diff --git a/public/images/docs/agent-playground/node-connection-handles.png b/public/images/docs/agent-playground/node-connection-handles.png new file mode 100644 index 00000000..a641ece1 Binary files /dev/null and b/public/images/docs/agent-playground/node-connection-handles.png differ diff --git a/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4 b/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4 new file mode 100644 index 00000000..12d59710 Binary files /dev/null and b/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4 differ diff --git a/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4 b/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4 new file mode 100644 index 00000000..c25ac221 Binary files /dev/null and b/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4 differ diff --git a/public/images/docs/simulation/agent-development-loop.svg b/public/images/docs/simulation/agent-development-loop.svg new file mode 100644 index 00000000..c48ebdf2 --- /dev/null +++ b/public/images/docs/simulation/agent-development-loop.svg @@ -0,0 +1,58 @@ + + + + + + + + + + + + + Replay + + + + + + + + + + + Build + develop your agent + + + + Simulate + rehearse voice or chat + conversations + + + + Score + evals judge every + conversation + + + + Fix + update the prompt + or config + + + + Deploy + ship to production + + + + Observe + trace live traffic + + The agent development loop + every change rehearses before it ships + diff --git a/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png b/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png new file mode 100644 index 00000000..ae493e7f Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-1.png b/public/images/docs/simulation/guides/connect-your-agent/connect-1.png new file mode 100644 index 00000000..8b4c6a59 Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-1.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-2.png b/public/images/docs/simulation/guides/connect-your-agent/connect-2.png new file mode 100644 index 00000000..b3d1c58c Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-2.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-3.png b/public/images/docs/simulation/guides/connect-your-agent/connect-3.png new file mode 100644 index 00000000..a674eb8b Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-3.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-4.png b/public/images/docs/simulation/guides/connect-your-agent/connect-4.png new file mode 100644 index 00000000..bfd43e0f Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-4.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-5.png b/public/images/docs/simulation/guides/connect-your-agent/connect-5.png new file mode 100644 index 00000000..48a9c1d0 Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-5.png differ diff --git a/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4 b/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4 new file mode 100644 index 00000000..bce36851 Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4 differ diff --git a/public/images/docs/simulation/guides/create-personas/chat-settings.png b/public/images/docs/simulation/guides/create-personas/chat-settings.png new file mode 100644 index 00000000..68497719 Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/chat-settings.png differ diff --git a/public/images/docs/simulation/guides/create-personas/choose-persona-type.png b/public/images/docs/simulation/guides/create-personas/choose-persona-type.png new file mode 100644 index 00000000..8bf979c9 Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/choose-persona-type.png differ diff --git a/public/images/docs/simulation/guides/create-personas/conversation-settings.png b/public/images/docs/simulation/guides/create-personas/conversation-settings.png new file mode 100644 index 00000000..2ed07e8a Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/conversation-settings.png differ diff --git a/public/images/docs/simulation/guides/create-personas/custom-properties.png b/public/images/docs/simulation/guides/create-personas/custom-properties.png new file mode 100644 index 00000000..73d19bef Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/custom-properties.png differ diff --git a/public/images/docs/simulation/guides/create-personas/custom-tab.png b/public/images/docs/simulation/guides/create-personas/custom-tab.png new file mode 100644 index 00000000..e89c55de Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/custom-tab.png differ diff --git a/public/images/docs/simulation/guides/create-personas/open-personas.png b/public/images/docs/simulation/guides/create-personas/open-personas.png new file mode 100644 index 00000000..8fd9012b Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/open-personas.png differ diff --git a/public/images/docs/simulation/guides/create-personas/persona-details.png b/public/images/docs/simulation/guides/create-personas/persona-details.png new file mode 100644 index 00000000..6c5b8e57 Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/persona-details.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png b/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png new file mode 100644 index 00000000..4686637c Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png b/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png new file mode 100644 index 00000000..81f1107f Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/basic-information.png b/public/images/docs/simulation/guides/create-scenarios/basic-information.png new file mode 100644 index 00000000..2d648db9 Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/basic-information.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png b/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png new file mode 100644 index 00000000..2354041e Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/import-datasets.png b/public/images/docs/simulation/guides/create-scenarios/import-datasets.png new file mode 100644 index 00000000..9582db74 Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/import-datasets.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png b/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png new file mode 100644 index 00000000..53087530 Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/upload-script.png b/public/images/docs/simulation/guides/create-scenarios/upload-script.png new file mode 100644 index 00000000..e265f2bf Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/upload-script.png differ diff --git a/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png b/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png new file mode 100644 index 00000000..240fdf6c Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png b/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png new file mode 100644 index 00000000..312154c5 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png b/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png new file mode 100644 index 00000000..b9d57fe6 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/configure-eval.png b/public/images/docs/simulation/guides/create-simulation/configure-eval.png new file mode 100644 index 00000000..896329d8 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/configure-eval.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/eval-library.png b/public/images/docs/simulation/guides/create-simulation/eval-library.png new file mode 100644 index 00000000..f1bd1927 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/eval-library.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png b/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png new file mode 100644 index 00000000..c74775a9 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png b/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png new file mode 100644 index 00000000..70dcc436 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/select-evaluations.png b/public/images/docs/simulation/guides/create-simulation/select-evaluations.png new file mode 100644 index 00000000..622abb5e Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/select-evaluations.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png b/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png new file mode 100644 index 00000000..bfc264cb Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png differ diff --git a/public/images/docs/simulation/guides/create-simulation/summary.mp4 b/public/images/docs/simulation/guides/create-simulation/summary.mp4 new file mode 100644 index 00000000..6e253667 Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/summary.mp4 differ diff --git a/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png b/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png new file mode 100644 index 00000000..3bc30c2c Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png b/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png new file mode 100644 index 00000000..a8e64825 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png b/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png new file mode 100644 index 00000000..3370e57d Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png new file mode 100644 index 00000000..243bab4e Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png new file mode 100644 index 00000000..5c0aac12 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png new file mode 100644 index 00000000..dfa4a20e Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png b/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png new file mode 100644 index 00000000..65150521 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png b/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png new file mode 100644 index 00000000..18d35745 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png b/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png new file mode 100644 index 00000000..20c7c441 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png b/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png new file mode 100644 index 00000000..9684a48d Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png b/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png new file mode 100644 index 00000000..83924100 Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png differ diff --git a/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png b/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png new file mode 100644 index 00000000..a0c253fe Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png differ diff --git a/public/images/docs/simulation/replay-loop.svg b/public/images/docs/simulation/replay-loop.svg new file mode 100644 index 00000000..8aed85a7 --- /dev/null +++ b/public/images/docs/simulation/replay-loop.svg @@ -0,0 +1,41 @@ + + + + + + + + + + + + + + + + Production + a real customer talks to + the agent you shipped + + + + Replay + the transcript becomes a scenario + and the config an agent definition + + + + Simulation + the simulator reruns it against + the recreated agent + + + + Compare and fix + measure against the original, + then ship the change + + The replay loop + the scenario stays as a regression test + diff --git a/public/images/docs/simulation/simulation-model-agent-highlighted.svg b/public/images/docs/simulation/simulation-model-agent-highlighted.svg new file mode 100644 index 00000000..7f8e2b65 --- /dev/null +++ b/public/images/docs/simulation/simulation-model-agent-highlighted.svg @@ -0,0 +1,53 @@ + + + + + + + + + + Your agent + + + + + conversation + + + + Simulated environment + having a simulated scenario + + FAGI + Simulator + + + + + + + List of scenarios + Refund request + Booking change + + + + + + + + List of personas + Frustrated customer + Polite regular + + + + + + + + Various personas + used to create + various scenarios + diff --git a/public/images/docs/simulation/simulation-model-personas-highlighted.svg b/public/images/docs/simulation/simulation-model-personas-highlighted.svg new file mode 100644 index 00000000..10a14de8 --- /dev/null +++ b/public/images/docs/simulation/simulation-model-personas-highlighted.svg @@ -0,0 +1,53 @@ + + + + + + + + + + Your agent + + + + + conversation + + + + Simulated environment + having a simulated scenario + + FAGI + Simulator + + + + + + + List of scenarios + Refund request + Booking change + + + + + + + + List of personas + Frustrated customer + Polite regular + + + + + + + + Various personas + used to create + various scenarios + diff --git a/public/images/docs/simulation/simulation-model-scenario-highlighted.svg b/public/images/docs/simulation/simulation-model-scenario-highlighted.svg new file mode 100644 index 00000000..cbbbca61 --- /dev/null +++ b/public/images/docs/simulation/simulation-model-scenario-highlighted.svg @@ -0,0 +1,53 @@ + + + + + + + + + + Your agent + + + + + conversation + + + + Simulated environment + having a simulated scenario + + FAGI + Simulator + + + + + + + List of scenarios + Refund request + Booking change + + + + + + + + List of personas + Frustrated customer + Polite regular + + + + + + + + Various personas + used to create + various scenarios + diff --git a/public/images/docs/simulation/simulation-model.svg b/public/images/docs/simulation/simulation-model.svg new file mode 100644 index 00000000..2936ceb4 --- /dev/null +++ b/public/images/docs/simulation/simulation-model.svg @@ -0,0 +1,53 @@ + + + + + + + + + + Your agent + + + + + conversation + + + + Simulated environment + having a simulated scenario + + FAGI + Simulator + + + + + + + List of scenarios + Refund request + Booking change + + + + + + + + List of personas + Frustrated customer + Polite regular + + + + + + + + Various personas + used to create + various scenarios + diff --git a/src/lib/navigation.ts b/src/lib/navigation.ts index af21b375..ddc6ebff 100644 --- a/src/lib/navigation.ts +++ b/src/lib/navigation.ts @@ -107,11 +107,32 @@ export const tabNavigation: NavTab[] = [ ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Create a Graph', href: '/docs/agent-playground/features/create-graph' }, - { title: 'Build a Workflow', href: '/docs/agent-playground/features/build-workflow' }, - { title: 'Run & Monitor', href: '/docs/agent-playground/features/run-and-monitor' }, + { title: 'Create an agent', href: '/docs/agent-playground/guides/create-agent' }, + { + title: 'Build a workflow', + items: [ + { title: 'Overview', href: '/docs/agent-playground/guides/build-workflow' }, + { title: 'Configure an LLM Prompt node', href: '/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node' }, + { title: 'Configure an Agent node', href: '/docs/agent-playground/guides/build-workflow/configure-an-agent-node' }, + { title: 'Set input variables', href: '/docs/agent-playground/guides/build-workflow/set-input-variables' }, + ] + }, + { title: 'Run an agent', href: '/docs/agent-playground/guides/run-an-agent' }, + { title: 'Manage versions', href: '/docs/agent-playground/guides/manage-versions' }, + ] + }, + { + title: 'Reference', + items: [ + { title: 'Limits & rules', href: '/docs/agent-playground/reference/limits-and-rules' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Agent Playground FAQ & fixes', href: '/docs/agent-playground/troubleshooting' }, ] }, ] @@ -124,28 +145,44 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ + { title: 'Understanding Annotation', href: '/docs/annotations/concepts/understanding-annotation' }, + { title: 'Labels', href: '/docs/annotations/concepts/labels' }, + { title: 'Queues & Items', href: '/docs/annotations/concepts/queues-and-items' }, { title: 'Scores', href: '/docs/annotations/concepts/scores' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Labels', href: '/docs/annotations/features/labels' }, - { title: 'Queues', href: '/docs/annotations/features/queues' }, - { title: 'Add Items to Queues', href: '/docs/annotations/features/add-items' }, - { title: 'Annotate Items', href: '/docs/annotations/features/annotate' }, - { title: 'Inline Annotations', href: '/docs/annotations/features/inline' }, - { title: 'Analytics & Agreement', href: '/docs/annotations/features/analytics' }, - { title: 'Export Annotations', href: '/docs/annotations/features/export' }, - { title: 'Automation Rules', href: '/docs/annotations/features/automation' }, + { title: 'Create a label', href: '/docs/annotations/guides/create-label' }, + { title: 'Create a queue', href: '/docs/annotations/guides/create-queue' }, + { + title: 'Explore a queue', + items: [ + { title: 'Overview', href: '/docs/annotations/guides/explore-queue' }, + { title: 'Add items', href: '/docs/annotations/guides/explore-queue/add-items' }, + { title: 'Track progress & agreement', href: '/docs/annotations/guides/explore-queue/progress-and-agreement' }, + { title: 'Automate item intake', href: '/docs/annotations/guides/explore-queue/automate-item-intake' }, + ] + }, + { title: 'Annotate items', href: '/docs/annotations/guides/annotate-items' }, + { title: 'Review submissions', href: '/docs/annotations/guides/review-submissions' }, + { title: 'Annotate without a queue', href: '/docs/annotations/guides/annotate-without-a-queue' }, + { title: 'Export annotations', href: '/docs/annotations/guides/export-annotations' }, ] }, { - title: 'SDK', + title: 'Reference', items: [ - { title: 'Python SDK', href: '/docs/annotations/sdk/python' }, - { title: 'JavaScript SDK', href: '/docs/annotations/sdk/javascript' }, - { title: 'Annotation Queue Using SDK', href: '/docs/annotations/sdk/annotation-queue-using-sdk' }, + { title: 'Label types & values', href: '/docs/annotations/reference/label-types-and-values' }, + { title: 'Queue settings & limits', href: '/docs/annotations/reference/queue-settings-and-limits' }, + { title: 'SDK & API', href: '/docs/annotations/reference/sdk-api' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Annotation FAQ & fixes', href: '/docs/annotations/troubleshooting' }, ] }, ] @@ -253,21 +290,33 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Understanding Datasets', href: '/docs/dataset/concept/understanding-dataset' }, - { title: 'Static Columns', href: '/docs/dataset/concept/static-column' }, - { title: 'Dynamic Columns', href: '/docs/dataset/concept/dynamic-column' }, - { title: 'Synthetic Data', href: '/docs/dataset/concept/synthetic-data' }, + { title: 'Understanding Datasets', href: '/docs/dataset/concepts/understanding-datasets' }, + { title: 'Static & Dynamic Columns', href: '/docs/dataset/concepts/static-and-dynamic-columns' }, + { title: 'Synthetic Data', href: '/docs/dataset/concepts/synthetic-data' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Create New Dataset', href: '/docs/dataset/features/create' }, - { title: 'Add Rows to Dataset', href: '/docs/dataset/features/add-rows' }, - { title: 'Add Columns to Dataset', href: '/docs/dataset/features/add-columns' }, - { title: 'Run Prompt in Dataset', href: '/docs/dataset/features/run-prompt' }, - { title: 'Experiments in Dataset', href: '/docs/dataset/features/experiments' }, - { title: 'Add Annotation', href: '/docs/dataset/features/annotate' }, + { title: 'Create a dataset', href: '/docs/dataset/guides/create-a-dataset' }, + { title: 'Add rows', href: '/docs/dataset/guides/add-rows' }, + { title: 'Add columns', href: '/docs/dataset/guides/add-columns' }, + { title: 'Run a prompt on every row', href: '/docs/dataset/guides/run-a-prompt-on-every-row' }, + { title: 'Run an experiment', href: '/docs/dataset/guides/run-an-experiment' }, + { title: 'Manage datasets', href: '/docs/dataset/guides/manage-datasets' }, + ] + }, + { + title: 'Reference', + items: [ + { title: 'Limits & Data Types', href: '/docs/dataset/reference/limits-and-data-types' }, + { title: 'Dynamic column methods', href: '/docs/dataset/reference/dynamic-column-methods' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Dataset FAQ & fixes', href: '/docs/dataset/troubleshooting' }, ] }, ] @@ -280,25 +329,34 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'How It Works', href: '/docs/error-feed/concepts/how-it-works' }, - { title: 'Error Taxonomy', href: '/docs/error-feed/concepts/taxonomy' }, - { title: 'Scoring', href: '/docs/error-feed/concepts/scoring' }, - { title: 'Severity and Status', href: '/docs/error-feed/concepts/severity-and-status' }, + { title: 'Understanding Error Feed', href: '/docs/error-feed/concepts/understanding-error-feed' }, + { title: 'Severity & Status', href: '/docs/error-feed/concepts/severity-and-status' }, + { title: 'Trace error analysis', href: '/docs/error-feed/concepts/trace-error-analysis' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'The Feed', href: '/docs/error-feed/features/the-feed' }, - { title: 'Issue Overview', href: '/docs/error-feed/features/issue-overview' }, - { title: 'Traces', href: '/docs/error-feed/features/traces' }, - { title: 'State Graph', href: '/docs/error-feed/features/state-graph' }, - { title: 'Trends', href: '/docs/error-feed/features/trends' }, - { title: 'Metadata Panel', href: '/docs/error-feed/features/metadata-panel' }, - { title: 'Triage Workflow', href: '/docs/error-feed/features/triage-workflow' }, - { title: 'Deep Analysis', href: '/docs/error-feed/features/deep-analysis' }, - { title: 'Linear Integration', href: '/docs/error-feed/features/linear-integration' }, - { title: 'Sampling', href: '/docs/error-feed/features/sampling' }, + { title: 'Turn on Error Feed', href: '/docs/error-feed/guides/turn-on-error-feed' }, + { title: 'Triage issues', href: '/docs/error-feed/guides/triage-issues' }, + { title: 'Investigate an issue', href: '/docs/error-feed/guides/investigate-an-issue' }, + { title: 'Run a root cause analysis', href: '/docs/error-feed/guides/run-root-cause-analysis' }, + { title: 'Create a Linear issue', href: '/docs/error-feed/guides/create-linear-issue' }, + ] + }, + { + title: 'Reference', + items: [ + { title: 'Issue fields & filters', href: '/docs/error-feed/reference/issue-fields' }, + { title: 'Error taxonomy', href: '/docs/error-feed/reference/error-taxonomy' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'No issues in the feed', href: '/docs/error-feed/troubleshooting/no-issues-in-the-feed' }, + { title: 'Issue counts look wrong', href: '/docs/error-feed/troubleshooting/issue-counts-look-wrong' }, + { title: "Analysis doesn't finish", href: '/docs/error-feed/troubleshooting/analysis-does-not-finish' }, ] }, ] @@ -501,11 +559,22 @@ export const tabNavigation: NavTab[] = [ items: [ { title: 'Overview', href: '/docs/falcon-ai' }, { - title: 'Features', + title: 'Concepts', + items: [ + { title: 'Understanding Falcon AI', href: '/docs/falcon-ai/concepts/understanding-falcon-ai' }, + { title: 'Skills', href: '/docs/falcon-ai/concepts/skills' }, + { title: 'MCP Connectors', href: '/docs/falcon-ai/concepts/mcp-connectors' }, + ] + }, + { + title: 'Guides', items: [ - { title: 'Using Falcon AI', href: '/docs/falcon-ai/features/chat' }, - { title: 'Skill Builder', href: '/docs/falcon-ai/features/skills' }, - { title: 'MCP Connectors', href: '/docs/falcon-ai/features/mcp-connectors' }, + { title: 'Chat with Falcon AI', href: '/docs/falcon-ai/guides/chat-with-falcon-ai' }, + { title: 'Manage conversations', href: '/docs/falcon-ai/guides/manage-conversations' }, + { title: 'Create a skill', href: '/docs/falcon-ai/guides/create-skill' }, + { title: 'Connect an MCP server', href: '/docs/falcon-ai/guides/connect-mcp-server' }, + { title: 'Choose connector tools', href: '/docs/falcon-ai/guides/choose-connector-tools' }, + { title: 'Use the MCP Server in your IDE', href: '/docs/falcon-ai/guides/use-the-mcp-server' }, ] }, ] @@ -518,14 +587,15 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Understanding Knowledge Base', href: '/docs/knowledge-base/concepts/concept' }, + { title: 'Understanding Knowledge Base', href: '/docs/knowledge-base/concepts/understanding-knowledge-base' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Create KB Using SDK', href: '/docs/knowledge-base/features/sdk' }, - { title: 'Create KB Using UI', href: '/docs/knowledge-base/features/ui' }, + { title: 'Create a knowledge base', href: '/docs/knowledge-base/guides/create-knowledge-base' }, + { title: 'Update a knowledge base', href: '/docs/knowledge-base/guides/update-knowledge-base' }, + { title: 'Manage with the SDK', href: '/docs/knowledge-base/guides/manage-with-the-sdk' }, ] }, ] @@ -589,20 +659,40 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Understanding Optimization', href: '/docs/optimization/concepts/concept' }, - { title: 'Bayesian Search', href: '/docs/optimization/optimizers/bayesian-search' }, - { title: 'Meta-Prompt', href: '/docs/optimization/optimizers/meta-prompt' }, - { title: 'ProTeGi', href: '/docs/optimization/optimizers/protegi' }, - { title: 'PromptWizard', href: '/docs/optimization/optimizers/promptwizard' }, - { title: 'GEPA', href: '/docs/optimization/optimizers/gepa' }, - { title: 'Random Search', href: '/docs/optimization/optimizers/random-search' }, + { title: 'Understanding optimization', href: '/docs/optimization/concepts/understanding-optimization' }, + { title: 'Choosing an optimizer', href: '/docs/optimization/concepts/choosing-an-optimizer' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Using Python SDK', href: '/docs/optimization/features/using-python-sdk' }, - { title: 'Using Platform', href: '/docs/optimization/features/using-platform' }, + { title: 'Run an optimization', href: '/docs/optimization/guides/run-an-optimization' }, + { title: 'Read optimization results', href: '/docs/optimization/guides/read-optimization-results' }, + { title: 'Optimize from the SDK', href: '/docs/optimization/guides/optimize-from-the-sdk' }, + ] + }, + { + title: 'Reference', + items: [ + { + title: 'Optimizers', + items: [ + { title: 'Overview', href: '/docs/optimization/reference/optimizers' }, + { title: 'Random Search', href: '/docs/optimization/reference/optimizers/random-search' }, + { title: 'Bayesian Search', href: '/docs/optimization/reference/optimizers/bayesian-search' }, + { title: 'ProTeGi', href: '/docs/optimization/reference/optimizers/protegi' }, + { title: 'Meta-Prompt', href: '/docs/optimization/reference/optimizers/meta-prompt' }, + { title: 'PromptWizard', href: '/docs/optimization/reference/optimizers/promptwizard' }, + { title: 'GEPA', href: '/docs/optimization/reference/optimizers/gepa' }, + ] + }, + { title: 'SDK & API', href: '/docs/optimization/reference/sdk-api' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Optimization FAQ & fixes', href: '/docs/optimization/troubleshooting' }, ] }, ] @@ -615,20 +705,33 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Prompt Engineering', href: '/docs/prompt/concepts/prompt-engineering' }, { title: 'Understanding Prompts', href: '/docs/prompt/concepts/understanding-prompts' }, - { title: 'Versions and Labels', href: '/docs/prompt/concepts/versions-and-labels' }, + { title: 'Versions & Labels', href: '/docs/prompt/concepts/versions-and-labels' }, + { title: 'Prompt Engineering', href: '/docs/prompt/concepts/prompt-engineering' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Create Prompt from Scratch', href: '/docs/prompt/features/create-from-scratch' }, - { title: 'Create from Existing Template', href: '/docs/prompt/features/create-from-template' }, - { title: 'Create with AI', href: '/docs/prompt/features/create-with-ai' }, - { title: 'Prompt Workbench Using SDK', href: '/docs/prompt/features/sdk' }, - { title: 'Linked Traces', href: '/docs/prompt/features/linked-traces' }, - { title: 'Manage Folders', href: '/docs/prompt/features/folders' }, + { title: 'Create a prompt', href: '/docs/prompt/guides/create-a-prompt' }, + { title: 'Run a prompt', href: '/docs/prompt/guides/run-a-prompt' }, + { title: 'Commit & compare versions', href: '/docs/prompt/guides/commit-and-compare-versions' }, + { title: 'Evaluate prompt outputs', href: '/docs/prompt/guides/evaluate-prompt-outputs' }, + { title: 'Track prompt performance', href: '/docs/prompt/guides/track-prompt-performance' }, + { title: 'Organize prompts in folders', href: '/docs/prompt/guides/organize-prompts-in-folders' }, + ] + }, + { + title: 'Reference', + items: [ + { title: 'Model configuration', href: '/docs/prompt/reference/model-configuration' }, + { title: 'SDK & API', href: '/docs/prompt/reference/sdk-api' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Prompt FAQ & fixes', href: '/docs/prompt/troubleshooting' }, ] }, ] @@ -641,35 +744,30 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Use Cases', href: '/docs/protect/concepts/concept' }, + { title: 'Understanding Protect', href: '/docs/protect/concepts/understanding-protect' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Run Protect via SDK', href: '/docs/protect/features/run-protect' }, + { title: 'Turn on a guardrail', href: '/docs/protect/guides/turn-on-a-guardrail' }, + { title: 'Test a guardrail', href: '/docs/protect/guides/test-a-guardrail' }, + { title: 'Review guardrail activity', href: '/docs/protect/guides/review-guardrail-activity' }, + { title: 'Run Protect from the SDK', href: '/docs/protect/guides/run-protect-from-the-sdk' }, ] }, - ] - }, - { - group: 'Prototype', - icon: 'flask', - items: [ - { title: 'Overview', href: '/docs/prototype' }, { - title: 'Concepts', + title: 'Reference', items: [ - { title: 'Understanding Prototype', href: '/docs/prototype/concepts/understanding-prototype' }, - { title: 'Versions and Runs', href: '/docs/prototype/concepts/versions-and-runs' }, + { title: 'Guardrail checks', href: '/docs/protect/reference/guardrail-checks' }, ] }, { - title: 'Features', + title: 'Troubleshooting', items: [ - { title: 'Set Up Prototype', href: '/docs/prototype/features/set-up-prototype' }, - { title: 'Evals', href: '/docs/prototype/features/evals' }, - { title: 'Choose Winner', href: '/docs/prototype/features/choose-winner' }, + { title: 'Guardrail changes not taking effect', href: '/docs/protect/troubleshooting/guardrail-changes-not-taking-effect' }, + { title: 'Guardrail fires on the wrong requests', href: '/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests' }, + { title: 'Protect SDK rejects an input', href: '/docs/protect/troubleshooting/protect-sdk-rejects-an-input' }, ] }, ] @@ -706,28 +804,84 @@ export const tabNavigation: NavTab[] = [ { title: 'Concepts', items: [ - { title: 'Agent Definition', href: '/docs/simulation/concepts/agent-definition' }, + { title: 'Understanding Simulation', href: '/docs/simulation/concepts/understanding-simulation' }, + { title: 'Agent definitions & versions', href: '/docs/simulation/concepts/agent-definitions' }, { title: 'Scenarios', href: '/docs/simulation/concepts/scenarios' }, { title: 'Personas', href: '/docs/simulation/concepts/personas' }, - { title: 'Global Nodes', href: '/docs/simulation/concepts/global-nodes' }, + { title: 'Runs & results', href: '/docs/simulation/concepts/runs-and-results' }, + { title: 'Replay', href: '/docs/simulation/concepts/replay' }, + { title: 'Optimization', href: '/docs/simulation/concepts/optimization' }, ] }, { - title: 'Features', + title: 'Guides', items: [ - { title: 'Run Voice Simulation', href: '/docs/simulation/features/run-simulation' }, - { title: 'Chat Simulation Using SDK', href: '/docs/simulation/features/simulation-using-sdk' }, + { title: 'Connect your agent', href: '/docs/simulation/guides/connect-your-agent' }, + { title: 'Create scenarios', href: '/docs/simulation/guides/create-scenarios' }, + { title: 'Create personas', href: '/docs/simulation/guides/create-personas' }, { - title: 'Replay', + title: 'Explore scenarios', items: [ - { title: 'Chat Replay', href: '/docs/simulation/features/observe-to-simulate' }, - { title: 'Voice Replay', href: '/docs/simulation/features/voice-replay' }, + { title: 'Overview', href: '/docs/simulation/guides/explore-scenarios' }, + { title: 'Explore scenario graph', href: '/docs/simulation/guides/explore-scenarios/scenario-graph' }, + { title: 'Add rows', href: '/docs/simulation/guides/explore-scenarios/add-rows' }, + { title: 'Add columns', href: '/docs/simulation/guides/explore-scenarios/add-columns' }, ] }, - { title: 'Prompt Simulation', href: '/docs/simulation/features/prompt-simulation' }, - { title: 'Evaluate Tool Calling', href: '/docs/simulation/features/evaluate-tool-calling' }, - { title: 'View Results', href: '/docs/simulation/features/view-results' }, - { title: 'Fix My Agent', href: '/docs/simulation/features/fix-my-agent' }, + { + title: 'Running simulations', + items: [ + { title: 'Create a simulation', href: '/docs/simulation/guides/create-simulation' }, + { title: 'Run a voice simulation', href: '/docs/simulation/guides/run-voice-simulation' }, + { title: 'Run a chat simulation', href: '/docs/simulation/guides/run-chat-simulation' }, + { title: 'Simulate a prompt', href: '/docs/simulation/guides/prompt-simulation' }, + ] + }, + { + title: 'Evaluations in simulation', + items: [ + { title: 'Edit evals in a simulation', href: '/docs/simulation/guides/edit-evals' }, + { title: 'Evaluate tool calls', href: '/docs/simulation/guides/evaluate-tool-calls' }, + ] + }, + { + title: 'Replay simulations', + items: [ + { title: 'Replay chat sessions', href: '/docs/simulation/guides/replay-chat' }, + { title: 'Replay voice calls', href: '/docs/simulation/guides/replay-voice' }, + ] + }, + { + title: 'Explore results', + items: [ + { title: 'Overview', href: '/docs/simulation/guides/explore-results' }, + { title: 'Calls & transcripts', href: '/docs/simulation/guides/explore-results/calls-and-transcripts' }, + { title: 'Analytics & metrics', href: '/docs/simulation/guides/explore-results/analytics' }, + ] + }, + { title: 'Fix My Agent', href: '/docs/simulation/guides/fix-my-agent' }, + { + title: 'Optimize using simulate', + items: [ + { title: 'Running optimizations', href: '/docs/simulation/guides/running-optimizations' }, + { title: 'Optimization runs', href: '/docs/simulation/guides/optimization-runs' }, + ] + }, + ] + }, + { + title: 'References', + items: [ + { title: 'Built-in personas', href: '/docs/simulation/reference/built-in-personas' }, + { title: 'Voice providers', href: '/docs/simulation/reference/voice-providers' }, + { title: 'Call metrics', href: '/docs/simulation/reference/call-metrics' }, + { title: 'SDK & API', href: '/docs/simulation/reference/sdk-api' }, + ] + }, + { + title: 'Troubleshooting', + items: [ + { title: 'Simulation FAQ & fixes', href: '/docs/simulation/troubleshooting' }, ] }, ] @@ -881,7 +1035,6 @@ export const tabNavigation: NavTab[] = [ title: 'Prompt', items: [ { title: 'Prompt Versioning: Create, Label, and Serve Prompt Versions', href: '/docs/cookbook/quickstart/prompt-versioning' }, - { title: 'Prototype and Iterate on LLM Applications', href: '/docs/cookbook/quickstart/prototype-llm-app' }, ] }, { diff --git a/src/lib/redirects.ts b/src/lib/redirects.ts index 8b6af070..006594e8 100644 --- a/src/lib/redirects.ts +++ b/src/lib/redirects.ts @@ -1,6 +1,32 @@ // Auto-generated redirect map: old Mintlify URLs → new docs URLs // 275 redirects from futureagi.mintlify.app export const redirectMap: Record = { + // Prototype was deprecated and removed from the dashboard; its pages now point at Evaluation + '/docs/prototype': '/docs/evaluation', + '/docs/prototype/concepts/understanding-prototype': '/docs/evaluation', + '/docs/prototype/concepts/versions-and-runs': '/docs/evaluation', + '/docs/prototype/features/set-up-prototype': '/docs/evaluation/guides/running-evaluations', + '/docs/prototype/features/evals': '/docs/evaluation/builtin', + '/docs/prototype/features/choose-winner': '/docs/evaluation', + '/docs/cookbook/quickstart/prototype-llm-app': '/docs/cookbook/quickstart/experimentation-compare-prompts', + // Error Feed revamp: Concepts/Features replaced by Concepts/Guides/Reference/Troubleshooting. + // NOTE /docs/error-feed/features/sampling is linked from inside the product + // (frontend ConfigureProject.jsx), so that redirect must not be removed. + '/docs/error-feed/concepts/how-it-works': '/docs/error-feed/concepts/understanding-error-feed', + '/docs/error-feed/concepts/taxonomy': '/docs/error-feed/reference/error-taxonomy', + '/docs/error-feed/concepts/scoring': '/docs/error-feed/concepts/trace-error-analysis', + '/docs/error-feed/features/sampling': '/docs/error-feed/guides/turn-on-error-feed', + '/docs/error-feed/features/the-feed': '/docs/error-feed/guides/triage-issues', + '/docs/error-feed/features/triage-workflow': '/docs/error-feed/guides/triage-issues', + '/docs/error-feed/features/issue-overview': '/docs/error-feed/guides/investigate-an-issue', + '/docs/error-feed/features/traces': '/docs/error-feed/guides/investigate-an-issue', + '/docs/error-feed/features/state-graph': '/docs/error-feed/guides/investigate-an-issue', + '/docs/error-feed/features/trends': '/docs/error-feed/guides/investigate-an-issue', + '/docs/error-feed/features/metadata-panel': '/docs/error-feed/guides/investigate-an-issue', + '/docs/error-feed/features/deep-analysis': '/docs/error-feed/guides/run-root-cause-analysis', + '/docs/error-feed/features/linear-integration': '/docs/error-feed/guides/create-linear-issue', + '/docs/error-feed/taxonomy': '/docs/error-feed/reference/error-taxonomy', + // Manual-instrumentation pages moved from Observe features into the traceAI SDK section '/docs/observe/features/manual-tracing/set-up-tracing': '/docs/sdk/tracing/set-up-tracing', '/docs/observe/features/manual-tracing/instrument-with-traceai-helpers': '/docs/sdk/tracing/instrument-with-traceai-helpers', @@ -18,10 +44,10 @@ export const redirectMap: Record = { '/docs/observe/features/manual-tracing/langfuse-integration': '/docs/sdk/tracing/langfuse-integration', '/docs/cookbook/observability': '/docs/cookbook/observe-langgraph-agent-and-obtain-insights', '/docs/cookbook/improve-langgraph-agent-with-observability': '/docs/cookbook/observe-langgraph-agent-and-obtain-insights', - '/docs/observe/features/annotation-queue-using-sdk': '/docs/annotations/sdk/annotation-queue-using-sdk', + '/docs/observe/features/annotation-queue-using-sdk': '/docs/annotations/reference/sdk-api', // SDK pages restructured: sdk/tracing.mdx flat page → sdk/tracing/ folder (index still serves /docs/sdk/tracing); annotation-queues moved under Annotations '/docs/sdk/tracing': '/docs/sdk/tracing/set-up-tracing', - '/docs/sdk/annotation-queues': '/docs/annotations/sdk/annotation-queue-using-sdk', + '/docs/sdk/annotation-queues': '/docs/annotations/reference/sdk-api', '/docs/observe/voice/set-up': '/docs/observe/features/voice', '/docs/quickstart/installation': '/docs/installation', '/docs/observability': '/docs/tracing/auto', @@ -37,14 +63,16 @@ export const redirectMap: Record = { '/docs/evaluation/features/futureagi-models': '/docs/evaluation/concepts/evaluator-models', '/docs/evaluation/concepts/eval-results': '/docs/evaluation/reference/output-types', '/docs/optimization/optimizers/overview': '/docs/optimization', - '/docs/dataset/add-annotations': '/docs/dataset/features/annotate', - '/docs/knowledge-base/concept': '/docs/knowledge-base/concepts/concept', + '/docs/dataset/add-annotations': '/docs/annotations/guides/explore-queue/add-items', + '/docs/knowledge-base/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base', '/docs/prompt-workbench': '/docs/prompt', - '/docs/prompt-workbench/sdk': '/docs/prompt/features/sdk', + '/docs/prompt-workbench/sdk': '/docs/prompt/reference/sdk-api', '/docs/tracing/manual/log-prompt-templates': '/docs/sdk/tracing/log-prompt-templates', '/docs/tracing/manual/in-line-evals': '/docs/sdk/tracing/in-line-evals', '/docs/simulation/set-up/scenarios': '/docs/simulation/concepts/scenarios', - '/docs/simulation/set-up/agent-definition': '/docs/simulation/concepts/agent-definition', + '/docs/simulation/set-up/agent-definition': '/docs/simulation/concepts/agent-definitions', + '/docs/simulation/concepts/agent-definition': '/docs/simulation/concepts/agent-definitions', + '/docs/simulation/concepts/global-nodes': '/docs/simulation/concepts/scenarios', '/docs/tracing/concepts/components': '/docs/tracing/concepts', '/docs/tracing/manual/add-attributes-metadata-tags': '/docs/sdk/tracing/add-attributes-metadata-tags', '/docs/tracing/manual/add-events-exceptions-status': '/docs/sdk/tracing/add-events-exceptions-status', @@ -186,9 +214,9 @@ export const redirectMap: Record = { '/future-agi/get-started/evaluation/future-agi-models': '/docs/evaluation/concepts/evaluator-models', '/future-agi/get-started/evaluation/running-your-first-eval': '/docs/evaluation/guides/running-evaluations', '/future-agi/get-started/evaluation/use-custom-models': '/docs/evaluation/guides/custom-models', - '/future-agi/get-started/knowledge-base/concept': '/docs/knowledge-base/concepts/concept', - '/future-agi/get-started/knowledge-base/how-to/create-kb-using-sdk': '/docs/knowledge-base/features/sdk', - '/future-agi/get-started/knowledge-base/how-to/create-kb-using-ui': '/docs/knowledge-base/features/ui', + '/future-agi/get-started/knowledge-base/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base', + '/future-agi/get-started/knowledge-base/how-to/create-kb-using-sdk': '/docs/knowledge-base/guides/manage-with-the-sdk', + '/future-agi/get-started/knowledge-base/how-to/create-kb-using-ui': '/docs/knowledge-base/guides/create-knowledge-base', '/future-agi/get-started/knowledge-base/overview': '/docs/knowledge-base', '/future-agi/get-started/observability/manual-tracing/add-attributes-metadata-tags': '/docs/sdk/tracing/add-attributes-metadata-tags', '/future-agi/get-started/observability/manual-tracing/add-events-exceptions-status': '/docs/sdk/tracing/add-events-exceptions-status', @@ -205,23 +233,23 @@ export const redirectMap: Record = { '/future-agi/get-started/observability/manual-tracing/set-session-user-id': '/docs/sdk/tracing/set-session-user-id', '/future-agi/get-started/observability/manual-tracing/set-up-tracing': '/docs/sdk/tracing/set-up-tracing', '/future-agi/get-started/optimization/dataset-optimization': '/docs/cookbook/quickstart/dataset-optimization', - '/future-agi/get-started/optimization/how-to/using-python-sdk': '/docs/optimization/features/using-python-sdk', - '/future-agi/get-started/optimization/optimizers/bayesian-search': '/docs/optimization/optimizers/bayesian-search', - '/future-agi/get-started/optimization/optimizers/gepa': '/docs/optimization/optimizers/gepa', - '/future-agi/get-started/optimization/optimizers/meta-prompt': '/docs/optimization/optimizers/meta-prompt', + '/future-agi/get-started/optimization/how-to/using-python-sdk': '/docs/optimization/guides/optimize-from-the-sdk', + '/future-agi/get-started/optimization/optimizers/bayesian-search': '/docs/optimization/reference/optimizers/bayesian-search', + '/future-agi/get-started/optimization/optimizers/gepa': '/docs/optimization/reference/optimizers/gepa', + '/future-agi/get-started/optimization/optimizers/meta-prompt': '/docs/optimization/reference/optimizers/meta-prompt', '/future-agi/get-started/optimization/optimizers/overview': '/docs/optimization', - '/future-agi/get-started/optimization/optimizers/promptwizard': '/docs/optimization/optimizers/promptwizard', - '/future-agi/get-started/optimization/optimizers/protegi': '/docs/optimization/optimizers/protegi', - '/future-agi/get-started/optimization/optimizers/random-search': '/docs/optimization/optimizers/random-search', + '/future-agi/get-started/optimization/optimizers/promptwizard': '/docs/optimization/reference/optimizers/promptwizard', + '/future-agi/get-started/optimization/optimizers/protegi': '/docs/optimization/reference/optimizers/protegi', + '/future-agi/get-started/optimization/optimizers/random-search': '/docs/optimization/reference/optimizers/random-search', '/future-agi/get-started/optimization/overview': '/docs/optimization', '/future-agi/get-started/optimization/quickstart': '/docs/optimization', - '/future-agi/get-started/protect/concept': '/docs/protect/concepts/concept', - '/future-agi/get-started/protect/how-to': '/docs/protect/features/run-protect', + '/future-agi/get-started/protect/concept': '/docs/protect/concepts/understanding-protect', + '/future-agi/get-started/protect/how-to': '/docs/protect/guides/run-protect-from-the-sdk', '/future-agi/get-started/protect/overview': '/docs/protect', - '/future-agi/get-started/prototype/evals': '/docs/prototype/features/evals', - '/future-agi/get-started/prototype/overview': '/docs/prototype', + '/future-agi/get-started/prototype/evals': '/docs/evaluation/builtin', + '/future-agi/get-started/prototype/overview': '/docs/evaluation', '/future-agi/get-started/prototype/quickstart': '/docs/observe/features/quickstart', - '/future-agi/get-started/prototype/winner': '/docs/prototype/features/choose-winner', + '/future-agi/get-started/prototype/winner': '/docs/evaluation', '/future-agi/products/observability/auto-instrumentation/overview': '/docs/tracing/auto', '/future-agi/products/observability/concept/core-components': '/docs/tracing/concepts', '/future-agi/products/observability/concept/otel': '/docs/tracing/concepts/otel', @@ -271,58 +299,58 @@ export const redirectMap: Record = { '/integrations/vertexai': '/docs/integrations/traceai/vertexai', '/product/agent-compass/overview': '/docs/error-feed', '/product/agent-compass/quickstart': '/docs/error-feed', - '/product/agent-compass/taxonomy': '/docs/error-feed/concepts/taxonomy', + '/product/agent-compass/taxonomy': '/docs/error-feed/reference/error-taxonomy', '/docs/cookbook/quickstart/agent-compass-debug': '/docs/error-feed', - '/product/annotations/concepts/labels': '/docs/annotations/features/labels', - '/product/annotations/concepts/queues': '/docs/annotations/features/queues', + '/product/annotations/concepts/labels': '/docs/annotations/reference/label-types-and-values', + '/product/annotations/concepts/queues': '/docs/annotations/reference/queue-settings-and-limits', '/product/annotations/concepts/scores': '/docs/annotations/concepts/scores', - '/product/annotations/features/add-items': '/docs/annotations/features/add-items', - '/product/annotations/features/analytics': '/docs/annotations/features/analytics', - '/product/annotations/features/annotate': '/docs/annotations/features/annotate', - '/product/annotations/features/automation': '/docs/annotations/features/automation', - '/product/annotations/features/export': '/docs/annotations/features/export', - '/product/annotations/features/inline': '/docs/annotations/features/inline', - '/product/annotations/features/labels': '/docs/annotations/features/labels', - '/product/annotations/features/queues': '/docs/annotations/features/queues', + '/product/annotations/features/add-items': '/docs/annotations/guides/explore-queue/add-items', + '/product/annotations/features/analytics': '/docs/annotations/guides/explore-queue/progress-and-agreement', + '/product/annotations/features/annotate': '/docs/annotations/guides/annotate-items', + '/product/annotations/features/automation': '/docs/annotations/guides/explore-queue/automate-item-intake', + '/product/annotations/features/export': '/docs/annotations/guides/export-annotations', + '/product/annotations/features/inline': '/docs/annotations/guides/annotate-without-a-queue', + '/product/annotations/features/labels': '/docs/annotations/reference/label-types-and-values', + '/product/annotations/features/queues': '/docs/annotations/reference/queue-settings-and-limits', '/product/annotations/overview': '/docs/annotations', - '/product/annotations/quickstart': '/docs/annotations/quickstart', - '/product/annotations/sdk/javascript': '/docs/annotations/sdk/javascript', - '/product/annotations/sdk/python': '/docs/annotations/sdk/python', - '/product/dataset/how-to/add-rows-to-dataset': '/docs/dataset/features/add-rows', - '/product/dataset/how-to/annotate-dataset': '/docs/dataset/features/annotate', - '/product/dataset/how-to/create-dynamic-column/by-executing-code': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/by-extracting-entities': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/by-extracting-json': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/using-api-calls': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/using-classification': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/using-conditional-node': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/using-run-prompt': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-dynamic-column/using-vector-db': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/create-new-dataset': '/docs/dataset/features/create', - '/product/dataset/how-to/create-static-column': '/docs/dataset/features/add-columns', - '/product/dataset/how-to/experiments-in-dataset': '/docs/dataset/features/experiments', - '/product/dataset/how-to/run-prompt-in-dataset': '/docs/dataset/features/run-prompt', + '/product/annotations/quickstart': '/docs/annotations/guides/create-queue', + '/product/annotations/sdk/javascript': '/docs/annotations/reference/sdk-api', + '/product/annotations/sdk/python': '/docs/annotations/reference/sdk-api', + '/product/dataset/how-to/add-rows-to-dataset': '/docs/dataset/guides/add-rows', + '/product/dataset/how-to/annotate-dataset': '/docs/annotations/guides/explore-queue/add-items', + '/product/dataset/how-to/create-dynamic-column/by-executing-code': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/by-extracting-entities': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/by-extracting-json': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/using-api-calls': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/using-classification': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/using-conditional-node': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/using-run-prompt': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-dynamic-column/using-vector-db': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/create-new-dataset': '/docs/dataset/guides/create-a-dataset', + '/product/dataset/how-to/create-static-column': '/docs/dataset/reference/dynamic-column-methods', + '/product/dataset/how-to/experiments-in-dataset': '/docs/dataset/guides/run-an-experiment', + '/product/dataset/how-to/run-prompt-in-dataset': '/docs/dataset/guides/run-a-prompt-on-every-row', '/product/dataset/overview': '/docs/dataset', - '/product/prompt/how-to/create-prompt-from-existing-template': '/docs/prompt/features/create-from-template', - '/product/prompt/how-to/create-prompt-from-scratch': '/docs/prompt/features/create-from-scratch', - '/product/prompt/how-to/linked-traces': '/docs/prompt/features/linked-traces', - '/product/prompt/how-to/manage-folders': '/docs/prompt/features/folders', - '/product/prompt/how-to/prompt-workbench-using-sdk': '/docs/prompt/features/sdk', + '/product/prompt/how-to/create-prompt-from-existing-template': '/docs/prompt/guides/create-a-prompt', + '/product/prompt/how-to/create-prompt-from-scratch': '/docs/prompt/guides/create-a-prompt', + '/product/prompt/how-to/linked-traces': '/docs/prompt/guides/track-prompt-performance', + '/product/prompt/how-to/manage-folders': '/docs/prompt/guides/organize-prompts-in-folders', + '/product/prompt/how-to/prompt-workbench-using-sdk': '/docs/prompt/reference/sdk-api', '/product/prompt/overview': '/docs/prompt', - '/product/simulation/agent-definition': '/docs/simulation/concepts/agent-definition', - '/product/simulation/how-to/chat-simulation-using-sdk': '/docs/simulation/features/simulation-using-sdk', - '/product/simulation/how-to/evaluate-tool-calling': '/docs/simulation/features/evaluate-tool-calling', - '/product/simulation/how-to/fix-my-agent': '/docs/simulation/features/fix-my-agent', - '/product/simulation/how-to/observe-to-simulate': '/docs/simulation/features/observe-to-simulate', - '/product/simulation/how-to/prompt-simulation': '/docs/simulation/features/prompt-simulation', - '/product/simulation/how-to/voice-observability': '/docs/simulation/features/voice-replay', + '/product/simulation/agent-definition': '/docs/simulation/concepts/agent-definitions', + '/product/simulation/how-to/chat-simulation-using-sdk': '/docs/simulation/guides/run-chat-simulation', + '/product/simulation/how-to/evaluate-tool-calling': '/docs/simulation/guides/evaluate-tool-calls', + '/product/simulation/how-to/fix-my-agent': '/docs/simulation/guides/fix-my-agent', + '/product/simulation/how-to/observe-to-simulate': '/docs/simulation/guides/replay-chat', + '/product/simulation/how-to/prompt-simulation': '/docs/simulation/guides/prompt-simulation', + '/product/simulation/how-to/voice-observability': '/docs/simulation/guides/replay-voice', '/product/simulation/overview': '/docs/simulation', '/product/simulation/personas': '/docs/simulation/concepts/personas', - '/product/simulation/run-tests': '/docs/simulation/features/run-simulation', + '/product/simulation/run-tests': '/docs/simulation/guides/run-voice-simulation', '/product/simulation/scenarios': '/docs/simulation/concepts/scenarios', '/quickstart/generate-synthetic-data': '/docs/quickstart/generate-synthetic-data', '/quickstart/running-evals-in-simulation': '/docs/quickstart/running-evals-in-simulation', - '/quickstart/setup-mcp-server': '/docs/quickstart/setup-mcp-server', + '/quickstart/setup-mcp-server': '/docs/falcon-ai/guides/use-the-mcp-server', '/quickstart/setup-observability': '/docs/quickstart/setup-observability', '/release-notes': '/docs/release-notes', '/sdk-reference/datasets': '/docs/sdk/datasets', @@ -334,4 +362,65 @@ export const redirectMap: Record = { '/sdk-reference/testcase': '/docs/sdk/testcase', '/sdk-reference/tracing': '/docs/sdk/tracing', '/docs/self-hosting/environment': '/docs/self-hosting/configuration/environment', + '/docs/simulation/features/run-simulation': '/docs/simulation/guides/run-voice-simulation', + '/docs/simulation/features/simulation-using-sdk': '/docs/simulation/guides/run-chat-simulation', + '/docs/simulation/features/observe-to-simulate': '/docs/simulation/guides/replay-chat', + '/docs/simulation/features/voice-replay': '/docs/simulation/guides/replay-voice', + '/docs/simulation/features/prompt-simulation': '/docs/simulation/guides/prompt-simulation', + '/docs/simulation/features/evaluate-tool-calling': '/docs/simulation/guides/evaluate-tool-calls', + '/docs/simulation/features/view-results': '/docs/simulation/guides/explore-results', + '/docs/simulation/features/fix-my-agent': '/docs/simulation/guides/fix-my-agent', + + // Product docs revamp: Features-shaped sections replaced by Concepts/Guides/Reference/Troubleshooting. + '/docs/agent-playground/features/build-workflow': '/docs/agent-playground/guides/build-workflow', + '/docs/agent-playground/features/create-graph': '/docs/agent-playground/guides/create-agent', + '/docs/agent-playground/features/run-and-monitor': '/docs/agent-playground/guides/run-an-agent', + '/docs/annotations/features/add-items': '/docs/annotations/guides/explore-queue/add-items', + '/docs/annotations/features/analytics': '/docs/annotations/guides/explore-queue/progress-and-agreement', + '/docs/annotations/features/annotate': '/docs/annotations/guides/annotate-items', + '/docs/annotations/features/automation': '/docs/annotations/guides/explore-queue/automate-item-intake', + '/docs/annotations/features/export': '/docs/annotations/guides/export-annotations', + '/docs/annotations/features/inline': '/docs/annotations/guides/annotate-without-a-queue', + '/docs/annotations/features/labels': '/docs/annotations/reference/label-types-and-values', + '/docs/annotations/features/queues': '/docs/annotations/reference/queue-settings-and-limits', + '/docs/annotations/quickstart': '/docs/annotations/guides/create-queue', + '/docs/annotations/sdk/annotation-queue-using-sdk': '/docs/annotations/reference/sdk-api', + '/docs/annotations/sdk/javascript': '/docs/annotations/reference/sdk-api', + '/docs/annotations/sdk/python': '/docs/annotations/reference/sdk-api', + '/docs/dataset/concept/dynamic-column': '/docs/dataset/concepts/static-and-dynamic-columns', + '/docs/dataset/concept/static-column': '/docs/dataset/concepts/static-and-dynamic-columns', + '/docs/dataset/concept/synthetic-data': '/docs/dataset/concepts/synthetic-data', + '/docs/dataset/concept/understanding-dataset': '/docs/dataset/concepts/understanding-datasets', + '/docs/dataset/features/add-columns': '/docs/dataset/reference/dynamic-column-methods', + '/docs/dataset/features/add-rows': '/docs/dataset/guides/add-rows', + '/docs/dataset/features/annotate': '/docs/annotations/guides/explore-queue/add-items', + '/docs/dataset/guides/annotate-rows': '/docs/annotations/guides/explore-queue/add-items', + '/docs/dataset/features/create': '/docs/dataset/guides/create-a-dataset', + '/docs/dataset/features/experiments': '/docs/dataset/guides/run-an-experiment', + '/docs/dataset/features/run-prompt': '/docs/dataset/guides/run-a-prompt-on-every-row', + '/docs/falcon-ai/features/chat': '/docs/falcon-ai/guides/chat-with-falcon-ai', + '/docs/falcon-ai/features/mcp-connectors': '/docs/falcon-ai/concepts/mcp-connectors', + '/docs/falcon-ai/features/skills': '/docs/falcon-ai/concepts/skills', + '/docs/knowledge-base/concepts/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base', + '/docs/knowledge-base/features/sdk': '/docs/knowledge-base/guides/manage-with-the-sdk', + '/docs/knowledge-base/features/ui': '/docs/knowledge-base/guides/create-knowledge-base', + '/docs/optimization/concepts/concept': '/docs/optimization/concepts/understanding-optimization', + '/docs/optimization/features/using-platform': '/docs/optimization/guides/run-an-optimization', + '/docs/optimization/features/using-python-sdk': '/docs/optimization/guides/optimize-from-the-sdk', + '/docs/optimization/optimizers/bayesian-search': '/docs/optimization/reference/optimizers/bayesian-search', + '/docs/optimization/optimizers/gepa': '/docs/optimization/reference/optimizers/gepa', + '/docs/optimization/optimizers/meta-prompt': '/docs/optimization/reference/optimizers/meta-prompt', + '/docs/optimization/optimizers/promptwizard': '/docs/optimization/reference/optimizers/promptwizard', + '/docs/optimization/optimizers/protegi': '/docs/optimization/reference/optimizers/protegi', + '/docs/optimization/optimizers/random-search': '/docs/optimization/reference/optimizers/random-search', + '/docs/prompt/features/create-from-scratch': '/docs/prompt/guides/create-a-prompt', + '/docs/prompt/features/create-from-template': '/docs/prompt/guides/create-a-prompt', + '/docs/prompt/features/create-with-ai': '/docs/prompt/guides/create-a-prompt', + '/docs/prompt/features/folders': '/docs/prompt/guides/organize-prompts-in-folders', + '/docs/prompt/features/linked-traces': '/docs/prompt/guides/track-prompt-performance', + '/docs/prompt/features/sdk': '/docs/prompt/reference/sdk-api', + '/docs/protect/concepts/concept': '/docs/protect/concepts/understanding-protect', + '/docs/protect/concepts/guardrail-pipeline': '/docs/protect/concepts/understanding-protect', + '/docs/protect/features/run-protect': '/docs/protect/guides/run-protect-from-the-sdk', + '/docs/quickstart/setup-mcp-server': '/docs/falcon-ai/guides/use-the-mcp-server', }; diff --git a/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx b/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx index 5bf045fd..4834c5bc 100644 --- a/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx +++ b/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx @@ -1,100 +1,69 @@ --- -title: "Agent Playground: Core Concepts" -description: "Learn the core building blocks of Agent Playground: graphs, LLM nodes, subgraph nodes, ports, edges, and node templates." +title: "Understanding Agent Playground" +description: "How nodes, connections, and nested agents combine into what an agent actually runs" --- -## About +## An agent is a set of connected nodes -Agent Playground is built around a small set of core building blocks. Understanding these helps you design and debug workflows effectively. This page explains how graphs, nodes, ports, edges, and templates fit together. +An **agent** is a set of nodes wired together. Each node takes named inputs and produces named outputs, and every input and output is typed: it carries a display name for the canvas and a JSON Schema that defines the shape of the data passing through it. A **connection** joins one node's output to another node's input, and that's how a value produced by one step reaches the next. ---- - -## Graphs - -A graph is the top-level container for your AI workflow. It is a series of connected steps where data flows from inputs through each node to outputs. +Take an agent called `support-triage`. It starts with two nodes: `classify` and `draft-reply`. `classify` takes a `message` input and produces a `response` output: it runs a linked [prompt](/docs/prompt/concepts/understanding-prompts), and that's what turns the input into the output. `classify`'s `response` output schema tracks whatever prompt is linked to it: change the linked prompt's response format, and `response`'s shape updates to match. A connection carries that `classify.response` value straight into `draft-reply`'s own `category` input. -Each graph has: -- **Name and description** for identification -- **Collaborators** who can view and edit the graph -- **One or more versions** (snapshots of the workflow at different points in time) +Both `classify` and `draft-reply` are atomic nodes, meaning each does the work itself rather than delegating to another agent; right now that means both are LLM Prompt nodes, since that's the only kind of atomic node the platform ships with. ---- - -## Nodes +A node doesn't have to do the work itself, either. It can instead be a reference to another agent's [saved version](/docs/agent-playground/concepts/versions-and-execution), a version you've saved rather than a draft still being edited, letting you reuse a whole agent as a single step, as covered in Composing agents with an Agent Node below. -Nodes are the building blocks of your workflow. Each node represents a single step that takes inputs, performs an operation, and produces outputs. +## How connections wire together -### LLM Prompt Nodes +Three properties hold for every connection in an agent, and they explain most of what you'll run into. -LLM Prompt nodes execute a prompt against a language model. They connect directly to the **Prompt Management** system: +- **One output can feed several inputs at once.** If `classify`'s `response` output is useful to more than one downstream node, connect it to as many inputs as you need; each one gets the same value +- **Every input accepts exactly one connection.** `draft-reply`'s `category` input can be fed by `classify` or by some other node, but never both at the same time. If two outputs could plausibly feed the same input, you pick one +- **The wiring can never loop back on itself.** Data only flows forward, from a node to the nodes downstream of it, never back to a node it already came from. An agent that tried to connect `draft-reply`'s output back into `classify`'s input would be forming a loop, and the platform rejects that connection -- **Prompt template** defines the prompt text with `{{variable}}` placeholders -- **Model** specifies which LLM to call (GPT-4, Claude, etc.) -- **Parameters** control generation behavior (temperature, max tokens, top-p) -- **Response format** determines output structure (plain text or JSON) +## What isn't connected becomes the agent's own input or output -When the linked prompt template is updated, the node's input ports automatically sync to match the new variables. +Not every input ends up fed by a connection, and not every output ends up feeding one, and that's what makes the model click. -### Agent (Subgraph) Nodes +Inputs that nothing feeds are the agent's own inputs: the [values you fill in before a run](/docs/agent-playground/guides/build-workflow/set-input-variables). `classify`'s `message` input has no connection into it, so `message` is what you provide when you run `support-triage`. -Agent nodes embed an entire other graph as a single step in your workflow. This enables: +Outputs that nothing consumes work the same way in reverse: they're what the run hands back. If `draft-reply`'s `response` output isn't wired into anything, `draft-reply`'s `response` is part of `support-triage`'s result. -- **Modularity**: break complex workflows into reusable sub-workflows -- **Composition**: combine multiple agents into a larger pipeline -- **Encapsulation**: the parent graph only sees the subgraph's exposed input and output ports +## Composing agents with an Agent Node - - Subgraph nodes can only reference **non-draft** versions of other graphs, never drafts. Circular references (Graph A embeds Graph B which embeds Graph A) are detected and blocked. - +An **Agent Node** is a node whose job is to run another saved agent as a single step, instead of doing the work itself. You point it at one of that other agent's saved versions, and because the referenced agent has its own inputs, the Agent Node exposes those same inputs as its own. You map each of them in the node's Input Mapping section: it's where you choose which value in the current agent feeds an input that belongs to the nested agent, though you don't have to map every one. Leave a mapping empty and that input becomes one of the agent's own inputs, the same rule covered above. See [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) for the form. ---- +Say `support-triage` needs to compress `draft-reply`'s output before it goes out, using a category-aware summarizer someone already built. Add an Agent Node, `summarize-step`, pointing at a saved version of a separate agent called `summarizer`. `summarizer` takes two inputs, `text` and `category`, and produces one output, `response`. Once `summarize-step` is in place, `text` and `category` become inputs on `summarize-step` itself: `classify`'s `response` output can now fan out to feed both `draft-reply` and `summarize-step`, and `draft-reply`'s `response` output connects into `summarize-step`'s `text` input. -## Ports + CL["classify"] + CL -->|"response"| DR["draft-reply"] + CL -->|"response"| SS["summarize-step"] + DR -->|"response"| SS + SS -.->|"points to"| SUM["summarizer (saved version)"] + SS -->|"response"| RES(("agent output"))`} /> -Ports are typed connection points on every node. They define the data contract: what a node expects as input and what it produces as output. - -Each port has: -- **Direction**: input or output -- **Key**: a unique identifier (e.g., `prompt`, `response`, `output`) -- **Display name**: a human-readable label -- **Data schema**: a JSON Schema definition that validates data at runtime - -### Exposed Ports - -When an input port has no incoming edge, it becomes an **exposed port**: an entry point for the graph. Similarly, output ports with no outgoing edges are exposed as graph outputs. Exposed input ports automatically become columns in the graph's dataset for execution. - ---- +That last connection changes what's exposed. `draft-reply`'s `response` is no longer unconnected, so it drops out of `support-triage`'s result, and `summarize-step`'s own `response` output takes its place as the new exposed output. Nothing else changes: the rest of the wiring, and the rules that govern it, are exactly the ones from the last two sections. -## Edges +What an Agent Node can't point at: -Edges are the connections that carry data between nodes. Each edge links one node's output port to another node's input port. - -**Rules:** -- **Fan-out is allowed**: one output port can connect to multiple input ports (data is broadcast to all targets) -- **Fan-in is blocked**: each input port accepts only one incoming edge -- **No cycles**: the graph cannot loop back on itself. The platform detects and prevents cycles at connection time -- **Type validation**: the platform checks that connected ports have compatible data schemas - ---- - -## Node Templates - -Node templates are the registry of available node types. They define the default configuration for each type of node, including: - -- **Port definitions**: what inputs and outputs the node type has -- **Port mode**: strict, extensible, or dynamic -- **Config schema**: JSON Schema for the node's configuration (model parameters, settings, etc.) - -The platform ships with built-in templates (LLM Prompt, Agent) and supports custom templates for specialized use cases. Templates are seeded system-wide and available to all users. - - - When you drag a node from the selection panel onto the canvas, the platform creates a new node instance from the matching template and auto-generates its ports based on the template's port definitions. - - ---- +- **A draft version.** Only a saved version of the other agent is a valid target +- **This agent itself.** An agent can't be a step inside itself +- **An agent that already contains this one as one of its own steps, however indirectly.** That would form a loop between agents instead of within one, and it's rejected the same way a loop inside a single agent is -## Next Steps +## Keep exploring -- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): How the version lifecycle and execution model work -- [Create a Graph](/docs/agent-playground/features/create-graph): Create your first workflow -- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them + + + How a draft becomes a saved version, and how a run moves data through an agent + + + Create the agent itself, before you add any nodes + + + Add nodes to that agent, configure them, and wire the connections between them + + diff --git a/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx b/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx index 8cfc810f..5bc94d76 100644 --- a/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx +++ b/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx @@ -1,100 +1,65 @@ --- -title: "Agent Workflow Versions & Execution" -description: "Understand Agent Playground's draft/non-draft version lifecycle, topological execution model, node states, and data routing between nodes." +title: "Versions & Execution" +description: "How drafts become saved versions, and what one run records" --- -## About +## Versions and executions: two models -This page covers how Agent Playground handles versioning and execution. Versions let you iterate safely on your workflow. Execution is how the platform runs your graph and tracks results per node. +A version is a numbered snapshot of an agent's graph. A version starts as a **draft**, the state you can edit; saving fixes it into an immutable snapshot. An **execution** is the record of one run of a version. ---- - -## Version Lifecycle - -Every graph manages its workflow through **versions**: immutable snapshots of the graph's structure (nodes, ports, edges, and configuration) at a point in time. - -### Draft vs Non-Draft - -There are two states a version can be in: - -| State | Editable | Executable | -|-------|----------|------------| -| **Draft** | Yes | No | -| **Non-draft** | No | Yes | - -### How Versions Work - -1. **Draft** - When you create or modify a graph, you work in a draft. Drafts are auto-saved as you make changes. This is your workspace for experimenting and iterating freely. - -2. **Save** - When a draft is ready, you save it. This finalizes the draft into a non-draft version, making it the version that runs when you execute the graph. The previous version is kept in your version history. +A run always executes one specific saved version, so the three stay linked: what you can currently change, what you saved, and what happened when it ran. -You can always view older versions in the **Changelog** tab and create a new draft from any of them to pick up where you left off. + Versions + subgraph Versions["support-triage's versions"] + History["Version 1, 2, 3 ..."] + Draft["Draft"] + Draft -->|"Save Agent"| V4["Version 4 (runs)"] + end + V4 -->|"Run"| ExecBox + subgraph ExecBox["Execution (one run)"] + Classify["classify: success"] + DraftReply["draft-reply: success"] + Summarize["summarize-step: success"] + end + Summarize --> Nested["Nested execution"]`} /> - - Saving a version triggers validation: the platform checks that all required connections exist and the graph has no cycles. If validation fails, the version stays as a draft. - +## Drafts and versions -### Version Workflow +Every change you make, adding a node, editing a connection, rewriting an input, lives in a draft. A draft is marked with the Draft badge, and it is the only kind of version you can edit. -``` -Create Graph → Draft v1 - ↓ (save) - v1 → Edit → Draft v2 - ↓ (save) - v1 v2 → Edit → Draft v3 - ↓ (save) - v1 v2 v3 -``` +[Save Agent](/docs/agent-playground/guides/build-workflow) turns the draft into a version: you write a commit message for it, and it becomes a numbered snapshot carrying that message. Saving validates the graph before that version can run: ---- - -## Execution Model - -When you run a graph, the platform creates a **graph execution**: a record of that specific run with its own ID, status, timing, and results. +- [Exposed output](/docs/agent-playground/concepts/understanding-agent-playground#what-isnt-connected-becomes-the-agents-own-input-or-output) names cannot duplicate +- Every node's required inputs must be present -### Execution States +The save is blocked until both checks pass. Once it succeeds, that new version becomes the one the agent runs, replacing whichever version ran before it. -| State | Meaning | -|-------|---------| -| **Pending** | Execution created, waiting to start | -| **Running** | Nodes are actively executing | -| **Success** | All nodes completed successfully | -| **Failed** | One or more nodes encountered an error | -| **Cancelled** | Execution was stopped by the user | +Older versions do not disappear. They stay available to read and preview, though only the draft can be edited. An agent always keeps at least one version, so there is never a state with nothing to run. -### Node Execution +## Executions and node records -Within a graph execution, each node gets its own **node execution** record tracking: -- Start and end timestamps -- Status (pending, running, success, failed, skipped) -- Input data received from upstream nodes -- Output data produced -- Error details (if failed) +Running an agent creates an execution: one record of that run as a whole, plus one record per node inside it. Each node record carries its own status, one of pending, running, success, failed, or skipped, along with the inputs it received and the outputs it produced. -Nodes that cannot execute because an upstream node failed are marked as **skipped**. - ---- +A node whose upstream step failed is marked skipped rather than being run at all, so a failure does not silently propagate as if the node had executed. In a run of `support-triage`, for example, if `draft-reply` fails, `summarize-step` is skipped rather than run, since it depends on `draft-reply`'s output, while `classify`, which has no dependency on `draft-reply`, is unaffected. -## Data Routing +Independent branches do not wait on each other: up to ten nodes run at the same time by default, so parts of the graph with no dependency between them finish in parallel instead of one after another. -The execution engine processes nodes in **topological order**: it determines which nodes can run first (those with no dependencies) and works forward through the graph. +Nesting closes the loop between the two models. `summarize-step`, for example, is an [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node): its record holds the nested run of the agent it points at. You can open a run inside a run and keep going as deep as the graph nests. -### How Data Flows - -1. **Graph inputs** are injected into the exposed input ports (ports with no incoming edges) -2. **Start nodes** (nodes with all inputs satisfied) execute first -3. When a node completes, its output data is **routed** along edges to downstream nodes' input ports -4. A downstream node becomes **ready** when all its required input ports have data -5. Ready nodes execute, and the process repeats until all nodes are done -6. **Graph outputs** are collected from exposed output ports (ports with no outgoing edges) - -Each piece of data flowing through a port is validated against the port's JSON Schema. Validation errors are recorded but do not block execution: you can inspect them after the run to identify data contract issues. - ---- +## Why the version matters -## Next Steps +A run doesn't just execute "the agent", it pins one specific saved version, and each execution stays tied to the version it ran. So two runs of the same agent differ only by what you changed between the versions they pinned. -- [Create a Graph](/docs/agent-playground/features/create-graph): Create your first workflow and manage versions -- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them -- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute workflows and inspect results +## Keep exploring + + + Open the Changelog to read and preview older versions + + + Run a workflow and read node status and output from the Executions tab + + diff --git a/src/pages/docs/agent-playground/features/build-workflow.mdx b/src/pages/docs/agent-playground/features/build-workflow.mdx deleted file mode 100644 index e2aff337..00000000 --- a/src/pages/docs/agent-playground/features/build-workflow.mdx +++ /dev/null @@ -1,126 +0,0 @@ ---- -title: "Build an AI Agent Workflow" -description: "Add LLM Prompt and Agent nodes to the canvas, configure models and parameters, draw edges, and set global variables in Agent Playground." ---- - -## About - -The Agent Builder is the visual graph editor where you assemble your workflow by adding nodes, configuring them, and connecting them with edges. For background on nodes, ports, and edges, see [Understanding Agent Playground](/docs/agent-playground/concepts/understanding-agent-playground). - -![Agent Builder showing the full workspace with nodes, canvas, and configuration drawer](/images/docs/agent-playground/builder-overview.png) - -The workspace has three main areas: -- **Node Selection Panel** (left): available node types to add -- **Canvas** (center): the graph editor where you arrange and connect nodes -- **Node Drawer** (right): configuration form for the selected node - ---- - -## Add Nodes - -The left panel shows the available node types. You can add nodes in two ways: - -- **Click** a node type to add it to the center of the canvas -- **Drag** a node type onto the canvas and drop it at the desired position - -![Node selection panel with LLM Prompt and Agent node types](/images/docs/agent-playground/node-selection-panel.png) - -### Available Node Types - -| Node Type | Purpose | -|-----------|---------| -| **LLM Prompt** | Execute a prompt against a language model. Configured via Prompt Templates. | -| **Agent** | Embed another graph as a sub-workflow for modular composition. | - -When you add a node, the platform automatically creates its ports based on the node template's definitions. - ---- - -## Configure Nodes - -Click any node on the canvas to open the **Node Drawer** on the right side. The drawer shows a configuration form specific to the node type. - -![Node configuration drawer showing LLM Prompt settings](/images/docs/agent-playground/node-drawer-config.png) - -### LLM Prompt Node Configuration - -| Field | Description | -|-------|-------------| -| **Prompt Template** | Select a prompt template from Prompt Management. The node's input ports automatically sync to the template's `{{variables}}`. | -| **Model** | Choose the LLM to call (e.g., GPT-4, Claude, Gemini). | -| **Temperature** | Controls randomness (0 = deterministic, 1 = creative). | -| **Max Tokens** | Maximum length of the generated response. | -| **Top-p** | Nucleus sampling threshold. | -| **Response Format** | Output as plain text or structured JSON. | - - - When you change the prompt template, the node's input ports update automatically to match the new template variables. Existing connections to removed variables are disconnected. - - -### Agent Node Configuration - -For Agent (subgraph) nodes, configure: -- **Agent** - choose which graph to embed as a sub-agent -- **Version** - select which version of that agent to use -- **Input mapping** - map variables from the parent graph to the sub-agent's exposed input ports - -![Agent node configuration showing agent selection, version, graph preview, and input mapping](/images/docs/agent-playground/agent-node-config.png) - ---- - -## Connect Nodes - -Create data flow connections by drawing edges between nodes. - - - - Hover over a node's **output handle** (the circle on the right side of the node). Your cursor changes to a crosshair. - - - Click and drag from the output handle toward the target node's **input handle** (the circle on the left side). - - - - Release the mouse over the target node's input handle. The platform creates the edge and validates that the port types are compatible. - - - -### Connection Rules - -- **One output to many inputs**: an output port can connect to multiple input ports (data is broadcast) -- **One input, one source**: each input port accepts only one incoming edge -- **No cycles**: the platform prevents connections that would create loops in the graph -- **Type checking**: connected ports must have compatible data schemas - -To **delete an edge**, select it and press Delete. - ---- - -## Global Variables - -Use the **Global Variables** panel (accessible from the right side of the builder) to define values for your prompt variables. These are the inputs your workflow needs to run , for example the customer message that gets passed into your first node. - -![Global variables panel showing prompt variable with a test value](/images/docs/agent-playground/global-variables.png) - ---- - -## Tips - - - Use the **+** button that appears below a node (when it has no outgoing edge) to quickly add and connect a new node in one step. - - - - Changes to a draft version are auto-saved as you work. You do not need to manually save after every edit. The platform saves node positions, configurations, and connections automatically. - - - - If you are viewing a non-draft version, the canvas is read-only. You must create a draft to make changes. The platform will prompt you to create a draft when you try to edit. - - ---- - -## Next Steps - -- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute your workflow and watch results in real time -- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): Understand version lifecycle and execution model diff --git a/src/pages/docs/agent-playground/features/create-graph.mdx b/src/pages/docs/agent-playground/features/create-graph.mdx deleted file mode 100644 index 0f839633..00000000 --- a/src/pages/docs/agent-playground/features/create-graph.mdx +++ /dev/null @@ -1,68 +0,0 @@ ---- -title: "Create an Agent Graph & Manage Versions" -description: "Create a new agent graph in Agent Playground, set metadata, manage draft and saved versions, and roll back to previous workflow snapshots." ---- - -## About - -Create a new graph to start building your AI workflow. A graph is the container for your entire pipeline. For background on what graphs are, see [Understanding Agent Playground](/docs/agent-playground/concepts/understanding-agent-playground). - ---- - -## Create a New Graph - - - - Go to **Agent Playground** from the main navigation. You will see the agent list view showing all your existing graphs. - - ![Agent list view showing existing graphs](/images/docs/agent-playground/agent-list-view.png) - - - Click **Create Agent** in the top-right corner. The platform creates a new graph with a blank draft version and takes you directly to the builder canvas. - - - - Give your graph a meaningful name and description. - - - ---- - -## Manage Versions - -Agent Playground uses a version system to track changes and let you roll back safely. Every graph starts with a draft version. - -### View Versions - -Switch to the **Changelog** tab to see all versions of your graph. The left sidebar lists every version with its status and creation date. Click a version to preview its workflow structure in read-only mode on the right panel. - -![Changelog view with version list and graph preview](/images/docs/agent-playground/changelog-versions.png) - -### Activate a Draft - -When your draft is ready for use: - -1. Click **Save** - -The platform validates the graph (checking for cycles, missing connections, and incomplete configurations). If validation passes, the draft is saved and becomes the version that runs when you execute the graph. The previous version is kept in your version history. - -### Create a Draft from a Previous Version - -To iterate on an older version: - -1. Open the **Changelog** tab -2. Select any previous version -3. Click **Create Draft** - -This creates a new draft that is a copy of that version's workflow structure. You can then modify it freely without affecting the saved version. - -### Edit a Non-Draft Version - -If you try to edit a node while viewing a non-draft version, the platform prompts you to create a draft first. Your edits go into the new draft, leaving the saved version unchanged until you explicitly save the draft. - ---- - -## Next Steps - -- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them into a pipeline -- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute your workflow and inspect results diff --git a/src/pages/docs/agent-playground/features/run-and-monitor.mdx b/src/pages/docs/agent-playground/features/run-and-monitor.mdx deleted file mode 100644 index 97e0f06f..00000000 --- a/src/pages/docs/agent-playground/features/run-and-monitor.mdx +++ /dev/null @@ -1,105 +0,0 @@ ---- -title: "Run & Monitor Agent Workflows" -description: "Execute AI agent workflows, watch per-node status in real time, inspect input/output data for each step, and browse full execution history." ---- - -## About - -Run your workflow and monitor each step as it executes. The platform shows real-time status per node, records full input/output data, and keeps a history of all past runs. - ---- - -## Run a Workflow - - - - Navigate to your graph and open the **Build** tab. Make sure all nodes are configured. The platform highlights unconfigured nodes with a red border. - - - Click the **Run** button (play icon) in the builder actions on the right side of the canvas. - - - **If you are on a draft version**: the platform validates the graph, prompts you to save, and then activates and executes the workflow. - - **If you are on a non-draft version**: the workflow executes immediately. - - Validation checks for: - - All nodes are fully configured - - No cycles in the graph - - All required ports are connected - - - The **Run Agent Panel** opens at the bottom of the builder. Nodes update in real time as they execute: - - - **Green animated border**: node is currently running - - **Green solid border**: node completed successfully - - **Red border**: node failed - - **Gray**: node is pending or was skipped - - Edges animate to show data flowing between nodes. - - ![Workflow running with real-time node status updates](/images/docs/agent-playground/workflow-running.png) - - - ---- - -## View Execution Results - -The **Run Agent Panel** at the bottom of the builder shows detailed results after (and during) execution. - -![Run Agent Panel showing graph status and node output details](/images/docs/agent-playground/run-agent-panel.png) - -The panel is split into two halves: - -### Left: Graph Visualization -A miniature view of your graph with nodes colored by execution status: -- Green = success -- Red = failed -- Gray = pending or skipped - -Click any node in this view to inspect its details on the right. - -### Right: Node Output Details -Shows the selected node's execution data: - -| Field | Description | -|-------|-------------| -| **Execution ID** | Unique identifier for this node's execution | -| **Status** | Success, failed, skipped, running, or pending | -| **Duration** | How long the node took to execute | -| **Input Data** | The data received from upstream nodes | -| **Output Data** | The data produced by this node (JSON or text) | -| **Error** | Error message and details (if the node failed) | - -The panel auto-selects the last executed node when the workflow completes. - ---- - -## Execution History - -The **Executions** tab shows a complete history of all runs for this graph. - -![Executions history with list and detail view](/images/docs/agent-playground/executions-history.png) - -### Browse Executions - -The left sidebar lists all executions, most recent first. Each entry shows: -- Timestamp -- Status badge (success, failed, running, pending) -- Version used - -Click an execution to load its details on the right. The same graph visualization and node output panel from the builder. - -### Inspect a Past Execution - -Select any execution to see: -1. The full graph with per-node status colors -2. Click individual nodes to see their input data, output data, timing, and errors -3. Compare different executions to understand how changes affected results - ---- - -## Next Steps - -- [Build a Workflow](/docs/agent-playground/features/build-workflow): Modify your workflow and add more nodes -- [Create a Graph](/docs/agent-playground/features/create-graph): Create another graph or manage versions -- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): Understand the execution model in depth diff --git a/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx new file mode 100644 index 00000000..35ee7223 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx @@ -0,0 +1,62 @@ +--- +title: "Configure an Agent node" +description: "Reference another agent's saved version in an Agent node and map your workflow's values onto its inputs." +--- + +The Agent Node's configuration form lives in the node drawer, and every field in it sets up which agent this step hands off to. It's where you tell the node which agent to run, which version of that agent, and which of the parent workflow's values feed its inputs. Beyond those three, it carries no configuration of its own. + + +This guide picks up once a workflow with an Agent Node already exists on the canvas; see [Build a workflow](/docs/agent-playground/guides/build-workflow) to add one first. + + +## Open the drawer + +Click the Agent Node on the canvas. Its drawer opens with the node's own configuration form. + +Agent Node drawer with the Agent and Version fields filled in, a preview of the nested agent, and the Input Mapping section showing unmapped rows still reading Select variable + +*The Agent Node drawer, with an agent and version selected and its Input Mapping rows ready to wire up* + +## Choose the agent and version + +Under **Agent**, select the agent to nest. The field starts empty, with the placeholder "Select agent". + +Under **Version**, select which version of that agent to run. Until an agent is chosen, this field stays on "Select an agent first"; once you choose an agent, its latest non-draft version is preselected here, and you can change it to any of its other versions. + +You can select any active or inactive saved version of a **different** agent. + + +The one reference rule Save can still refuse is referencing a second version of an agent you've already referenced elsewhere in this workflow. + + +## Map the inputs + +Input Mapping lists one row per input the nested agent expects. Row labels are the nested agent's own input names, set when that agent was built rather than here; they double as the mapping keys. If the nested agent has no inputs, the Input Mapping section doesn't appear at all. + +Each row has a **Variable** select, with the placeholder "Select variable", and its options are the output ports of the nodes connected directly into this one, labeled `node_name.output_name`. Pick the upstream output that should feed that input. + +For example, say the nested agent expects an `invoice_details` input, and a `classify_invoice` node feeds into this Agent Node. The `invoice_details` row is where you'd pick `classify_invoice.response_1` from the **Variable** select to pass that output down. + + +Leave a row unmapped and no edge is created for it; that input becomes one of the parent workflow's own input variables instead, and per [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables), the run refuses to start until it has a value. + + +## Save the node + +Click **Save**. A successful save closes the drawer. If the save fails, a toast reads "Failed to save agent node" and the node reverts. + +The nested agent's own run appears inside the parent run's results; see [Run an agent](/docs/agent-playground/guides/run-an-agent) for what that looks like. + +## Dive deeper + + + + The full set of limits and validation rules Agent Playground enforces + + + Where the parent workflow's own values come from + + + See where a nested run lands in the parent's results + + diff --git a/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx new file mode 100644 index 00000000..e91ac2b5 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx @@ -0,0 +1,57 @@ +--- +title: "Configure an LLM Prompt node" +description: "Name the node, pick its prompt version and model, then save without breaking what's wired into it" +--- + +The LLM Prompt node's configuration form lives in the node drawer, and most of it comes from the prompt you pick. Walk the form top to bottom and each choice sets up the next. + + +This guide picks up once a workflow with an LLM Prompt node already exists on the canvas; see [Build a workflow](/docs/agent-playground/guides/build-workflow) to add one first. + + +## Open the drawer + +Click the LLM Prompt node on the canvas. Its drawer opens with the node's own configuration form. + +## Name the node + +**Prompt Name** is required and sits at the top of the form. Typing here sets this node's name: what you type is lowercased, every character other than `a`-`z`, `0`-`9`, and `_` is replaced with an underscore, and leading underscores are stripped. + +A name that already belongs to another node on the canvas is refused with "A node with this name already exists". Pick a different name and try again. + +## Pick a version + +A version select sits beside Prompt Name. Its options are labeled with the version, uppercased. When the selected version hasn't been saved yet, a **Draft** badge appears beside the select, not on the option inside the dropdown. + +Not every version can be used here. If the version you pick has an output format the builder doesn't support, the form shows: "This prompt uses an unsupported output format. Only text-based prompts are supported in the agent builder." While that alert shows, the model picker, the **Tools** control, and **Save prompt** are all disabled. + +## Choose the model + +The model picker selects which LLM this node uses. It's also what unblocks the **Tools** button beside it: Tools stays disabled until a model is chosen, and hovering it before then shows why, "Select a model first". + +## Inputs follow the prompt + +The node's inputs are generated from the prompt's `{{variable}}` placeholders, and those placeholder names become the node's input names. See [Limits & rules](/docs/agent-playground/reference/limits-and-rules) for naming restrictions. Swapping the prompt text or picking a different version changes that set of placeholders, so it also changes the set of inputs the node exposes. + +## Save the prompt + +**Save prompt**, at the bottom of the form, writes your changes. If the save fails, a toast reads "Failed to save prompt" and the form stays open so you can retry. On a successful save, the drawer closes. + + +Picking a different version of the prompt can change what this node returns: the output shape downstream nodes expect may shift, since the node's response follows the response format set on the linked version. + + +## Closing with unsaved changes + +Close the drawer while a change is unsaved and a dialog titled "Unsaved Changes" asks "You have unsaved changes. Are you sure you want to discard them?" Confirm with **Discard** to drop the edits, or back out of the dialog to go save first. + +## Dive deeper + + + + Wire the node's inputs to the workflow's values + + + Try the workflow with the node configured + + diff --git a/src/pages/docs/agent-playground/guides/build-workflow/index.mdx b/src/pages/docs/agent-playground/guides/build-workflow/index.mdx new file mode 100644 index 00000000..d132b3bc --- /dev/null +++ b/src/pages/docs/agent-playground/guides/build-workflow/index.mdx @@ -0,0 +1,72 @@ +--- +title: "Build a workflow" +description: "Add nodes to the canvas, wire them up, and save the graph" +--- + +This guide picks up once you've [created an agent](/docs/agent-playground/guides/create-agent) and opened it, on an empty canvas. As the running example, say you're building Invoice Triage, a workflow that needs to read each invoice and decide where it goes: an **LLM Prompt** node to classify the invoice, feeding an **Agent Node** that routes it to the right approver. Getting that shape onto the canvas, wiring the two nodes together, and saving the result is what this guide walks through. + + +What you type into either node's own settings isn't covered here; see [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) and [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node). + + +## Tour the builder + +Open an agent and you land on the **Agent Builder** tab, one of three tabs across the top alongside **Changelog** and **Executions**. Agent Builder holds the canvas itself; the other two sit outside what this guide covers. + +The builder splits into three regions. The **node palette** sits on the left, listing the node types you can place. The **canvas** fills the middle and holds the graph as you build it, the nodes and the connections between them. The **node drawer** opens on the right once you select a node, carrying that node's own settings. + +The Agent Builder with the Agent Builder, Changelog, and Executions tabs across the top, the node palette on the left, a two-node graph on the canvas, and the node drawer open on the right +*Agent Builder, Changelog, and Executions sit across the top; palette, canvas, and drawer make up the three regions underneath* + +## Add a node to the canvas + +The node palette lists the node types you can add, among them LLM Prompt, described as "Run a prompt against an LLM", and Agent Node, described as "Run an agent through LLM". Get either one onto the canvas by clicking its card, which drops the node straight onto the canvas, or by dragging the card and releasing it wherever you want the node to land. + +The node palette listing LLM Prompt, described as Run a prompt against an LLM, and Agent Node, described as Run an agent through LLM +*Click or drag either card onto the canvas* + +Reach for an LLM Prompt node for a single, focused call to an LLM, the way Invoice Triage uses one to classify an invoice. Reach for an Agent Node when the step needs to run an agent, the way Invoice Triage uses one to route the invoice to the right approver. + +There's a third way to add a node, once you already have one down: a node with no outgoing connection carries a **+** button to its right. Click it and the same node picker opens, so the new node lands already connected to the one before it. + +## Connect nodes + +A single configured node already runs on its own; wiring is how one node's output becomes the next node's input. Every node has an output handle and an input handle; drag from one node's output handle to the next node's input handle, and the builder draws the connection between them. On Invoice Triage, that means dragging from the classifier's output to the router's input. + +An output handle on the classifier connected by a dashed line to an input handle on the router +*Output and input handles sit on the edge of each node; drag from the classifier's output to the router's input to connect them* + +Connections follow a few rules: +- One output can feed as many inputs as you connect it to, so a single node's result can branch into several downstream nodes at once +- One input takes only one source +- A connection that loops back into a node's own upstream path draws fine but won't save + +Invoice Triage's LLM Prompt node still needs the invoice text itself to classify, and that value comes from outside any node, as an [input variable](/docs/agent-playground/guides/build-workflow/set-input-variables). + +## Delete a node + +A node can go from two places: a delete icon sits right on the node itself on the canvas, and the node drawer offers the same action for whichever node you have open. The canvas icon removes the node immediately, with no confirmation. The drawer's delete icon opens a dialog titled **Delete Node** that asks "Are you sure you want to delete this node? This action cannot be undone." Click **Delete** to confirm, or close the dialog to keep the node. If the deletion doesn't go through, a "Failed to delete node" toast tells you. + +## Save Agent + +Click **Save Agent** once the graph looks right. If the graph contains a cycle or a node is left unconfigured, **Save Agent** toasts the error instead of opening anything. Once the graph passes that check, the dialog opens, carrying a **Version** field you can't edit, a **Commit Message** box for a note about the change, and a single action button. That button reads **Save**, or **Save & Run** if you reached the dialog by clicking **Run Agent Workflow** on an unsaved draft, in which case it also runs the agent after saving, the same run covered in [Run an agent](/docs/agent-playground/guides/run-an-agent). + +The dialog closes and the graph is saved as a new version; find it later, commit message and all, in the [Changelog tab](/docs/agent-playground/guides/manage-versions). + +**Save Agent** itself stays disabled until you have permission to edit the agent, the agent is on a draft, you're on the Agent Builder tab, no run is already in progress, and the canvas has finished loading. Hover a disabled button to see why; without edit permission, the tooltip reads "You don't have permission to edit this agent." + +With that saved, Invoice Triage has its shape: an LLM Prompt node feeding an Agent Node, connected and versioned. Configuring what each node actually does comes next. + +## Dive deeper + + + + The settings behind an LLM Prompt step + + + The settings behind an Agent Node step + + + Feed values into the graph from outside any one node + + diff --git a/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx b/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx new file mode 100644 index 00000000..97f13135 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx @@ -0,0 +1,41 @@ +--- +title: "Set input variables" +description: "Open the Variables drawer, fill in each field, and see what a run does when one's still empty." +--- + +A run needs a value for every input variable before it can start. Set them from the builder before you run anything. + + +This guide picks up once a workflow with at least one LLM Prompt node already exists on the canvas; see [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) to add one first. + + +## Open the Variables drawer + +In the builder, click **Add input variables** at the top right of the canvas. The drawer opens headed **Variables**, with the subtext "Define values for your prompt variables". The list comes from your workflow's saved version, so a variable you just added won't show up until the node and the agent are saved. + +## Fill in each variable + +The drawer lists each variable by name, with a field beside it for the value you want this run to use. Give every listed variable a concrete value. If your prompt asks for the invoice text to classify, for example, fill that variable with the actual invoice, such as "Invoice #4521 from Acme Supplies, $2,400 due in 30 days." Only inputs with no incoming connection show up here, since those are [the agent's own inputs](/docs/agent-playground/concepts/understanding-agent-playground#what-isnt-connected-becomes-the-agents-own-input-or-output). + +## Save your values + +Click **Save** to store your values. If a run is waiting on these variables, the button reads **Save & Run Workflow** instead, and saving starts that run. + + +Close the drawer instead while an edit is unsaved, and a dialog titled "Unsaved Changes" asks "You have unsaved changes. Are you sure you want to close without saving?" Confirm with **Discard Changes** to close without keeping them, or cancel to go back and save first. + + +## A run won't start with an empty variable + +Leave any variable blank and a run refuses to start. You'll see the warning "Fill in all variables before running", and the drawer opens on its own with the run held until you fill in what's missing and save. Once every variable has a value, [Run an agent](/docs/agent-playground/guides/run-an-agent) covers starting the run itself. + +## Dive deeper + + + + Start a run once every variable has a value + + + What a saved version is and how a run turns it into node records + + diff --git a/src/pages/docs/agent-playground/guides/create-agent.mdx b/src/pages/docs/agent-playground/guides/create-agent.mdx new file mode 100644 index 00000000..52bbcf57 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/create-agent.mdx @@ -0,0 +1,62 @@ +--- +title: "Create an agent" +description: "Create a new agent from scratch or a template" +--- + +Every agent in Agent Playground starts in the same place: a list you open from the sidebar, and a button that drops you onto a blank canvas. This guide gets a new agent onto that canvas and back out again if you ever need to delete it, nothing more. + + +The example built up across this section is an agent called **Invoice Triage**. Create yours under that name so later guides line up with what's on your screen. + + +## Open the list and create an agent + +Click **Agents** in the sidebar to see every agent in the workspace. Clicking anywhere on a row opens that agent, so there's no separate open button to look for. Once there are more agents than fit on one page, use the Search box above the table to find one by name, and the pagination controls beneath it to page through the rest. If the workspace has no agents yet, the list is replaced by an empty state instead: a **Create your first agent** heading, the line "Break down complex tasks into sequential steps that build upon each other." underneath, and a **Start creating** button in place of **Create Agent**. It opens the same empty canvas. + +Click **Create Agent** in the top right to start a new one. You land straight on an empty builder canvas for a brand-new agent. That agent already exists as an empty draft the moment the canvas opens. + + +If **Create Agent** or, later, **Delete** looks greyed out, hover it: a tooltip explains why, either `You don't have permission to create agents.` or `You don't have permission to delete agents.` + + +## Name your agent + +The new agent already has a name at the top of the canvas, auto-generated from the timestamp, like `Agent Aug 11, 2026 4:52 PM`. Click the edit icon next to it, type **Invoice Triage**, and press Enter to save it before you start building. + +## Choose how to start + +The canvas offers two ways to build the same agent: node by node, or from a template. + +### Build node by node + +Click **Add first node** to open the [node](/docs/agent-playground/concepts/understanding-agent-playground) picker and start building the canvas yourself. + +### Start from a template + +Use the **or start from a template** link instead to open a drawer titled **Agent Templates**, with a **Search templates** box at the top for finding one by name. A template is the quicker start when your agent fits a common use case like writing, coding, or research. Pick one and it loads a ready-made agent onto the canvas in place of the empty one, then continue from [Build a workflow](/docs/agent-playground/guides/build-workflow) to keep building on what it gave you. + + +A **Stop** control appears while a template is loading; using it warns that stopping now will erase your progress and restart the setup. + + +## Remove agents you no longer need + +Back in the agent list, tick the checkbox on one or more rows. The header above the list swaps to a count of how many you've picked, like 3 Selected, with **Delete** and **Cancel** next to it. + +Press **Delete**, and a confirmation dialog titled **Delete agents** asks `Are you sure you want to delete 3 agents?`. Click **Delete** to remove them for good, or **Cancel** to back out and keep them. + +Deleting an agent fails when another agent's node still references one of its versions: the error names both the agent you tried to delete and the agent whose node depends on it. Open that referencing agent, find the node pointing to the version you're trying to remove, and repoint or delete that node first. See [Versions & execution](/docs/agent-playground/concepts/versions-and-execution) for more on how those references work. + +## Dive deeper + + + + Build out the canvas you just created + + + Run the agent you just created and inspect what each step produced + + + Save drafts, browse the Changelog, and restore old versions + + diff --git a/src/pages/docs/agent-playground/guides/manage-versions.mdx b/src/pages/docs/agent-playground/guides/manage-versions.mdx new file mode 100644 index 00000000..4ba42cd2 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/manage-versions.mdx @@ -0,0 +1,29 @@ +--- +title: "Manage versions" +description: "Find a past save by its commit message and preview it on a read-only canvas." +--- + +The Changelog tab lets you look back through an agent's version history without disturbing whatever you're currently building, and preview any version on a read-only canvas. + +## Open the Changelog tab + +From the agent list, open Invoice Triage. It opens on the **Agent Builder** tab, with **Changelog** and **Executions** alongside it across the top; switch to **Changelog** and the view splits into a version list on the left and a canvas preview on the right. + +## Read the version list + +Each entry in the list carries its version number and the [commit message](/docs/agent-playground/guides/build-workflow) you wrote when you saved it, for example version 6 with the message "Add duplicate invoice check," so you can tell what changed without opening anything. Version numbers only increase with each save, so the entry with the highest number is the newest. + +## Preview a version + +Click version 6 and its graph renders on the right, exactly as it looked the moment you saved it. This preview is read-only, draft included. Editing only happens on a [draft](/docs/agent-playground/concepts/versions-and-execution), marked with the Draft badge, back in Agent Builder. + +## Dive deeper + + + + The limits and validation rules Agent Playground enforces + + + Run a workflow and see what each node produced + + diff --git a/src/pages/docs/agent-playground/guides/run-an-agent.mdx b/src/pages/docs/agent-playground/guides/run-an-agent.mdx new file mode 100644 index 00000000..0be90e54 --- /dev/null +++ b/src/pages/docs/agent-playground/guides/run-an-agent.mdx @@ -0,0 +1,57 @@ +--- +title: "Run an agent" +description: "What has to be ready before Run Agent Workflow works, and where a run goes once you step away from the builder." +--- + +Running a graph in Agent Playground executes every node your edges connect, and gives you a live account of what happened at each one: which node ran, what it received, and what it returned. This guide covers pressing **Run Agent Workflow**, reading a run while it's in progress, and finding it again after you've left the builder. + +Running assumes a graph already exists and is saved. This guide runs Invoice Triage, built in [Build your workflow](/docs/agent-playground/guides/build-workflow); build yours first if you haven't. Two more things have to be true before a run starts. + +- Every node on the canvas needs to be configured, or the run is refused with "Node not configured" for one node, or "3 nodes are not configured" when more than one is missing setup. [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) or [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) to clear this +- Every variable the graph needs has to be filled in, or the run is held and the **Variables** drawer opens on its own with "Fill in all variables before running". See [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables) + +Fix whichever one is blocking you and try again. + +## Start the run + +Open your graph in the **Agent Builder** tab, then press **Run Agent Workflow** to execute the graph. The label switches to **Rerun Agent Workflow** the next time you run it, so the button itself tells you whether this is a first run or a repeat. + +If you've made changes you haven't saved, a dialog titled "Unsaved Changes" stops you before anything runs: "You have unsaved node changes. Running now will use the last saved configuration." Click **Run Anyway** to go ahead with the [last saved version](/docs/agent-playground/concepts/versions-and-execution), or close the dialog and save first if the run needs to reflect your edits. + +## Watch it run + +Once the run starts, a run panel opens at the bottom of the builder. It lists each node by name, and **Show Outcome** and **Hide Outcome** fold the panel away or bring it back without stopping anything underneath. + +While it runs, the node currently executing is the one animating on the canvas. A node marked failed didn't complete, and any node downstream of it is marked skipped rather than running. See [Limits & rules](/docs/agent-playground/reference/limits-and-rules) for the full list of statuses. + +Click a node inside the panel, including one marked failed, to see what went into it and what came out. Click the classifier and you'll see the output it handed off, the same value the router receives as its input. + +Once the agent finishes, the panel selects the last node that ran. + +If the run as a whole can't complete for a reason that isn't pinned to one node, the error you see falls back to a generic "Workflow execution failed". Check the panel's per-node statuses for one marked failed, and click it to see what it received and returned. + +## Leave it running + +Click **Exit Workflow** while a run is still going and a dialog titled "Leave running workflow?" asks first: "Your workflow will run in the background. You can find it in the Execution tab." Choose **Leave** to step away, or **Cancel** to stay and keep watching. + +Exit Workflow doesn't stop the run. A toast confirms it: "Exited workflow. It will continue running in the background." The run keeps executing with the builder closed. + +## Find it in Executions + +The **Executions** tab sits across the top of the agent alongside **Agent Builder** and **Changelog**; open it to find that run again, or any other run of this graph. A list of runs sits on the left; select one and its per-node detail loads on the right, the same kind of detail the run panel showed while it was in progress. When a step is an [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node), opening it opens the nested run inside it too. + +If nothing has run yet, the tab shows "No executions yet", with "Run your workflow from the Agent Builder to see results here" underneath. + +## Dive deeper + + + + Open the Changelog tab, read the version list, and preview any saved version + + + The limits and validation rules Agent Playground enforces + + + Symptom-first fixes for the builder's most common blockers + + diff --git a/src/pages/docs/agent-playground/index.mdx b/src/pages/docs/agent-playground/index.mdx index 9b2349f5..6e208bff 100644 --- a/src/pages/docs/agent-playground/index.mdx +++ b/src/pages/docs/agent-playground/index.mdx @@ -1,77 +1,37 @@ --- -title: "Agent Playground: Visual Workflow Builder" -description: "Design and run multi-step AI agent workflows on a drag-and-drop canvas. Chain LLM calls, embed sub-agents, version changes, and trace every execution." +title: "Overview" +description: "Where to start: what Agent Playground is, and the guides for creating, building, and running an agent." --- -## About +## What is Agent Playground? -Agent Playground is Future AGI's workflow builder for AI agents. It lets you design multi-step AI workflows by dragging nodes onto a canvas and connecting them, no code required. +Agent Playground is the visual builder under **Agents** in the sidebar, where you assemble a multi-step agent from nodes on a canvas, run it, and read what each step produced. Reach for it once a single prompt can't do the whole job on its own, such as when one step's output needs to feed the next step, or a step needs to call another agent. -Most AI applications are not a single prompt. They chain steps together: call one model, pass its output to another, combine results, and so on. As these pipelines grow, keeping track of every connection and debugging failures across steps gets harder. When something breaks, tracing which step went wrong means digging through logs. +The graph you build on the canvas is that agent, one step per node. Every node is one of two types: -Agent Playground makes this visual. You build your workflow as a graph of connected steps. Each step is a **node**: like an LLM call or a sub-agent. You draw connections between nodes to define how data flows from one step to the next. When you are ready, you hit **Run** and watch each step execute in real time, with results visible per node. - -Key capabilities: - -- **Visual builder**: drag-and-drop canvas to design workflows without writing code -- **Real-time execution**: run your workflow and watch each step light up as it completes -- **Version control**: draft changes safely, activate when ready, roll back if needed -- **Batch testing**: connect a dataset and run your workflow against hundreds of inputs at once -- **Reusable components**: embed one workflow inside another for modular, composable designs -- **Full traceability**: every run is recorded with complete input/output details per step - ---- - -## How Agent Playground Connects to Other Features - -- **Prompt**: LLM Prompt nodes use prompts you have already built in the Prompt Management system. Update a prompt once, and every workflow using it picks up the change automatically. -- **Dataset**: Each graph has a linked dataset where you can set up input variables and run experiments. Go to Dataset to add rows, fill in values for each input, and execute your workflow across all of them. +- **LLM Prompt** runs a prompt you already built in [Prompt Management](/docs/prompt) +- **Agent Node** runs another saved agent as a single step --- -## Know the Parts - -Before diving in, here is what each term means and how they fit together. - - - - A **graph** is the container for your entire workflow. Think of it as a project: it has a name, description, team collaborators, and one or more saved versions. You build and run workflows inside a graph. - - - A **node** is one step in your workflow. There are two kinds: +## Start here - - **LLM Prompt nodes** call a language model using a prompt template you have set up in Prompt Management. You pick the model, set parameters like temperature, and the node handles the rest. - - **Agent nodes** embed an entire other workflow as a single step, useful for breaking complex pipelines into reusable building blocks. - - - An **edge** is the line connecting two nodes. It defines how data flows from one step to the next. You create edges by dragging from one node's output to another node's input on the canvas. - - - Versions let you iterate safely. Make changes in a **Draft**, then **Activate** it when you are happy with the result. Previous versions are saved, so you can always go back and pick up from an earlier state. - - - An **execution** is one run of your workflow. It records the status of every step (success, failed, running), how long each took, and what data went in and came out. You can browse past executions to debug issues. - - - ---- - -## Getting Started - - - - Learn how graphs, nodes, and connections work together to form workflows. + + + Create your first agent and manage its versions - - Understand the version lifecycle and how workflows run. + + Add nodes, configure them, and connect them on the canvas - - Create your first workflow and start building. + + Run an agent and inspect what each step produced - - Add steps, configure them, and connect them into a pipeline. + + How graphs, nodes, ports, and edges fit together - - Execute workflows, watch results in real time, and browse history. + + How the draft/active version lifecycle and execution model work + +Something not working, or need to know a limit before you build? See [Agent Playground FAQ & fixes](/docs/agent-playground/troubleshooting) and [Limits & rules](/docs/agent-playground/reference/limits-and-rules). diff --git a/src/pages/docs/agent-playground/reference/limits-and-rules.mdx b/src/pages/docs/agent-playground/reference/limits-and-rules.mdx new file mode 100644 index 00000000..0c60d6a0 --- /dev/null +++ b/src/pages/docs/agent-playground/reference/limits-and-rules.mdx @@ -0,0 +1,90 @@ +--- +title: "Limits & rules" +description: "Character limits, connection rules, and paging behavior across Agent Playground" +--- + +These are the limits and validation rules Agent Playground enforces. + +## Naming + +| Name | Maximum length | Cannot contain | +|---|---|---| +| Agent name | 255 characters | No restriction | +| Node name | 255 characters | `.` `[` `]` `{` `}` | +| Input name | 100 characters | No restriction | +| Output name | 100 characters | `.` `[` `]` `{` `}` | + +A name that breaks one of these rules can't be saved. + +## Connections + +Break any of these and the connection won't attach. + +| Rule | Behavior | +|---|---| +| One source per input | An input accepts exactly one incoming connection; an output can feed any number of inputs | +| Direction | A connection always runs from an output to an input | +| Same version | Both connected nodes must belong to the same [version](/docs/agent-playground/concepts/versions-and-execution) | +| No loops | A connection cannot create a cycle, and a node cannot connect to itself | + +## Agent Nodes + +See [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node) for what it is. + +Breaking any of these means the node can't be saved with that target. + +| Rule | Behavior | +|---|---| +| Version status | Points only at an active or inactive version of another agent, never a draft | +| Self-reference | Cannot point at the agent it belongs to | +| One version per target agent | All Agent Nodes in a version that point at the same target agent must point at the same version of it, for example Agent B v3, not v3 and v4 together | +| Cycles | Cannot form a cycle of Agent Nodes pointing at each other | + +## Versions + +Breaking any of these blocks the edit, the activation, or the run. + +| Rule | Behavior | +|---|---| +| Editable | Only the draft version can be edited | +| Running version | Exactly one version of an agent runs at a time | +| Minimum versions | The last remaining version of an agent cannot be removed | +| Output names | Two outputs left unconnected can't share a name within the same version | +| Required inputs | Every node's required inputs must be present | + +## Runs + +### Statuses + +| Level | Possible statuses | +|---|---| +| Run | pending, running, success, failed, cancelled | +| Node step | pending, running, success, failed, skipped | + +### Execution limits + +| Limit | Value | +|---|---| +| Concurrent nodes | Up to 10 nodes run at the same time, per run | +| Step attempts | A step is retried automatically, up to 3 attempts, before it's marked failed | +| Step timeout | A step is cut off after 1 hour if it hasn't finished | + +## Lists + +| List | Page size | +|---|---| +| All lists | 10 rows per page | + +## Keep exploring + + + + Nodes, connections, and nested agents, the pieces these rules apply to + + + How drafts, versions, and runs relate + + + Run a workflow and read node status from the Executions tab + + diff --git a/src/pages/docs/agent-playground/troubleshooting.mdx b/src/pages/docs/agent-playground/troubleshooting.mdx new file mode 100644 index 00000000..b48ed5dc --- /dev/null +++ b/src/pages/docs/agent-playground/troubleshooting.mdx @@ -0,0 +1,102 @@ +--- +title: "Agent Playground FAQ & fixes" +description: "Symptom-first fixes for the builder's most common blockers" +--- + +## In this page + +Hit a wall in the Agent Builder? Each fix below leads with what you see on screen, so scan for the message or symptom that matches yours. If your problem isn't listed, reach out via [support](https://futureagi.com/contact-us) with the error text and the run or agent ID. + +## While building + +### The input I'm connecting to already has a source + +An input port only accepts one incoming edge. Delete the existing edge into that input before drawing the new one. + +See the full set in [Limits & rules](/docs/agent-playground/reference/limits-and-rules#connections). + +### The node name I typed gets rejected + +`A node with this name already exists` + +Every node on the canvas needs a unique name, so pick a different one. + +### The prompt shows an inline error in the drawer + +`This prompt uses an unsupported output format. Only text-based prompts are supported in the agent builder.` + +Pick a text-based version of the prompt; see how to pick a prompt version in [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node). + +## When saving or running + +### Save or Run shows an error toast + +`Node not configured`, or `3 nodes are not configured` when several are; the flagged nodes get highlighted on the canvas so you know which ones to open. + +Open each flagged node's drawer and finish its form; [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) and [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) cover what each form needs. + +`Graph contains a cycle. Remove the circular connection before saving.` + +The canvas lets you draw a connection that loops the graph back on itself; it's only caught here, when you save or run. Remove the connection that closes the loop and save again. + +### Save Agent is greyed out + +Several things gate this button, and it stays disabled until all of them clear: + +- You're not on the builder tab: switch back from Changelog or Executions +- You're viewing a saved version, not the draft: see [Manage versions](/docs/agent-playground/guides/manage-versions) for how to get back into a draft +- A run is still in flight: wait for it to finish +- The canvas is still loading: wait for it to finish +- You don't have permission to edit this agent. Hover the button for the tooltip: `You don't have permission to edit this agent.` See [Roles & Permissions](/docs/roles-and-permissions) for who can grant you edit access + +### Run Agent Workflow is greyed out + +This is gated by the same edit permission as Save Agent. Hover the button for the tooltip: `You don't have permission to run this agent.` See [Roles & Permissions](/docs/roles-and-permissions) for who can grant you edit access. + +### The run won't start + +`Fill in all variables before running`, or `Failed to validate variables` if the check itself can't complete. + +The builder opens the Variables drawer for you and queues the run behind it, so fill in what's missing there; see [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables). + +### The whole run shows as failed + +If the toast reads `Workflow execution failed`, the run never started. Retry it, and if it keeps failing to start, send support the agent ID. + +Otherwise, the toast carries the failed node's own error message. Open the run panel, find the failed node, and fix it; see [Run an agent](/docs/agent-playground/guides/run-an-agent). + +## During a run + +### The run started, but a node's status badge shows skipped + +A node is marked skipped, rather than run, when the step feeding it failed. Fix the failed node upstream and rerun; see [Run an agent](/docs/agent-playground/guides/run-an-agent) for how the run panel shows each node's status. + +## Managing agents + +### I got bounced to the agents list with a missing-agent message + +`Missing agent` + +The builder shows this and sends you back to the agent list when the URL you opened carries no agent ID at all. Open the agent again from the list instead; see [Create an agent](/docs/agent-playground/guides/create-agent). + +### An agent won't delete + +An agent stays undeletable while another agent's node still references one of its versions. Open the referencing agent, remove or swap out that node, then delete again; see [Create an agent](/docs/agent-playground/guides/create-agent#remove-agents-you-no-longer-need). + +### The last version of an agent won't delete + +An agent always keeps at least one version, so its last one can't be removed. Add a new version first if you want to retire the old one; see [Manage versions](/docs/agent-playground/guides/manage-versions). + +## Keep exploring + + + + Open the agent list, create an agent, and clear out the ones you don't need + + + What has to be ready before a run starts, and where it goes once you step away + + + How a draft becomes a saved version, and how a run turns it into node records + + diff --git a/src/pages/docs/annotations/concepts/labels.mdx b/src/pages/docs/annotations/concepts/labels.mdx new file mode 100644 index 00000000..3e313364 --- /dev/null +++ b/src/pages/docs/annotations/concepts/labels.mdx @@ -0,0 +1,66 @@ +--- +title: "Labels" +description: "Why every score built on a label keeps the same shape and options" +--- + +## A reusable question with a fixed answer type + +A **label** is a reusable question plus the answer type it accepts. Attach it to a [queue](/docs/annotations/concepts/queues-and-items) and every annotator working that queue answers the same question, the same way, and each answer lands as a [score](/docs/annotations/concepts/scores). The full object model connecting labels, queues, and scores lives in [Understanding Annotation](/docs/annotations/concepts/understanding-annotation); this page is about the label on its own. + +## Three rules that follow + +**A label's type is locked the moment you create the label** and can't be changed afterward, though you can still edit its name, description, or settings. You can't turn a numeric label into a categorical one, for example: the type fixes both the control the annotator sees and the shape of the value written into every resulting score. Picked the wrong type? [Create a new label](/docs/annotations/guides/create-label) with the right type instead of trying to change this one. + +**A label belongs to the org you created it in, not to any single queue, so editing it changes every queue it's attached to.** A queue attaches an existing label rather than copying it, so editing a label's name, description, or settings changes what every attached queue shows and validates against from that point on. Scores already submitted aren't touched; the edit applies to answers submitted after it. + +**A queue needs at least one label to exist.** You can't create or save a queue with zero labels attached. The label is what turns a pile of items into something answerable. + + C["Annotator's control · the label's option list"] + L --> V["Value shape in every score · the selected option(s)"] + L -->|attached to| Q1["Queue · Support quality review · needs 1+ label"] + L -->|attached to| Q2["Queue · Onboarding review · needs 1+ label"]`} /> + +## Five types, five kinds of judgement + +- **Categorical** collects one option from a list you define, or several if you allow multiple selection, best for a defect category, a sentiment, or an escalation reason +- **Numeric** collects a number within the range you set, best for relevance on a 1-to-10 scale or a quality score out of 100 +- **Text** collects free-text feedback in the annotator's own words, best for detail that doesn't reduce to an option or a number +- **Star Rating** collects a star count on a scale you choose when creating the label, from 1 up to 10 stars, best for a fast overall impression +- **Thumbs Up/Down** collects a single up or down call, best for a binary pass or fail judgement + +The settings each type requires and the exact validation applied to a submitted value live on [Label types & values](/docs/annotations/reference/label-types-and-values). + +## Picking a type + +Match the type to the shape of the judgement, not the topic: + +| If you need | Pick | +| --- | --- | +| The answer to be one or more of a few known outcomes | Categorical | +| The answer to be a quantity | Numeric | +| The answer explained, not selected | Text | +| A quick, coarse gut check | Star Rating | +| The answer to be strictly one of two | Thumbs Up/Down | + +If an item needs more than one kind of judgement, attach more than one label to the queue rather than stretching a single label to cover two jobs. + +## Why it matters + +Standardizing on a small set of labels (one for quality, one for tone) keeps scores comparable across teams. `Response quality`, for instance, is a categorical label with three options, `Good`, `Needs work`, `Wrong`, so everyone answering it is choosing from that exact same set, not inventing their own scale. + +## Keep exploring + + + + How labels attach to a queue + + + What a label's answer becomes + + + Get a usable label into your org + + diff --git a/src/pages/docs/annotations/concepts/queues-and-items.mdx b/src/pages/docs/annotations/concepts/queues-and-items.mdx new file mode 100644 index 00000000..bd700da1 --- /dev/null +++ b/src/pages/docs/annotations/concepts/queues-and-items.mdx @@ -0,0 +1,80 @@ +--- +title: "Queues & Items" +description: "The operational layer that turns annotation into a managed campaign" +--- + +## A queue is the campaign, an item is what's inside it + +A **queue** is the campaign. It carries: + +- a status +- the [labels](/docs/annotations/concepts/labels) it's collecting answers for +- the people working it and the role each one holds +- **submissions per item**: how many independent annotators must complete an item before it counts as done +- whether review is required before a submission counts + +An **item** is one thing inside that campaign, pulled from one of the sources Annotation covers (see [Understanding Annotation](/docs/annotations/concepts/understanding-annotation)). It lands in the queue when someone adds it there (see [Add items](/docs/annotations/guides/explore-queue/add-items)), moving from untouched to answered as work happens. + +A queue moves through four statuses of its own: + +- `draft`: doesn't accept annotations yet +- `active`: the only status that accepts submissions, flips on the moment someone activates it +- `paused`: temporarily closed to new submissions without marking the queue done +- `completed`: closed to further submissions, either because a manager marked it done directly or because every item finished on its own; a skipped item blocks that automatic path, so a manager has to close the queue by hand instead + +## How a queue and its items relate + +An item carries a status of its own, plus a review state layered on top of it when the queue requires review. + +The status moves through four values: + +- `pending`: waiting to be picked up +- `in_progress`: being worked on +- `completed`: done +- `skipped`: passed over without an answer + +On a queue where submissions per item is 1, opening an item reserves it for whoever opened it, whether it's still `pending` or already `in_progress`, so a second annotator can't pick up the same one while it's being worked. If the queue's reservation timeout passes before they act on it, the reservation simply lapses and the item becomes claimable again, the status itself doesn't move. On a queue that needs more than one submission per item, that reservation doesn't apply, so more than one annotator can open the same item at once. See [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the exact timeout options. + +An item is done once the required number of annotators have each completed it, with every label the queue requires scored, unless the queue calls for review. When review is required, a submitted item's review state moves to pending review instead of the item completing outright. A reviewer's approval then completes it, and sending it back moves the status to `in_progress` so the annotator can act on the feedback. + +|"has"| QL["Labels attached to the queue"] + Q -->|"has"| QA["Annotators: roles annotator, reviewer, manager"] + Q -->|"holds"| IT["Item: pending"] + IT --> INP["Item: in_progress, reserved"] + INP --> D{"Every label the queue requires scored by enough annotators?"} + D -->|"no"| INP + D -->|"yes, review off"| DONE["Item: completed"] + D -->|"yes, review on"| REV["Item: pending review"] + REV -->|"approve"| DONE + REV -->|"send back"| INP`} /> + +## Roles decide who can do what + +Everyone on a queue holds one or more of three roles. An **annotator** submits [scores](/docs/annotations/concepts/scores). A **reviewer** approves or sends back submissions when the queue requires it. A **manager** configures the queue and its people. Only annotators and managers can actually submit an annotation: holding the reviewer role by itself doesn't grant that. + +Two shortcuts save you from adding people by hand. Whoever creates a queue is added to it as a manager the moment it's saved, and admins act as managers automatically without ever being added as members: org admins on every queue in their org, workspace admins on every queue in their workspace. + +## The default queue you meet before you build one + +You'll often run into a queue before you ever set one up yourself. Future AGI creates at most one default queue per project, dataset, or agent definition, and only the first time that scope actually needs one, not automatically for every one of them, so a project nobody has annotated yet has no default queue. Unlike an ordinary queue, it never auto-completes: because it's meant to keep collecting whatever lands in it indefinitely, no amount of finished items closes it on its own. + +## Why it matters + +The queue is what makes annotation a managed workflow instead of something one person does off to the side. The item is what keeps that workflow honest one row at a time, holding its own status so a single stuck item never quietly skews the read on how the whole batch is progressing. It's also why the same submissions per item rule works whether that count is one or five: the queue sets the bar, and every item is measured against it independently. + +## Keep exploring + + + + What an answer becomes once it's submitted + + + Stand up an active queue with labels and annotators + + + The tabs, toolbar, and rules inside the queue detail view + + diff --git a/src/pages/docs/annotations/concepts/scores.mdx b/src/pages/docs/annotations/concepts/scores.mdx index 38abd050..5fc5c9d9 100644 --- a/src/pages/docs/annotations/concepts/scores.mdx +++ b/src/pages/docs/annotations/concepts/scores.mdx @@ -1,113 +1,84 @@ --- -title: "Annotation Scores: Unified Data Model" -description: "Understand the Score model — the unified annotation primitive storing label values, annotator, source type, and queue context across all entity types." +title: "Scores" +description: "How a score's identity decides when judgements merge and when they don't" --- -## About +## What a score is -A score is the atomic data record created every time an annotation label is applied to a source entity. It is the unified annotation primitive in FutureAGI, replacing the legacy TraceAnnotation model with a single structure that works identically across traces, spans, sessions, dataset rows, prototype runs, and simulation executions. +A **score** is one answer, from a person or a system, to one [label](/docs/annotations/concepts/labels) about one [source](/docs/annotations/concepts/understanding-annotation). It's the single record type behind every judgement in Annotation, whether it came from working a [queue item](/docs/annotations/concepts/queues-and-items) directly or from an inline score submitted against a source. -Every score answers five questions: **what** was annotated (source), **how** it was annotated (label and value), **who** annotated it (annotator), **when** (timestamps), and **why** (optional notes and queue context). +A score carries the annotator's [answer](/docs/annotations/reference/label-types-and-values) to the label, who or what produced it, and where it came from, and it looks the same regardless of which surface wrote it. -## Score fields +## What makes a score unique -| Field | Type | Description | -|-------|------|-------------| -| `id` | UUID | Unique identifier for the score. | -| `label_id` | UUID | The annotation label that was used. Determines the expected value format. | -| `value` | JSON | The annotation value. Format varies by label type (string, number, boolean, string array). | -| `source_type` | string | What kind of entity was annotated. One of the six supported source types. | -| `source_id` | UUID | The ID of the annotated entity (e.g. trace ID, dataset row ID). | -| `annotator` | string | Who created the annotation -- a user email or system identifier. | -| `score_source` | string | Origin of the score: `human` (manual annotation), `model` (LLM-generated), or `auto` (rule-based). | -| `notes` | string | Optional free-text notes attached to the annotation. Available when **Allow Notes** is enabled on the label. | -| `queue_item` | UUID | Optional. Links the score to a specific queue item if it was created through the queue workflow. Null for inline annotations. | -| `created_at` | datetime | When the score was created. | -| `updated_at` | datetime | When the score was last modified. | +What decides whether a new judgement becomes its own score or lands on an existing one is a key made of four fields. Change any one of them and you get a different score, not an update to an old one. -## Source types +- **Source**: the trace or item the score is about +- **Label**: the label being answered +- **Annotator**: the person account that submitted it, whether they worked a queue item or scored it inline +- **Queue item**: the queue item the score was submitted through, if any -Scores can target any of the following entity types. The `source_type` and `source_id` fields together form a polymorphic foreign key to the annotated entity. +This key applies to scores that have an annotator; a score written without one keys differently: see [Scores with no annotator](#scores-with-no-annotator) below. The field names above are conceptual: for their exact names in the API and SDK, see [Data models](/docs/sdk/annotation-queues/data-models). -| Source Type | Entity | Where it appears | -|-------------|--------|-----------------| -| `trace` | An LLM trace from Observe | Trace detail view, LLM Tracing grid | -| `observation_span` | A specific span within a trace | Span detail within trace tree | -| `trace_session` | A conversation session (group of traces) | Sessions grid and session detail | -| `dataset_row` | A row in a dataset | Dataset table view | -| `call_execution` | A simulation call execution | Simulation results view | -| `prototype_run` | A prototype/experiment run | Prototype results view | +Say `priya@yourteam.com` scores a trace's **Response quality** label. Two cases show what the key decides. -## Two ways to create scores +**Two queues, two scores.** Priya works the trace's item in the `Support quality review` queue and scores it. That's Score A. The same trace later lands in a second queue, `Escalation audit`, and she scores it again with the same label. That's Score B. Source, label, and annotator match between A and B, but the queue item doesn't: `Support quality review` on A, `Escalation audit` on B. Two submissions, two independent scores, neither overwrites the other. The same person's judgement through two different queues is two separate pieces of evidence, not a correction. -### Queue workflow (managed) +**Two writes, one queue item.** Now say Priya instead scores that same trace on Response quality inline, then does it again the same way. Both writes resolve to the same queue item, the source's default one, so all four fields match this time: source, label, annotator, and queue item are identical between the two writes. That's Score C: the second write lands on it and updates it, instead of creating a new one. -Scores created through an annotation queue are linked to a queue item via the `queue_item` field. The queue manages assignment, progress tracking, and completion logic. + SC1["Score A"] + SRC --> SC2["Score B"] + SRC --> SC3["Score C"] + LBL --> SC1 + LBL --> SC2 + LBL --> SC3 + ANN --> SC1 + ANN --> SC2 + ANN --> SC3 + QI1 --> SC1 + QI2 --> SC2 + QI3 --> SC3 + W2["Second inline write · same key"] -->|merges, old value kept in history| SC3`} /> - - - Select entities (traces, dataset rows, etc.) in their respective views and click **Add to Queue**. Each becomes a queue item. - - - Click **Start Annotating** on the queue detail page. The workspace presents items one at a time with the queue's labels. Each submitted annotation creates a score. - - - When all labels are scored by the required number of annotators, the queue item auto-completes. - - +### Scores with no annotator -### Inline annotation (direct) +A score written without an annotator uses a narrower key: source and label alone, with no queue item and no annotator in it. That key guarantees at most one such score can exist for a given source and label: see [What a score remembers](#what-a-score-remembers) below for what happens when its value changes. -Scores can also be created directly from the detail view of any supported entity -- without going through a queue. Inline annotations have `queue_item` set to null. +## What a score remembers -- **Trace detail**: Open a trace, expand the annotation panel, and apply any label from your organization. -- **Session grid**: Annotation columns appear directly in the sessions table for quick scoring. -- **Dataset view**: Annotate individual rows from the dataset table. +Editing a score's value doesn't erase what was there before. The prior value is appended to the score's history before the new one is written, so the current answer and the trail of what it used to be both live on the same record, visible in the queue's annotation panel. -Inline annotations are ideal for ad-hoc feedback during investigation or review. They produce the same Score records and appear alongside queue-created scores in all views and exports. +Every score also carries a `score_source` tag recording where it came from: `human` or `api`. It's a record of origin you can read back, not a setting you choose. -## Value formats by label type +## A score is its own record -The `value` field in a score is JSON. Its shape depends on the label type: +A score doesn't depend on the queue item it came from. The queue item it names is provenance, a record of which pass through which queue produced it, not something the score needs to keep existing. That's why an inline score and a queue score show up side by side in the same list: they're the same kind of record, and the queue item is optional context on either one. -| Label Type | Value Example | JSON Type | -|------------|--------------|-----------| -| Categorical (single) | `"Positive"` | string | -| Categorical (multi) | `["Relevant", "Accurate"]` | string array | -| Numeric | `7` | number | -| Text | `"Consider rephrasing the second paragraph."` | string | -| Star Rating | `4` | number | -| Thumbs Up/Down | `true` | boolean | +## Why it matters -## Where scores appear +- Scoring the same source in more than one queue never collides, so a support audit and a compliance audit can run over the same traces independently +- Correcting a score is safe: the old value doesn't disappear, it's superseded on the same record +- Every view of a source, and every export, shows one list of scores no matter which surface wrote them -Scores are surfaced everywhere the annotated entity is displayed: +## Keep exploring -- **Trace detail view** -- Annotation panel shows all scores for the trace and its spans. -- **Sessions grid** -- Dynamic annotation columns display score values inline with session data. Filter and sort by annotation values. -- **Dataset table** -- Score values appear as columns alongside dataset row data. -- **Queue detail** -- The items tab shows all scores submitted for each queue item. -- **API** -- Query scores programmatically with filters on source type, label, annotator, and date range. - -## Bidirectional sync - -Scores created through different paths stay synchronized: - -- An annotation submitted on a trace via Observe **automatically** creates a corresponding score visible in any queue containing that trace. -- A score submitted through a queue workflow is **immediately** visible in the trace detail and session grid views. - -This ensures a single source of truth regardless of where the annotation originated. - -## Next Steps - - - - Learn about the five label types that define score value formats. + + + Work a queue and watch each submission become a score - - Understand how queues manage the annotation lifecycle and produce scores. + + Score a source inline, no queue involved - - Walk through the full flow from label creation to submitted scores. + + Pull scores out as a dataset or a file diff --git a/src/pages/docs/annotations/concepts/understanding-annotation.mdx b/src/pages/docs/annotations/concepts/understanding-annotation.mdx new file mode 100644 index 00000000..07553513 --- /dev/null +++ b/src/pages/docs/annotations/concepts/understanding-annotation.mdx @@ -0,0 +1,67 @@ +--- +title: "Understanding Annotation" +description: "One shared model so every judgement, however it's made, lands as a comparable score" +--- + +## Label, queue, item, score + +**Annotation** is how a person turns their judgement on a piece of AI output (a rating, a category, a correction) into a record Future AGI can compare against every other judgement made the same way. An annotator, someone on your team, works through a queue, or judges a source directly. Either path ends at the same four objects: a label, a queue, an item, and a score. + +A [label](/docs/annotations/concepts/labels) is a reusable question with a fixed answer type: text, a number, a category, a star rating, thumbs up or down. Define it once and reuse it wherever you want that same question answered. + +A [queue](/docs/annotations/concepts/queues-and-items) attaches one or more labels and holds the items waiting to be judged against them. A queue can't exist with zero labels; the labels are what an annotator sees when they open an item. + +Each item a queue holds points at exactly one **source**: + +- A [trace](/docs/observe/concepts/traces) +- A [span](/docs/observe/concepts/spans) +- A [session](/docs/observe/concepts/sessions) +- A Simulation (a simulated voice or text call) +- A prototype run (a prompt run over a dataset) +- A dataset row + +There's no such thing as an item pointing at two sources, or none. + +Every answer to a label, whether it came from working an item or from judging a source directly, becomes one [score](/docs/annotations/concepts/scores). + +## How the four pieces fit + +|attached to| Q["Queue · one or more labels"] + Q -->|holds| I["Item"] + SRC{{"Source · exactly one of:
trace, span, session,
Simulation, prototype run,
dataset row"}} + I -->|is about| SRC + I -->|produces| SC["Score"] + SC -->|points at| SRC + L -->|shapes the answer on| SC`} /> + +### One trace, judged two ways + +Take a trace from the `support-agent` project. Route it into the `Support quality review` queue, and it becomes an item whose one source is that trace. An annotator opens the item, answers the queue's `Response quality` label, and that answer lands as a score referencing the trace, the label, and the item. + +The same trace is judged directly with no queue picked, in the trace's own view in Observe, and someone answers `Response quality` on it there. That judgement doesn't skip the item: the inline judgement lands as the same kind of score record, pointing at the trace and the label. Both show up wherever the trace's judgements are read back. + +## Why it matters + +- An item points at exactly one source, so anything reading the score back never has to guess which of six possible sources it was about +- A judgement made inline is never a lesser record: it resolves to a queue item just like a queue-worked one, so filtering, exporting, or displaying scores never has to special-case where they came from +- A label defined once and attached wherever it's needed means the same question, `Response quality` in a support queue and in a compliance queue, produces answers that land in one comparable set of scores + +## Keep exploring + + + + Answer types, from a star rating to free text + + + Roles, statuses, and how an item gets to complete + + + The uniqueness grain behind every judgement + + + Attach labels, add items, and activate it for annotators + + diff --git a/src/pages/docs/annotations/features/add-items.mdx b/src/pages/docs/annotations/features/add-items.mdx deleted file mode 100644 index a779842e..00000000 --- a/src/pages/docs/annotations/features/add-items.mdx +++ /dev/null @@ -1,91 +0,0 @@ ---- -title: "Add Items to Annotation Queues" -description: "Add traces, spans, sessions, dataset rows, prototype runs, and simulation calls to annotation queues for structured human review in Future AGI." ---- - -## About - -Items are the bridge between your data and your annotation workflow. Each item links a source -- a trace, span, session, dataset row, prototype run, or simulation call -- to a queue. When you add items to a queue, annotators can review the source content and apply the queue's labels. - -## Supported source types - -| Source Type | Where to find | Description | -|-------------|--------------|-------------| -| Trace | Observe > Traces | Full LLM trace with input, output, metadata, latency, tokens, and cost | -| Observation Span | Observe > Trace detail > specific span | An individual span within a trace | -| Session | Observe > Sessions | A conversation session (group of related traces) | -| Dataset Row | Datasets > select dataset | An individual row in a dataset | -| Prototype Run | Prototype > execution history | A prototype experiment run | -| Simulation Call | Simulation > call logs | A simulated voice or text call execution | - -## How to add items from Observe - - - - Go to your **Observe** project and open the **Traces** view. - - - - Use the checkboxes to select one or more traces you want to annotate. - - - - Click the **Add to Queue** button in the toolbar. A dialog opens where you can search for and select the target queue. - - - - Select the queue and click **Add**. The selected traces appear as items in the queue's **Items** tab with a **Pending** status. - - - -## How to add from other sources - -The flow is the same across all source types: - -- **Datasets** -- Navigate to a dataset, select rows using checkboxes, and click **Add to Queue**. -- **Sessions** -- Open Observe > Sessions, select sessions, and click **Add to Queue**. -- **Prototyping** -- Open a prototype's execution history, select runs, and click **Add to Queue**. -- **Simulation** -- Open simulation call logs, select calls, and click **Add to Queue**. - -## Managing items in a queue - -Open a queue's detail page and go to the **Items** tab to see all items and their statuses. - -![Queue items](/images/docs/annotations/queue-detail-items.png) - -### Filtering items - -- **By status** -- Filter by Pending, In Progress, Completed, or Skipped. -- **By source type** -- Show only items from a specific source (e.g., traces only). -- **My Items** -- Toggle to see only items assigned to you. - -### Removing items - -- Select one or more items using checkboxes and click **Remove Selected**. -- Or click the `x` button on an individual row to remove a single item. - -### Bulk operations - -Use the select-all checkbox to select all visible items, then apply bulk actions like remove. - - -Duplicate items are silently skipped. If a source is already in the queue, adding it again has no effect. The response shows how many items were added versus how many were duplicates. - - - -For large-scale annotation campaigns, use the SDK to programmatically add items to queues. See the [Python SDK](/docs/annotations/sdk/python) or [JavaScript SDK](/docs/annotations/sdk/javascript) guide. - - -## Next steps - - - - Start annotating items in the annotation workspace. - - - Understand how queue items flow through statuses and assignment. - - - Add items programmatically via the REST API. - - diff --git a/src/pages/docs/annotations/features/analytics.mdx b/src/pages/docs/annotations/features/analytics.mdx deleted file mode 100644 index 350f29ec..00000000 --- a/src/pages/docs/annotations/features/analytics.mdx +++ /dev/null @@ -1,96 +0,0 @@ ---- -title: "Annotation Analytics & IAA Metrics" -description: "Track queue progress, annotator throughput, label distribution, and inter-annotator agreement using Cohen's and Fleiss' Kappa metrics." ---- - -## About - -Every annotation queue includes a built-in analytics dashboard that shows progress, throughput, and quality metrics. Use it to monitor how your annotation campaign is going and to identify issues before they compound. - -## Accessing analytics - -Open a queue and click the **Analytics** tab. - -![Queue analytics](/images/docs/annotations/queue-detail-analytics.png) - -## Overview stats - -The top of the analytics view shows four key numbers at a glance: - -- **Total items** -- Number of items currently in the queue. -- **Completed** -- Number of items that have been fully annotated. -- **Completion rate** -- Percentage of items completed out of the total. -- **Average completions per day** -- Rolling daily throughput across the queue's lifetime. - -## Status breakdown - -A visual bar displays the distribution of item statuses: - -- **Completed** (green) -- All required annotations collected. -- **In Progress** (blue) -- At least one annotation submitted, more required. -- **Pending** (gray) -- No annotations yet. -- **Skipped** (orange) -- Annotator chose to skip the item. - -## Daily throughput chart - -A bar chart showing the number of completions over the last 30 days. Use it to spot trends, identify slowdowns, and measure annotator velocity over time. - -## Annotator performance table - -| Column | Description | -|--------|-------------| -| Annotator | Name and email of the team member | -| Completed | Number of items this annotator has completed | -| Last Active | Timestamp of their most recent annotation | - -## Label distribution - -For each label attached to the queue, the analytics view shows the frequency of each value: - -- **Categorical** -- Option counts (e.g., "Positive: 45, Negative: 23, Neutral: 12"). -- **Numeric / Star** -- Distribution histogram across the value range. -- **Thumbs** -- Up vs. down counts. -- **Text** -- Total annotation count (text values are not aggregated). - -## Inter-Annotator Agreement - -Switch to the **Agreement** tab to see consistency metrics between annotators scoring the same items. - -**Metrics used:** - -- **Cohen's Kappa** -- Used when exactly 2 annotators have scored the same items. -- **Fleiss' Kappa** -- Used when 3 or more annotators have scored the same items. - -The view shows a per-label agreement breakdown so you can pinpoint which labels have the most disagreement. - -**Interpreting Kappa values:** - -| Kappa Value | Interpretation | -|-------------|---------------| -| < 0.20 | Poor | -| 0.21 -- 0.40 | Fair | -| 0.41 -- 0.60 | Moderate | -| 0.61 -- 0.80 | Substantial | -| 0.81 -- 1.00 | Almost perfect | - - -Agreement metrics require `annotations_required` to be set to 2 or more in your queue settings, and at least 2 annotators must have scored the same items for results to appear. - - - -If agreement is low, review your annotation instructions and consider adding clearer guidelines or simplifying label options. Small improvements to instructions often produce large jumps in agreement. - - -## Next steps - - - - Export completed annotations as datasets for fine-tuning or evaluation. - - - Learn the annotation workspace and keyboard shortcuts. - - - Understand queue architecture, assignment modes, and lifecycle. - - diff --git a/src/pages/docs/annotations/features/annotate.mdx b/src/pages/docs/annotations/features/annotate.mdx deleted file mode 100644 index 0922f2dd..00000000 --- a/src/pages/docs/annotations/features/annotate.mdx +++ /dev/null @@ -1,107 +0,0 @@ ---- -title: "Annotate Items in the Workspace" -description: "Use the annotation workspace to label traces, sessions, and datasets with categorical, numeric, star, and thumbs inputs plus keyboard shortcuts." ---- - -## About - -The annotation workspace is where annotators provide feedback on queue items. It presents the source content alongside the queue's labels in a focused, distraction-free view designed for fast, consistent annotation. - -## How to start annotating - - - - Navigate to a queue and click the **Start Annotating** button. You can also go to the queue's **Items** tab and click on any individual item. - - The workspace opens in a dedicated view. - - ![Annotation workspace](/images/docs/annotations/annotate-workspace.png) - - - - The **left panel** (~60% of the screen) displays the source content. What you see depends on the source type: - - | Source Type | What is displayed | - |-------------|-------------------| - | Trace | Full trace tree with expandable spans -- input, output, metadata, latency, tokens, cost | - | Dataset Row | All fields and values from the dataset row | - | Session | Conversation history with expandable individual traces | - | Prototype Run | Prompt, response, and model information | - | Simulation Call | Transcript, analytics, and audio player (for voice calls) | - - - - The **right panel** (~40% of the screen) shows each label as a section with a colored header. Fill in values based on the label type: - - - **Categorical** -- Click a radio button (single-choice) or checkbox (multi-choice). Use number keys `1`--`9` for quick selection. - - **Numeric** -- Drag the slider or type directly in the input field. Values are enforced within the configured min/max bounds. - - **Star** -- Click a star to set the rating. Use number keys `1`--`N` where N is the number of stars. - - **Thumbs Up/Down** -- Click the **Yes** or **No** button. Use key `1` for thumbs up or `2` for thumbs down. - - **Text** -- Type in the text area. A character count is shown. Input is saved with a 300ms debounce. - - - - If the queue's labels have **Allow Notes** enabled, an optional free-text field appears at the bottom of the labels panel. Use it to add context or comments about your annotation. - - - - Click **Submit & Next** or press `Ctrl+Enter` (`Cmd+Enter` on Mac) to save your annotations and advance to the next item. - - - An item is marked as **Completed** when all required labels have been scored. - - If the queue requires multiple annotators, the item stays **In Progress** until the required number of annotators have submitted. - - - -## Keyboard shortcuts - -Use keyboard shortcuts for significantly faster annotation speed. - -| Shortcut | Action | -|----------|--------| -| `Tab` / `Shift+Tab` | Navigate between labels | -| `1`--`9` | Select a categorical option or set a star rating | -| `Ctrl+Enter` / `Cmd+Enter` | Submit and move to next item | -| `S` | Skip current item | -| `←` / `→` | Previous / next item | -| `?` | Toggle keyboard shortcuts help | - - -Keyboard shortcuts can increase annotation speed by 3--5x. Press `?` in the workspace at any time to see all available shortcuts. - - -## Instructions panel - -If the queue creator wrote annotation instructions, they appear in a collapsible section above the labels. Instructions are rendered as markdown and typically include criteria, examples, and edge-case guidance. Review them before starting your first annotation. - -## Skipping items - -Click the **Skip** button in the header or press `S` to skip the current item and move to the next one. Skipped items can be revisited later -- they are not marked as completed. - -## Navigation - -- Use the **Previous** and **Next** buttons in the footer to move between items. -- A position indicator shows your current item (e.g., "5 of 50"). -- The workspace maintains a history of up to 50 visited items for easy back-navigation. -- A progress bar in the header shows overall completion (X of Y completed). - -## Completion - -When all items in the queue have been annotated, a success screen appears with completion statistics. - - -If an item's source has been deleted, the workspace displays a "Source item has been deleted" message. If another annotator has reserved the item, a lock icon is shown and you will be routed to the next available item. - - -## Next steps - - - - Annotate directly from trace detail, session grid, or dataset views without opening a queue. - - - Export annotated data as training or evaluation datasets. - - - View completion rates, annotator activity, and label distributions. - - diff --git a/src/pages/docs/annotations/features/automation.mdx b/src/pages/docs/annotations/features/automation.mdx deleted file mode 100644 index dfe8c83d..00000000 --- a/src/pages/docs/annotations/features/automation.mdx +++ /dev/null @@ -1,73 +0,0 @@ ---- -title: "Annotation Automation Rules" -description: "Create condition-based rules to automatically add matching traces, spans, or sessions to annotation queues without manual curation." ---- - -## About - -Automation rules let you define conditions that automatically trigger actions on queue items -- such as auto-adding items that match certain criteria or pre-filling label values based on span attributes. Instead of manually curating queue contents, you set the rules once and let matching items flow in automatically. - -## How to set up an automation rule - - - - Open a queue and go to the **Rules** tab. - - - - Click the **Create Rule** button. - - - - Fill in the rule configuration: - - - **Name** -- A descriptive rule name so your team knows what it does at a glance. - - **Source Type** -- Which type of items this rule applies to (e.g., traces, spans). - - **Conditions** -- Define match criteria: - - **Field** -- The attribute to evaluate (e.g., span attribute, metric name). - - **Operator** -- The comparison operator (equals, greater than, less than, contains). - - **Value** -- The threshold or match string. - - **Enabled** -- Toggle the rule on or off. - - - - Click **Save**. The rule is now active and will be evaluated when new items are added to the queue. - - - -## Preview and evaluate - -Before enabling a rule in production, use these tools to validate it: - -- **Preview** -- Click the **Preview** button to see which existing queue items would match the conditions without actually triggering any actions. -- **Evaluate** -- The **Evaluate** action tests the rule against current items and shows detailed match results, so you can fine-tune conditions before going live. - -## Example rules - -| Rule Name | Condition | Action | -|-----------|-----------|--------| -| Flag low scores | eval_score < 0.5 | Auto-add to review queue | -| Long responses | token_count > 1000 | Auto-add for quality check | -| Error traces | status = "error" | Auto-add for analysis | - - -Automation rules are evaluated when new items are added to the queue. Existing items can be tested using the Evaluate action but are not retroactively processed unless you trigger evaluation manually. - - - -Automation rules are a powerful feature still being expanded. Check back for new condition types and actions as they become available. - - -## Next steps - - - - Set up the queues that your automation rules feed into. - - - Learn about manual and programmatic ways to add items alongside automation. - - - Monitor the items your rules are adding and track annotation progress. - - diff --git a/src/pages/docs/annotations/features/export.mdx b/src/pages/docs/annotations/features/export.mdx deleted file mode 100644 index 373276d0..00000000 --- a/src/pages/docs/annotations/features/export.mdx +++ /dev/null @@ -1,86 +0,0 @@ ---- -title: "Export Annotations to Dataset or File" -description: "Export completed annotation queue results to a Future AGI dataset or download as JSON/CSV for fine-tuning, evaluation, and offline analysis." ---- - -## About - -Export lets you turn annotation results from a queue into a structured dataset you can use for fine-tuning, evaluation, or offline analysis. You can export directly into a FutureAGI dataset or download as JSON/CSV. - -## Export to Dataset - - - - Open queue detail and click the **Export to Dataset** button in the header. - - - - Create a **new dataset** by entering a name, or select an **existing dataset** from the dropdown. - - - - Optionally filter by item status. By default, only completed items are included. - - - - Click **Export**. The annotations are written as rows in the target dataset with all label values as columns. - - - -## Export as JSON/CSV - - - - Open queue detail and click the **Export** button. Choose your format -- **JSON** or **CSV**. - - - - Optionally filter by item status to include only the records you need. - - - - Click **Download**. The file is generated and saved to your local machine. - - - -## Export data structure - -Each exported record contains the following fields: - -| Field | Description | -|-------|-------------| -| item_id | Queue item ID | -| source_type | Type of annotated source (trace, span, session, etc.) | -| source_id | ID of the annotated entity | -| status | Item status (completed, skipped, etc.) | -| annotations | Array of label values with annotator info | -| notes | Annotator notes (if any) | - -## When to use exported data - -- **Fine-tuning** -- Use annotated traces as training data for model improvement. -- **Evaluation datasets** -- Create golden datasets for automated eval pipelines. -- **Quality reports** -- Analyze annotation patterns and model failure modes offline. -- **Model comparison** -- Compare model outputs across annotated dimensions. - - -Export to Dataset creates a full FutureAGI dataset that you can use with all dataset features including experiments, evaluations, and prompt management. - - - -For programmatic export, use the [Queues API](/docs/api/annotations/queues/export) or the [SDK export methods](/docs/annotations/sdk/python). - - -## Next steps - - - - Review annotation progress and agreement before exporting. - - - Learn about FutureAGI datasets and what you can do with exported data. - - - Export annotations programmatically via the REST API. - - diff --git a/src/pages/docs/annotations/features/inline.mdx b/src/pages/docs/annotations/features/inline.mdx deleted file mode 100644 index a64920ec..00000000 --- a/src/pages/docs/annotations/features/inline.mdx +++ /dev/null @@ -1,88 +0,0 @@ ---- -title: "Inline Annotations: Ad-Hoc Feedback" -description: "Score traces, sessions, dataset rows, and prototype runs directly from their detail views without setting up an annotation queue." ---- - -## About - -Inline annotations let you score any trace, session, or prototype execution directly from its detail view -- no queue setup required. The InlineAnnotator component appears in the right sidebar of every detail drawer, so you can leave feedback the moment you spot something interesting. - -Best for one-off feedback, quick quality checks, or ad-hoc labeling during debugging. - -## How to annotate inline from Observe - - - - Go to your Observe project and click any trace to open the detail drawer. - - - - Click the **Annotations** tab in the right panel. - - - - Click the **Annotate** button to enter edit mode. - - - - Select labels and provide values. The input types are the same as queue-based annotation -- categorical, numeric, text, star, or thumbs up/down. - - - - Optionally add free-text notes to provide extra context for your annotation. - - - - Click **Save** to store your annotations. They are immediately visible to your team and available via the API. - - - -## From Sessions - -Same flow -- open a session, switch to the **Annotations** tab, and click **Annotate**. Session-level annotations are tracked separately from individual trace annotations within the session. - -## From Prototyping - -Open a prototype execution, then click into the trace detail drawer. The **Annotations** tab is available in the right panel -- click **Annotate** to score the execution. - -## From Simulation Call Logs - -Open a call log detail. The **Annotations** tab appears in the right section of the detail view -- click **Annotate** to score the call. - -## Adding new labels inline - -You can create labels without leaving the annotation sidebar: - -- Click the **Add Label** button in the annotation sidebar. -- Create a new label or select from your existing labels. -- The label immediately appears in the annotation form, ready to use. - -## Inline vs Queue-based - -| Feature | Inline | Queue-based | -|---------|--------|-------------| -| Best for | Quick one-off annotations | Structured campaigns | -| Setup required | None | Create queue, add items | -| Assignment | Self-serve | Manual, Round Robin, Load Balanced | -| Progress tracking | Per-score only | Full queue progress + analytics | -| Multi-annotator | Manual coordination | Built-in agreement metrics | -| Export | Individual scores | Bulk export to dataset | -| Keyboard shortcuts | No | Yes (full shortcut support) | - - -Use inline annotations for quick feedback during debugging. Switch to queues when you need structured annotation campaigns with progress tracking and inter-annotator agreement. - - -## Next steps - - - - Create the labels you'll use for inline annotation. - - - Set up queues for structured annotation campaigns. - - - Understand how scores unify inline and queue-based annotations. - - diff --git a/src/pages/docs/annotations/features/labels.mdx b/src/pages/docs/annotations/features/labels.mdx deleted file mode 100644 index fe710e97..00000000 --- a/src/pages/docs/annotations/features/labels.mdx +++ /dev/null @@ -1,132 +0,0 @@ ---- -title: "Annotation Labels: 5 Types Explained" -description: "Create and configure annotation labels: categorical, numeric, text, star rating, and thumbs up/down. Reusable across all queues in your organization." ---- - -## About - -An annotation label is a reusable template that defines what feedback annotators provide. Labels are organization-scoped: once created, any queue in your workspace can use them. This keeps annotation criteria consistent across teams and projects. - -Each label has a type that determines the UI control annotators see and the value format stored in the resulting score. - ---- - -## Label Types - -| Type | Description | Example Use Case | Value Format | -|------|-------------|------------------|--------------| -| **Categorical** | Predefined list of options. Supports single-choice or multi-choice. Can be used for auto-annotation. | Sentiment analysis: Positive, Negative, Neutral | `string` (single) or `string[]` (multi) | -| **Numeric** | A number within a defined range. | Relevance score from 1 to 10 | `number` | -| **Text** | Free-form text input for open-ended feedback. | Grammar corrections or rewrite suggestions | `string` | -| **Star Rating** | Visual star selector for quick quality ratings. | Overall response quality | `number` (1 to N) | -| **Thumbs Up/Down** | Binary pass/fail toggle. The fastest annotation type. | Helpfulness check: was this answer useful? | `boolean` | - -### Which type should I use? - -| Scenario | Recommended Type | Why | -|----------|-----------------|-----| -| Classify responses into fixed categories | **Categorical** | Predefined options ensure consistency and enable aggregation | -| Rate quality on a fine-grained scale | **Numeric** | Continuous range captures nuance that categories miss | -| Collect corrections, rewrites, or explanations | **Text** | Free-form input gives annotators maximum flexibility | -| Quick quality gut-check (1-5 stars) | **Star Rating** | Visual stars are fast and intuitive for subjective quality | -| Binary accept/reject decisions | **Thumbs Up/Down** | Fastest annotation type: one click per item | -| Multiple dimensions per item | Combine multiple labels in one queue | Attach several labels to a single queue for multi-dimensional annotation | - -### UI appearance by type - -| Type | Annotator UI | -|------|-------------| -| Categorical (single) | Radio buttons for each option | -| Categorical (multi) | Checkboxes for each option | -| Numeric | Number input with stepper or slider | -| Text | Multi-line text area | -| Star Rating | Clickable star icons | -| Thumbs Up/Down | Thumb up and thumb down buttons | - ---- - -## Creating a Label - - - - Go to **Annotations** in the left sidebar, then open the **Labels** tab. Click **Create Label**. - - ![Labels list](/images/docs/annotations/labels-list.png) - - - - Fill in the **Name** field (required) and an optional **Description** to help annotators understand the label's purpose. - - - - Choose the label **Type**. This cannot be changed after creation, so choose carefully. - - - - Each type has its own configuration options: - - - **Categorical**: Add at least two options. Toggle **Allow multiple selection** if annotators should be able to pick more than one option. - - **Numeric**: Set **Min**, **Max**, and **Step size** values. Choose the display format: **Slider** or **Buttons**. - - **Text**: Set **Placeholder text**, **Min character length**, and **Max character length**. - - **Star**: Set the **Number of stars** (1-10, default 5). - - **Thumbs Up/Down**: No additional settings needed. - - ![Create label form](/images/docs/annotations/create-label-categorical.png) - - - - Toggle **Allow Notes** if you want annotators to add free-text commentary alongside their label value. Notes are stored in the `notes` field of the resulting score and are available in exports and the API. - - - - Click **Save**. The label is now available for use in any queue. - - - ---- - -## Managing Labels - -| Action | How | -|---|---| -| Edit | Click a label row or use the menu and select **Edit**. You can change the name, description, and type-specific settings, but the type itself is immutable. | -| Duplicate | Use the menu and select **Duplicate**. Creates a copy you can customize. | -| Archive | Use the menu and select **Archive**. Soft-deletes the label. Archived labels can be restored. | -| Search | Use the search bar at the top to filter labels by name. | -| Filter by type | Use the type dropdown to show only labels of a specific type. | - ---- - -## Label Type Settings Reference - -| Type | Settings | Default | -|------|----------|---------| -| Categorical | `options` (list), `multi_choice` (bool) | `multi_choice`: false | -| Numeric | `min`, `max`, `step_size` | 0, 10, 1 | -| Text | `placeholder`, `min_length`, `max_length` | empty string, 0, 5000 | -| Star | `no_of_stars` | 5 | -| Thumbs Up/Down | none | none | - - -Labels are shared across your entire organization. Any queue can use any label, and changes to a label's settings apply everywhere the label is used. Deleting a label does not remove existing scores that were created with it. - - - -Start with a few simple labels (e.g. a 5-star quality rating and a categorical sentiment label) before creating complex ones. You can always duplicate and customize later. - - ---- - -## Next Steps - - - - Set up annotation queues that use your labels. - - - Learn how to use labels in the annotation workspace. - - - How label values are stored as scores and queried via the API. - - diff --git a/src/pages/docs/annotations/features/queues.mdx b/src/pages/docs/annotations/features/queues.mdx deleted file mode 100644 index 3bd1c067..00000000 --- a/src/pages/docs/annotations/features/queues.mdx +++ /dev/null @@ -1,168 +0,0 @@ ---- -title: "Annotation Queues: Setup & Management" -description: "Create annotation queues with round-robin or load-balanced assignment, multi-annotator support, reservation timeouts, and review workflows." ---- - -## About - -An annotation queue is a managed campaign that groups items to annotate, assigns them to annotators, tracks progress, and enforces quality controls. Queues sit between labels (what to measure) and scores (the resulting data), providing the operational layer that turns annotation from an ad-hoc activity into a structured workflow. - ---- - -## Creating a Queue - - - - Go to **Annotations** in the left sidebar, then open the **Queues** tab. - - ![Queues list](/images/docs/annotations/queues-list.png) - - - - Click the **Create Queue** button to open the creation form. - - - - Fill in the **Name** field (required) and an optional **Description** to help your team understand the queue's purpose. - - - - Select which annotation labels annotators will use when reviewing items in this queue. You can add as many labels as needed. - - ![Create queue](/images/docs/annotations/create-queue.png) - - - - Select workspace members who will annotate items. Only selected members can access and annotate items in this queue. - - - - | Setting | Options | Default | - |---------|---------|---------| - | Annotations Required | 1-10 annotators per item | 1 | - | Assignment Strategy | Manual, Round Robin, Load Balanced | Manual | - | Reservation Timeout | 15 min, 30 min, 1 hour, 4 hours | 30 min | - | Require Review | On / Off | Off | - - - - Write markdown-formatted guidelines for annotators. These appear in a collapsible panel in the annotation workspace. Use guidelines to define criteria, provide examples of correct/incorrect annotations, specify when to skip, and link to reference material. - - - - Click **Save**. The queue is created in **Draft** status. Add items and review settings before activating it. - - - ---- - -## Assignment Strategies - -| Strategy | Behavior | Best For | -|----------|----------|----------| -| **Manual** | Annotators browse and pick items themselves from the queue list. | Small queues or exploratory annotation where annotators need context to choose. | -| **Round Robin** | Items are distributed cyclically across annotators in rotation. | Even distribution when annotators work at similar speeds. | -| **Load Balanced** | Items are distributed based on each annotator's current workload. | Teams with varying availability or part-time annotators. | - ---- - -## Multi-Annotator Support - -For tasks that benefit from agreement between multiple reviewers, set the **Annotations Required** field (1-10). - -- Each item must receive the configured number of complete annotations before it transitions to **Completed**. -- Different annotators independently annotate the same item. They do not see each other's responses. -- The queue analytics tab shows inter-annotator agreement metrics once multiple annotators have scored the same items. - - -An item is considered fully annotated by a single annotator only when all labels attached to the queue have been scored. Partial submissions are saved but do not count toward the required annotation count. - - ---- - -## Reservation System - -When an annotator opens an item, the system reserves it for a configurable timeout period. This prevents two annotators from working on the same item simultaneously. - -- **Default timeout**: 30 minutes -- **Configurable range**: 15 minutes to 4 hours -- **Expiry behavior**: If the annotator does not submit or skip within the timeout, the reservation expires and the item returns to **Pending** for another annotator - ---- - -## Review Workflow - -Enable **Requires Review** on a queue to add a review step after annotation: - -1. Annotators complete their work as usual. When all required annotations are submitted, the item moves to **Pending Review** instead of **Completed**. -2. A designated reviewer opens the item, sees all submitted annotations, and either **Approves** (moves to Completed) or **Rejects** (sends back to Pending for re-annotation). - -This is useful for high-stakes labeling tasks where a senior reviewer must validate annotations before they become final. - ---- - -## Queue Lifecycle - -| Status | Description | Can transition to | -|--------|-------------|-------------------| -| Draft | Queue is being set up, not yet accepting annotations | Active | -| Active | Annotators can annotate items | Paused, Completed | -| Paused | Temporarily stopped, no new annotations allowed | Active, Completed | -| Completed | All items done or manually completed | Active (re-open) | - -### Activating a queue - -A newly created queue starts in **Draft**. To begin accepting annotations, use the menu and select **Activate**, or open the queue detail page and change the status. - -### Auto-completion - -Items auto-complete when: -1. All labels attached to the queue have been scored for the item -2. The required number of annotators have each fully annotated the item -3. If **Requires Review** is enabled, the reviewer has approved the item - - -When a completed queue receives new items, it automatically transitions back to **Active** so annotators can continue. - - ---- - -## Item Statuses - -| Status | Meaning | -|--------|---------| -| **Pending** | Waiting for an annotator to pick it up | -| **In Progress** | An annotator has opened the item and is actively annotating | -| **Completed** | All required annotations have been submitted | -| **Skipped** | An annotator chose to skip this item. It remains available for others. | -| **Pending Review** | Annotations are done but awaiting reviewer approval (when review workflow is enabled) | - ---- - -## Managing Queues - -| Action | How | -|---|---| -| Edit | Open the queue detail page, use the **Settings** tab to modify name, labels, annotators, or workflow settings | -| Duplicate | Use the menu and select **Duplicate**. Creates a copy in Draft status. | -| Archive | Use the menu and select **Archive**. Soft-deletes the queue. | -| Search and filter | Use the search bar to filter by name and the status dropdown to filter by queue status | - ---- - -## Next Steps - - - - Populate your queue with traces, sessions, dataset rows, and more. - - - Walk through the annotation workspace and keyboard shortcuts. - - - Track progress, annotator performance, and inter-annotator agreement. - - - How annotation values are stored and queried. - - diff --git a/src/pages/docs/annotations/guides/annotate-items.mdx b/src/pages/docs/annotations/guides/annotate-items.mdx new file mode 100644 index 00000000..0481dcac --- /dev/null +++ b/src/pages/docs/annotations/guides/annotate-items.mdx @@ -0,0 +1,81 @@ +--- +title: "Annotate items" +description: "The step-by-step workflow annotators follow inside an annotation queue" +--- + +This is the annotator's view: opening a [queue](/docs/annotations/concepts/queues-and-items), working through its items one at a time, and submitting an answer for every [label](/docs/annotations/concepts/labels) it carries. Everything below plays out inside the annotation workspace itself. + +## Open the workspace + +Get in the same way [Explore a queue](/docs/annotations/guides/explore-queue) describes: click **Start Annotating** (or **Resume Skipped**) from the queue, or open an item directly from the Items tab. Either way lands you in the workspace on a specific item. + +### If you land on a message instead + +Sometimes the workspace hands you back a message instead of an item: + +- **Item Reserved**, when someone else already has that item open. Click **Skip to Next Item** to move on +- **Queue Not Active**, if the queue isn't active. A manager has to reactivate it, see Explore a queue for where +- **Assigned to {`{name}`}**, if the item belongs to someone else, a queue manager controls that assignment. Click **Skip to Next Item** to move on +- **All Done!**, once there's nothing left assigned to you + +## Read the source + +The workspace splits into two resizable panes: the source on the left, the labels on the right. Drag the divider between them if you want more room for either side. The left pane renders whatever the item points to, a trace, a span, a session, a dataset row, a prototype run, or a voice call, so you can judge it before answering anything on the right. + +## Answer the labels + +If the queue's creator wrote instructions, they sit in a collapsible section above the labels, open by default so you see them on your first item. Once you know the guidance, collapse it to get it out of the way. + +The right pane lists every label the queue carries under a **Labels** heading. What each label's control looks like depends on its type. Labels covers what each type is for and [Label types & values](/docs/annotations/reference/label-types-and-values) has the exact constraints. + +You can't submit until every label has an answer. Leave one blank and the submit button stays disabled; nothing more happens until you try to submit. Press Cmd/Ctrl+Enter and a reminder names exactly which labels are still missing. + +Numeric and text answers are also checked against the label's configured bounds. Even if the control lets a value slip past, the check still runs again on the server when you submit. Anything out of range gets rejected; bring it back within the label's min, max, step, or length and submit again. + +## Add a note + +Below the labels, an optional **Notes** field lets you leave free-text context on the item ("Add notes for this item..."). It's your own commentary alongside the labels, not a label itself. + +## Submit + +The submit button's text tells you what submitting will actually do: + +- **Submit & Next** is the default: save your answers and move to the next item +- **Submit for Review** shows instead when the queue requires reviewer approval, your answers go to a reviewer before the item counts as done +- **Update & Next** shows when you're revising an answer you already submitted + +If a reviewer sends an item back with feedback, it shows up in the header's **Comments** button and as a **Reviewer feedback** alert at the top of the right pane. See [Review submissions](/docs/annotations/guides/review-submissions). + +## Skip an item + +Click **Skip** in the header, or press **S**. A skipped item isn't marked complete, it just steps out of your way so you can come back to it later. Skip is disabled once an item is completed or already pending review. On a completed item the button's tooltip flips to explain why; on a pending-review item it's just greyed out. + +## Move between items + +Move through the queue with these controls: + +- **Previous** and **Next** in the footer, alongside a position indicator (`n / total`) +- **Back to Queue** in the header, which takes you out of the workspace and back to the queue's detail view +- **Show completed**, which toggles whether items you've already finished are included as you move through the list + +## Keyboard shortcuts + +| Key | Action | +|---|---| +| Tab / Shift+Tab | Move between labels | +| 1-9 | Quick-select a categorical option or star rating | +| Cmd/Ctrl+Enter | Submit and move to the next item | +| S | Skip the current item | +| ← / → | Previous / next item | +| ? | Toggle this shortcuts overlay | + +## Dive deeper + + + + What happens to your answers once a reviewer looks at them + + + Score a single item on the spot, no queue required + + diff --git a/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx b/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx new file mode 100644 index 00000000..d3a1063c --- /dev/null +++ b/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx @@ -0,0 +1,40 @@ +--- +title: "Annotate without a queue" +description: "Score a source on the spot, right from its own view, with no queue involved" +--- + +Not every [score](/docs/annotations/concepts/scores) needs a [queue](/docs/annotations/concepts/queues-and-items) behind it. If you already have a [source](/docs/annotations/concepts/understanding-annotation) open somewhere in Future AGI, a trace, span, dataset row, or prompt run, and something's worth flagging, score it right there instead of setting one up. + +## Open the source and find the Annotations tab + +Open the source's detail view and go to the **Annotations** tab. It lists whatever's already been scored on that source. + +## Switch to edit mode and pick your labels + +Click **Annotate** to switch into edit mode, then pick the labels that apply and fill in a value for each. The labels on offer are the ones set up for that workspace. See [Labels](/docs/annotations/concepts/labels) for how labels get organized. + +## Add a note, if a label supports it + +Not every label carries a note field, only the ones set up for it. Where one is available, it sits right under the label as its own box, already visible, no click needed. Type into it to add context that doesn't fit into the value itself. + +## Save to record the score + +Click **Save**. This writes a score exactly like one submitted through a queue, minus the queue-item link. It shows up wherever the source appears and in exports, right next to any queue-based scores on the same source. + +## When to reach for a queue instead + +A score saved on the spot is right for a one-off: you noticed something while looking at a source and want it on record. It stops being enough once more than one person needs to score the same batch of sources, or you need to see how far through that batch you've gotten. That's what a queue is for: it organizes who scores what, and whether a reviewer has to approve before it counts as done. [Create a queue](/docs/annotations/guides/create-queue) walks through setting one up. + +## Dive deeper + + + + The identity a score carries, in or out of a queue + + + Set one up when scoring turns into a coordinated campaign + + + Write the same scores from the API instead of the UI + + diff --git a/src/pages/docs/annotations/guides/create-label.mdx b/src/pages/docs/annotations/guides/create-label.mdx new file mode 100644 index 00000000..842e3049 --- /dev/null +++ b/src/pages/docs/annotations/guides/create-label.mdx @@ -0,0 +1,70 @@ +--- +title: "Create a label" +description: "Pick a type, configure its settings, and save a label ready to attach to a queue" +--- + +This walks through creating a label: a categorical label called `Response quality` with the options `Good`, `Needs work`, and `Wrong`. A [label](/docs/annotations/concepts/labels) needs a name, a type, and that type's settings. + +## Open the Labels tab + +Labels aren't tied to a project. Go to **Annotations** in the sidebar and switch to the **Labels** tab. Click **Create Label** to open the form. + +If this is your first label, the tab shows an empty state instead of a table, "No labels created yet." The same **Create Label** button sits there too. + +## Name it and pick a type + +Fill in **Name** and an optional **Description**. Those greyed-out examples in the Name field, `Relevance`, `Tone`, `Accuracy`, are placeholder text, not a fixed list to pick from, so `Response quality` fits just as well. + +Then pick a **Type**: + +- **Categorical**: a predefined set of options to choose from +- **Numeric**: a score within a range +- **Text**: free-text feedback +- **Star Rating**: a star-based rating +- **Thumbs Up/Down**: binary feedback + +Not sure which fits? Labels covers what each type is for and when to reach for it. Pick **Categorical** for `Response quality`. + +The Create Label form with Categorical selected as the type, showing the Name, Description, Type, and Options fields +*The Create Label form, with Categorical selected as the type* + +Building one of the other four types instead? [Label types & values](/docs/annotations/reference/label-types-and-values) lists every setting, its default, and its validation rule by type. + + +The type is locked once you save. Editing a label later lets you change its name, description, and settings, but not its type, so get this one right before you save. + + +## Configure the settings and save + +Type-specific settings appear once you've picked a type. For Categorical, that's a list of options: add `Good`, `Needs work`, and `Wrong`. + + +A categorical label needs at least two distinct, non-empty options. Duplicates aren't allowed either, `Good` and `good` count as the same option. The form doesn't stop you from entering duplicates: clicking **Create** fails with an error toast, so fix the options and click **Create** again. + + +Check **Allow notes** if you want annotators to attach free-text commentary alongside their `Response quality` value. Leave it unchecked for a label that should stay a single, quick choice. + +Click **Create**. `Response quality` is now available to attach to a [queue](/docs/annotations/concepts/queues-and-items) or score an item [inline](/docs/annotations/guides/annotate-without-a-queue) (on the spot, with no queue involved). + +## Managing labels + +The same tab handles the rest, once you have labels to manage: + +- **Edit** a label to change its name, description, or settings +- **Duplicate** a label to open the form pre-filled with its settings under a new name, so you can adjust and save it as a separate label +- **Archive** a label to take it out of active use, and **Restore** it from the **Archived** view when you need it back +- **Search** narrows the list by name, and the **type** filter shows only labels of one type + +## Dive deeper + + + + Score an item inline, on the spot, with no queue involved + + + Every settings key, default, and validation rule by type + + + Attach your new label to a queue and start collecting scores + + diff --git a/src/pages/docs/annotations/guides/create-queue.mdx b/src/pages/docs/annotations/guides/create-queue.mdx new file mode 100644 index 00000000..753ff222 --- /dev/null +++ b/src/pages/docs/annotations/guides/create-queue.mdx @@ -0,0 +1,77 @@ +--- +title: "Create a queue" +description: "Fill out the create-queue drawer field by field, then activate the queue so annotators can start" +--- + +A [queue](/docs/annotations/concepts/queues-and-items) attaches one or more [labels](/docs/annotations/concepts/labels) to a set of items, adds the people who'll answer them, and sets how many independent submissions each item needs. This walks through the create-queue drawer in the order it presents its fields, building one running example: `Support quality review`, carrying the `Response quality` label with two submissions per item. + + +- You need at least one label before you open this drawer. Queues won't save without one. Build `Response quality` first if you haven't; see [Create a label](/docs/annotations/guides/create-label) +- This example needs one other workspace member picked as an annotator alongside you, since submissions per item can't exceed the number of annotators picked there +- Invite them from [User Management](/docs/admin-settings/user-management) if they're not in your workspace yet +- Working solo? Leave your own row as it is and set submissions per item to 1 + + +## Open the drawer + +Go to **Annotations** in the sidebar, switch to the **Queues** tab, and click **Create Queue** to open the drawer. A queue can belong to a project, a dataset, or an agent definition, or to none of them at all, an org-level queue. This drawer doesn't ask you to pick one, so `Support quality review` comes out org-level. + +## Name it and describe it + +**Queue Name** is required and free text: type `Support quality review`. The name has to be unique, case-insensitively, among the queues you haven't archived in that same scope, so a name already used there is rejected. + +**Description** is optional, a line or two on the queue's purpose. + +## Attach labels + +Pick the labels annotators will answer for every item in this queue. At least one is required. The drawer won't let you save without it. Attach `Response quality`. + +## Add annotators + +Add the workspace members who'll work this queue as annotators. The picker opens with your own row already selected as Annotator, Reviewer, and Manager, tagged (creator), so you already count toward submissions per item. Untick Annotator on your row if you don't want to answer items yourself. For `Support quality review`, add one more member: you and them make two. + +**Auto-assign items to all annotators** is a checkbox alongside the picker. Turn it on and every annotator is assigned to every item, so anyone can open anything. Leave it off and assignment is manual instead, which means someone has to hand items out (covered in [Add items](/docs/annotations/guides/explore-queue/add-items)). + +## Set submissions per item + +**Submissions per item** is how many different annotators have to answer each item before it's done. It can't exceed the number of annotators you've added. Set it to `2` for `Support quality review`, so every item needs two independent takes on `Response quality` before it's complete. + +## Write instructions + +**Instructions** is a free-text field, markdown supported, that shows up in the annotation workspace itself. Optional, but worth using for queue-wide guidance that applies across every item and label in the queue, rather than notes on a single label. For `Support quality review`, something like: + +```markdown +Rate the agent's final response, not the whole conversation. + +- **Good**: fully answers the question and matches our tone guidelines +- **Needs work**: correct but incomplete, robotic, or missing context +- **Wrong**: factually incorrect or answers a different question than the one asked +``` + +## Open Advanced settings + +Advanced settings is collapsed by default and holds three fields: + +- **Assignment strategy**: Manual is the only one you can pick today. Round Robin and Load Balanced are visible with a **Coming soon** chip, disabled +- **Reservation timeout**: how long an item stays reserved for the annotator who opened it, one of 15 minutes, 30 minutes, 1 hour (the default), or 4 hours. When it expires, the item is released back to the queue for another annotator to pick up +- **Require reviewer approval**: a gated feature that needs the review workflow entitlement. Turn it on and a fully annotated item lands in review instead of finishing outright, which [Review submissions](/docs/annotations/guides/review-submissions) covers; without the entitlement, turning it on fails with an upgrade prompt + +## Save, then activate it + +Click **Create annotation queue**. It saves in Draft, and Draft can only move to Active, nowhere else. Annotating doesn't start until you make that move: open the queue's row menu and click **Activate**. + +Activating doesn't add any items either: the queue starts empty, add them next. From Active, a queue can move to Paused and on to Completed; see [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the exact transitions. + +## Dive deeper + + + + Orient yourself in the queue you just created + + + Hand-pick items or add them by filter from the Items tab + + + Every field, status, role, and cap the queue carries + + diff --git a/src/pages/docs/annotations/guides/explore-queue/add-items.mdx b/src/pages/docs/annotations/guides/explore-queue/add-items.mdx new file mode 100644 index 00000000..54b993a3 --- /dev/null +++ b/src/pages/docs/annotations/guides/explore-queue/add-items.mdx @@ -0,0 +1,70 @@ +--- +title: "Add items" +description: "Pull sources into a queue and hand them to the right people" +--- + +A [queue](/docs/annotations/concepts/queues-and-items)'s **Items** tab is empty until you put something in it. This guide covers getting items into `Support quality review`, the running example from [Create a queue](/docs/annotations/guides/create-queue), then filtering and assigning them once they're there. Adding and assigning are both manager work: if you're not a manager on the queue, you won't see these controls at all. + +## Add items + +Open `Support quality review` and switch to the **Items** tab. A queue with nothing in it shows **No items in this queue** with its own **Add Items** button. Once the queue has items, that button moves into the toolbar above the item table, next to the item filters covered below. Either way, click **Add Items** to open the picker: two ways to fill the queue, hand-pick specific items or set a filter and add everything it matches. Hand-pick when you already know the exact items you want; use filter mode when you want everything matching a rule, however many that turns out to be. + +You can also push items from the source instead of pulling them from the queue. Select rows in a [traces](/docs/observe) table, then **Actions > Add to annotation queue** opens a popover listing your queues: search for one, pick it, or create a new queue on the spot. + +### Hand-pick items + +1. Choose where the items come from: **From Datasets**, **From Traces**, **From Spans**, **From Sessions**, or **From Simulation** +2. Use the checkbox column to select the specific rows you want +3. Click **Add to queue** + +### Add by filter + +1. Choose a source the same way as hand-picking +2. Set the filters that describe what you want, using the filter controls above that source's table +3. Instead of checking rows one by one, tick the header checkbox, then click **Select all N matching your filter** in the banner that appears above the table +4. Click **Add to queue**. For traces, spans, sessions, and simulation, the queue resolves everything the filter matches on the server at that moment, not just what's loaded on screen + + +For a hand-picked selection, the picker splits a large add into multiple requests for you. Filter mode selections are capped at 10,000 items: past that, the whole add is rejected and you're asked to narrow the filter first. Call the endpoint yourself instead of using the picker, and an enumerated list over 1,000 items is rejected outright with an HTTP 413. Full caps and gated behavior live on [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits). + + +Two things worth knowing about what happens once items are added: + +- **Duplicates are skipped, not errored.** If a source is already in the queue, or shows up twice in what you're adding, it's dropped, and the confirmation names the count, for example "12 items added · 3 already in queue" +- **Assignment can happen automatically.** If the queue has auto-assign turned on, incoming items get an assignee the moment they land. Otherwise they arrive unassigned, and someone has to assign them by hand + + +The preview shown in the Items table is a snapshot captured the moment the item was added, not a live read of the source. Opening the item to annotate it always shows the current source. + + +## Filter and find items + +Above the item table sit the item filters: + +- Item status +- Source +- Review status, shown only when the queue has review turned on + +There's also a **My Items** toggle that narrows the table down to whatever's assigned to you. + +## Assign and remove items + +Select one or more rows and buttons join that same toolbar, scoped to your selection: **Assign Selected** and **Remove Selected**. + +**Assign Selected** opens the Assign Selected Items dialog, and like adding items, it's manager-only. Whoever you check becomes the full set of assignees on the selected items, replacing whatever was there before; check nobody and it clears the assignment instead. You can only check people who are already members of the queue: you can't assign an item to someone who hasn't been added yet. + +**Remove Selected** takes the selected items out of the queue entirely, along with any annotations already submitted on them. There's no undo: getting an item back means adding it again as a new item, starting from zero submissions. + +## Dive deeper + + + + Work through what you just added + + + See how the queue is filling up + + + Every cap and gated feature in one place + + diff --git a/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx b/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx new file mode 100644 index 00000000..c6542b82 --- /dev/null +++ b/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx @@ -0,0 +1,60 @@ +--- +title: "Automate item intake" +description: "Build a rule that automatically feeds a queue with matching items" +--- + +A rule checks new candidates against conditions you set and adds the matches to a [queue](/docs/annotations/concepts/queues-and-items) on its own, so nobody has to go looking for fresh items to hand out. This walks through building one for `Support quality review`, the running example from [Create a queue](/docs/annotations/guides/create-queue), running it once to see what it catches, and what actually happens once you trigger it. Creating, editing, deleting, and running a rule are all manager work: if you're not a manager on the queue, following these steps just gets you an error. If you'd rather add items yourself, see [Add items](/docs/annotations/guides/explore-queue/add-items). + +## Create a rule + +Only queue managers can create or run rules. If you're not one, step 1 below returns "Only queue managers can manage automation rules." instead of opening the form. + +From `Support quality review`'s **Rules** tab, click **Add Rule** in the top right to start a new rule. + +1. Give it a name +2. Choose a **Source type** to say what kind of candidate the rule looks at, one of: + - Dataset Row + - Trace + - Span + - Session + - Simulation +3. Set the **Trigger**, one of: + - **Manually**: the rule never fires on its own, someone has to run it every time + - **Every hour**, **Daily**, **Weekly**, or **Monthly**: puts it on a recurring schedule instead +4. Pick the specific target the rule reads from, one of: + - Dataset + - Project + - Agent Definition + + If the queue is already scoped to a dataset or project, that field is locked and reads "Locked by this queue" +5. Add the conditions that decide which candidates match under **Conditions**. The available fields and operators depend on the Source type you picked +6. Click **Create Rule**. It stays disabled until the rule has a name and a source + +Once a rule exists, click its row to open **Edit Automation Rule** and change its name, source type, conditions, or trigger. Delete is the **x** at the end of the row; it asks for confirmation first. + +## Run it once before you trust it + +Whatever trigger you picked, click **Run Now** on the rule's row in the Rules tab to see what it catches from its source right now, before you let it run unattended. The first run always scans the whole backlog against the conditions, whether it fires by schedule or because you clicked Run Now. After that first run, a scheduled rule only rescans what's new since it last ran, so Run Now on a rule that's already fired once is a smaller check, not a full rescan; a Manually-triggered rule has no schedule to fall back on, so every run stays a full rescan. Run Now stays disabled until the rule is enabled with the **Enabled** switch in the rule's row; its tooltip reads "Enable this rule before running it". Click **Run Now** again while a run is still going and it's refused with "A run is already in progress for this rule". + +## When a rule runs + +This is the part that trips people up: + +- A **scheduled** rule (Every hour, Daily, Weekly, Monthly) doesn't fire at the exact minute its trigger implies, so treat the trigger as "within about an hour of," not "on the dot" +- **Running a rule yourself** either reports what it added in the toast right away, or shows "We're preparing your data" and finishes in the background +- When a run finishes in the background, the person who ran it, the rule's creator, and every manager on the queue get an email. Scheduled runs don't send it + +## New items don't arrive silently + +Even without that email, annotators find out. Everyone gets a new-item email at most once an hour, plus a daily summary at their own local digest hour, unless they've snoozed notifications. Whatever a rule adds to `Support quality review` reaches annotators on both cadences, so a rule firing while you're not watching still gets to the people who need to work the items. + +## Dive deeper + + + + Work through what the rule and your team add + + + See where the queue's items stand once they're flowing in + + diff --git a/src/pages/docs/annotations/guides/explore-queue/index.mdx b/src/pages/docs/annotations/guides/explore-queue/index.mdx new file mode 100644 index 00000000..f97f0f55 --- /dev/null +++ b/src/pages/docs/annotations/guides/explore-queue/index.mdx @@ -0,0 +1,63 @@ +--- +title: "Explore a queue" +description: "Find your way around a queue's detail view: its header, tabs, and toolbar actions" +--- + +Once a [queue](/docs/annotations/concepts/queues-and-items) has items and annotators in it, this detail view is where the work actually happens. This guide walks that view using `Support quality review`, the queue built earlier, as the example; any queue of your own works the same way. + +These guides all start from a queue you've already created. If you don't have one yet, [Create a queue](/docs/annotations/guides/create-queue) makes the first one. + +## Open the queue + +Go to **Annotations → Queues** and click the `Support quality review` row to open it. + +## Read the header + +The header carries the queue's name, a status badge next to it, progress underneath, and a row of actions on the right, covered below. + +The badge reads **Draft**, **Active**, **Paused**, or **Completed**. It's the one place to check whether the queue is unlocked for annotating before you try to open an item. + +If you have items assigned to you, you get two progress bars: your own progress first, then the queue's overall progress with a breakdown of how many items are pending, in progress, in review, or skipped. + +## The five tabs + +Below the header, the view splits into five tabs: + +- **Items**, the list of items in the queue and where you open one to annotate it +- **Settings**, the queue's configuration, managers only +- **[Analytics](/docs/annotations/guides/explore-queue/progress-and-agreement)**, performance across the queue +- **Agreement**, how consistently annotators score the same items +- **[Rules](/docs/annotations/guides/explore-queue/automate-item-intake)**, automation rules that feed items into the queue, managers only + +**Settings** and **Rules** only show up if you're a manager on the queue. Everyone else sees Items, Analytics, and Agreement. + +## The toolbar + +Four actions live in the header toolbar. Two of them switch labels depending on the queue's state, and which ones you see at all depends on your role: + +- **Activate** shows up for managers only, and only while the queue isn't already active +- **Export** opens a menu with **Download** and **Export to Dataset**. It's grayed out until the queue has items in it +- **Review Items** is for reviewers and managers, and shows up once the queue has items and is active or completed. If the queue doesn't [require review](/docs/annotations/guides/create-queue#open-advanced-settings), this button reads **View Submissions** instead +- **Start Annotating** opens the annotation workspace for anyone who can annotate, picking the next available item for you. Once the queue is complete and has skipped items left, this button reads **Resume Skipped** instead. Opening an item straight from the **Items** tab lands you in that same workspace, just on the item you clicked instead of the next one in line + +## The two rules for annotating an item + +- **You can only annotate while the queue is active, except to resume skipped items once it's completed.** Try to open an item outside that state and you're told to manage the status from the Settings tab first. If you're a manager, the toolbar's **Activate** button does the same thing in one click whenever the queue isn't already active; use Settings for any other status change, or if you don't see the button +- **You can only open an item assigned to you, unless auto-assign is on, or you're a manager or reviewer on the queue.** Auto-assign is a checkbox in the queue's Settings tab, the same one you set when you built the queue. For plain annotators, an item assigned to someone else stays closed even if you can see it in the list + +## Dive deeper + + + + Open the workspace and work through the labels + + + Put more items in front of your annotators + + + Read the Analytics and Agreement tabs + + + Set up rules so items feed into the queue on their own + + diff --git a/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx b/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx new file mode 100644 index 00000000..3be9ee41 --- /dev/null +++ b/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx @@ -0,0 +1,82 @@ +--- +title: "Track progress & agreement" +description: "Track how a queue's work is coming along, including annotator agreement" +--- + +The **Analytics** and **Agreement** tabs sit after **Items** on a queue's detail view (managers also see **Settings** there). Analytics tells you how the work is going: how much is done, how fast, and who's doing it. Agreement tells you something Analytics can't: whether two people looking at the same item score it the same way. This walks both tabs using `Support quality review`, the queue from [Create a queue](/docs/annotations/guides/create-queue), which carries the `Response quality` label and two submissions required per item. + +## Read the Analytics tab + +Open `Support quality review` and switch to the **Analytics** tab. + +### Headline numbers + +Four cards summarize the queue at a glance: + +- **Total Items**: a raw count of everything in the queue +- **Completed**: a raw count of items that are done +- **Completion Rate**: Completed turned into a percentage of Total Items +- **Avg / Day**: completions averaged over the last 30 days + +### Status breakdown + +Below the headline cards, a bar breaks total items into six buckets, each with its own count: + +- **Completed** +- **In Review** +- **Needs Changes** +- **Resubmitted** +- **Pending Annotation** +- **Skipped** + +Needs Changes and Resubmitted come out of the [review workflow](/docs/annotations/guides/review-submissions): an item only moves through them when the queue requires reviewer approval. In Review holds both: items part-way to the queue's required submissions, and items that have already reached those submissions on a review-enabled queue but are still waiting on a reviewer's verdict. On `Support quality review`, which requires two submissions per item, the first annotator's submission alone puts the item in In Review, no reviewer needed. + +### Throughput over time + +**Daily Throughput (Last 30 Days)** charts completions per day over that same window. Use it to spot a slowdown, or to confirm that adding annotators actually moved the queue faster. + +### Label distribution + +**Label Distribution** shows one card per label, breaking down every value annotators have submitted for it. For `Response quality`, that's a bar for each option, `Good`, `Needs work`, `Wrong`, with a count of how many times annotators picked it. A numeric or star label shows the same idea by rating instead of option, and a thumbs label shows up versus down ([Label types & values](/docs/annotations/reference/label-types-and-values) covers every type). + +### Annotator performance + +**Annotator Performance** lists everyone with activity in the queue. Completed counts items that have met the queue's required submissions across every required label, so it credits every annotator whose submission contributed to that item, not just the one who finished it. + +## Read the Agreement tab + +Switch to the **Agreement** tab. + + +Agreement only has something to compare once two different annotators have actually scored the same item, not just been assigned to it. That means the queue's submissions-per-item setting has to be above its default of 1 ([Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) covers where that lives), with submissions from two or more people on the same item. `Support quality review` is already set to 2, so its Agreement tab fills in as soon as a second annotator submits on the same item. + +Agreement is also a gated feature that needs the Agreement Metrics entitlement: without it, the tab comes up empty instead of loading. + + +### Overall agreement + +At the top, **Overall Agreement** shows one percentage: the share of item/label pairs where every annotator who scored it landed on the same value. Until the precondition above is met, the number reads N/A, with "Need at least 2 annotators per item to calculate agreement" underneath it. + +### Per-label agreement + +**Per-Label Agreement** breaks that number down one row per label. Agreement here is the same raw percentage, scoped to that one label; Disagreements counts how many items its annotators didn't match on. + +Cohen's Kappa is the only agreement statistic the tab reports, whether two annotators scored an item or five. It isn't swapped for a different multi-rater statistic once a third annotator joins. Kappa corrects the raw Agreement percentage for how often annotators would land on the same value purely by chance, so it usually reads lower, and more honestly, than Agreement alone, especially on a label with few options. As a rough guide: below 0.20 is poor agreement, 0.21–0.40 fair, 0.41–0.60 moderate, 0.61–0.80 substantial, and above 0.80 almost perfect. Kappa only computes for categorical, numeric, star, and thumbs labels; a free-text label shows a dash instead. + +### Annotator pair agreement + +As soon as any single pair of annotators has overlapping work, **Annotator Pair Agreement** lists that pair with their agreement percentage and a Comparisons count: the number of item/label comparisons they share, not items, so one item with three labels counts as three. It's the fastest way to tell an annotator who disagrees with everyone else apart from a label that's just genuinely hard to agree on. + +## Dive deeper + + + + Where submissions per item and every other queue field lives + + + Turn completed annotations into a dataset or a file + + + Keep the queue fed so Analytics has something to track + + diff --git a/src/pages/docs/annotations/guides/export-annotations.mdx b/src/pages/docs/annotations/guides/export-annotations.mdx new file mode 100644 index 00000000..ca7a7990 --- /dev/null +++ b/src/pages/docs/annotations/guides/export-annotations.mdx @@ -0,0 +1,64 @@ +--- +title: "Export annotations" +description: "Download a queue's results as a file, or write them into a Future AGI dataset" +--- + +`Support quality review`, the [queue](/docs/annotations/concepts/queues-and-items) from [Create a queue](/docs/annotations/guides/create-queue), now has completed items worth keeping. This guide walks through both routes, starting with the quick download and ending with a write into a [dataset](/docs/dataset). + +## Download as JSON or CSV + +Open `Support quality review` and click **Export > Download** in the header. Every item in the queue, any status, downloads immediately as a JSON file: one entry per item, carrying its annotations, review status, and the source's own content resolved onto it. + + +The endpoint behind Download also accepts a CSV format, which flattens item, review, and annotation fields to one row per label value. It drops `source`, `evals`, `source_id`, `item_notes`, and `annotation_metrics`. There's no format picker in the UI for it yet, so pull CSV directly through the API if you need rows instead of nested JSON; see [SDK & API](/docs/annotations/reference/sdk-api). + + + +Download tops out at 1,000 items. Push past that and it returns an error instead of a file, since there's no status filter on this button to narrow the set first. For a queue that big, use Export to Dataset instead, which carries no such cap, or filter by status through the API. Self-hosted deployments can raise the ceiling with the `ANNOTATION_EXPORT_SYNC_MAX` Django setting. + + +### What's in an export + +| Field | What it holds | +|---|---| +| `item_id` | The queue item's ID | +| `source_type` | trace, observation_span, trace_session, prototype_run, call_execution, or dataset_row | +| `source_id` | ID of the annotated source | +| `status` | pending, in_progress, completed, or skipped | +| `order` | The item's position in the queue | +| `review` | Review status and reviewer, filled in once the item's been reviewed | +| `item_notes` | The latest note left on the item | +| `annotations` | Every label value submitted, with the annotator and score source | +| `annotation_metrics` | The item's annotations keyed by label name | +| `evals` | Eval scores already attached to the item's source, if any | +| `source` | The resolved content of the source itself | + +## Export to Dataset + +Click **Export > Export to Dataset** in the header to open the export drawer. + +1. Choose **Create new dataset** and name it, or **Add to existing dataset** and search for one +2. Set **Items to export**: it defaults to Completed only, and can widen to All items, or switch to Pending only or In Progress only +3. Review the column mapping: each source field, label, and review detail maps to a dataset column, and you can rename, add, or drop columns before running +4. Click **Export** + +Unlike Download, Export to Dataset has no item cap, so a queue past 1,000 items still exports in full. + +## What you do with it + +- **Fine-tuning**: the annotated examples become training data for a model update +- **Eval datasets**: completed items become a golden set you run other evals against + +## Dive deeper + + + + What a dataset is and what you can run against one + + + Pull an export, including CSV, straight from the API + + + Every cap and gated feature in one place + + diff --git a/src/pages/docs/annotations/guides/review-submissions.mdx b/src/pages/docs/annotations/guides/review-submissions.mdx new file mode 100644 index 00000000..5f7dc1bd --- /dev/null +++ b/src/pages/docs/annotations/guides/review-submissions.mdx @@ -0,0 +1,64 @@ +--- +title: "Review submissions" +description: "Compare annotators' answers side by side, approve or send items back for changes, and follow the back-and-forth in comment threads" +--- + +This walks through the reviewer's side of a review-gated queue: reading what came in, approving or sending it back, and clearing a stack in bulk. + +`Support quality review`, the [queue](/docs/annotations/concepts/queues-and-items) from [Create a queue](/docs/annotations/guides/create-queue), requires reviewer approval. That changes what happens once an item collects its [`Response quality`](/docs/annotations/concepts/labels) submissions: instead of finishing on its own, a fully annotated item lands in pending review, and the button its annotators see reads **Submit for Review** rather than **Submit & Next** (covered in [Annotate items](/docs/annotations/guides/annotate-items)). + +## Get into review mode + +Reviewer approval is itself a gated feature: an org needs the entitlement before **Require reviewer approval** can be turned on for a queue at all ([Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) has the details). Everything below assumes it's on. + +If you hold both the annotator and reviewer roles on the queue, the annotation workspace carries a **Workspace action** toggle: **Annotate my answers** next to **Review submissions**. The toggle only appears when you hold both roles, so a reviewer without the annotator role, or an annotator without the reviewer role, never sees it. + +If you only hold the reviewer role, open the queue from its list and use the **Review Items** button in the header instead, which opens the review workspace directly on the first item pending review. + +Whether that button reads **Review Items** or **View Submissions** depends on the queue's **Require reviewer approval** setting, not on the entitlement: a queue can have the review-workflow entitlement and still show **View Submissions** if that setting is off. With it off, the workspace is read-only, labelled **View submissions** instead of **Review submissions** in the toggle: you can read what was submitted, but there's nothing to approve or send back. + +## Compare answers side by side + +Opening an item that's pending review puts you in the comparison panel, which lays out every annotator's answer for the item side by side instead of one at a time, so you can weigh `Response quality` from both submissions before deciding. Each answer has its own feedback field for comments scoped to just that answer, and the panel carries the Approve and Request changes actions for the item as a whole. + +## Approve or request changes + +Two actions sit in the comparison panel: **Approve**, which moves the item to completed, and **Request changes**, which sends it back to the annotator. + +- **Approve** takes no note at all, and is disabled the moment you type any feedback into the panel +- **Request changes** stays disabled until you've left a note in the **Whole-item feedback** box or targeted feedback on at least one answer +- **Approve** is also unavailable while an earlier request for changes on the item is still open; it has to be addressed first before Approve is available again + + +You can't review an item you annotated yourself. Approve and Request changes don't render at all on an item you submitted answers for, even if you also hold the reviewer role on the queue. + + +Once the annotator resubmits, the item lands back in pending review the same way it did the first time, so it reappears in your Items tab list for another look. + +## Leave targeted feedback + +When the problem is one specific answer rather than the whole item, click **Feedback** on that annotator's row to open a feedback field scoped to that answer: **Clear** discards what you've typed, **Done** closes it. That's the targeted alternative to a whole-item note when only one annotator's answer needs fixing. Filling in this field, or the whole-item note, is what enables Request changes; clearing it back out is what re-enables Approve. + +## Approve in bulk + +When there's nothing to argue with, you don't have to open every item on its own. From the queue's **Items** tab, select the items you want to clear and click **Approve Selected**. It counts only the items in your selection that are pending review, and it only appears once at least one selected item is. + +## Comment threads + +Every review action (comment, approve, or request changes) drops into a thread scoped to the item or to one specific answer. A thread moves through open, addressed, resolved, and reopened as reviewers and annotators go back and forth. + +You can @mention teammates in a comment, up to 50 per comment. + +## Dive deeper + + + + See how review activity shows up in the queue's analytics + + + Get the approved answers out as a file or a dataset + + + The reviewer role, the review-workflow entitlement, and the caps on comments + + diff --git a/src/pages/docs/annotations/index.mdx b/src/pages/docs/annotations/index.mdx index 77a17818..99358b04 100644 --- a/src/pages/docs/annotations/index.mdx +++ b/src/pages/docs/annotations/index.mdx @@ -1,88 +1,49 @@ --- -title: "Annotations: Human-in-the-Loop Feedback" -description: "Capture human feedback on AI outputs using labels, queues, and scores across traces, spans, sessions, datasets, prototypes, and simulations." +title: "Overview" +description: "The three objects behind every human judgement on your AI's output" --- -## About +## What is Annotation? -Annotations are human labels applied to AI outputs -- traces, spans, sessions, dataset rows, prototype runs, and simulation executions. They capture subjective judgments (sentiment, quality, helpfulness) and factual assessments (correctness, safety, relevance) that automated evals alone cannot provide. +Annotation captures human judgement on AI output and stores every judgement as a score you can filter, export, and turn into a dataset. The judged thing can be a trace, a span, a session, a call execution, a prototype run, or a dataset row. -Human-in-the-loop (HITL) feedback is essential for GenAI systems because: +## Labels, queues, and scores -- **Quality control** -- Catch hallucinations, off-topic responses, and policy violations before they reach users. -- **Feedback loops** -- Route human judgments back into prompt tuning, guardrail configuration, and model selection. -- **Fine-tuning data** -- Build high-quality labeled datasets from production traffic to improve your models. -- **Safety and compliance** -- Document human review for regulated or high-stakes use cases. +Three objects carry the whole model: -## Architecture +- A **[label](/docs/annotations/concepts/labels)** is the question you ask: a reusable definition of what you're judging, with a fixed answer type. `Response quality`, for instance, is categorical with three options: `Good`, `Needs work`, `Wrong` +- A **[queue](/docs/annotations/concepts/queues-and-items)** organises who answers it and on what: a managed campaign that assigns items, the individual pieces of output being judged, to annotators and tracks their progress. Attach `Response quality` to a `Support quality review` queue and every annotator working it answers that same question +- A **[score](/docs/annotations/concepts/scores)** is the answer itself, one record per judgement. An annotator answering `Response quality` on an item in `Support quality review` produces one score: `Good` -Annotations are built on three primitives: +You can produce a score two ways: work an item through a queue, or [annotate it inline](/docs/annotations/guides/annotate-without-a-queue), on the spot, with no queue involved. -| Primitive | Purpose | -|-----------|---------| -| **Labels** | Reusable annotation templates (categorical, numeric, text, star rating, thumbs up/down) shared across your organization. | -| **Queues** | Managed annotation campaigns that assign items to annotators, track progress, and enforce review workflows. | -| **Scores** | The unified data record created each time an annotator (or automation) applies a label to a source. | +## Start here -Labels define *what* you measure. Queues organize *how* the work gets done. Scores store *every individual annotation*. - -## Supported source types - -Annotations can target any of the following entities: - -| Source Type | Description | -|-------------|-------------| -| `trace` | An LLM trace from Observe | -| `observation_span` | A specific span within a trace | -| `trace_session` | A conversation session (group of traces) | -| `dataset_row` | A row in a dataset | -| `call_execution` | A simulation call execution | -| `prototype_run` | A prototype/experiment run | - -## How it works - -The typical annotation workflow follows three steps: - -1. **Define labels** -- Create the annotation templates your team will use (e.g. a "Sentiment" categorical label or a "Quality" star rating). -2. **Set up a queue** -- Build an annotation campaign by choosing labels, adding annotators, and configuring assignment rules. -3. **Annotate and review** -- Add items (traces, dataset rows, etc.) to the queue. Annotators score each item. Reviewers optionally approve results. - -Annotations can also be created **inline** -- directly from any trace, session, or dataset view -- without a queue, for ad-hoc feedback. - -## Key capabilities - -- **5 label types** -- Categorical, numeric, free-text, star rating, and thumbs up/down to cover any feedback need. -- **Managed queues** -- Round-robin, load-balanced, or manual assignment strategies with reservation timeouts. -- **Inline annotations** -- Annotate directly from trace detail, session grid, or dataset views without opening a queue. -- **Multi-annotator support** -- Require 1-10 annotators per item for inter-annotator agreement. -- **Review workflows** -- Route completed items through a reviewer before finalizing. -- **Export to dataset** -- Turn annotated data into training or eval datasets. -- **Python and JS SDK** -- Create labels, manage queues, and submit scores programmatically. - -## Common use cases - -| Use Case | Label Type | Example | -|----------|------------|---------| -| Sentiment classification | Categorical | Positive / Negative / Neutral | -| Factual accuracy | Thumbs up/down | Correct vs. hallucinated | -| Toxicity screening | Categorical | Safe / Borderline / Toxic | -| Response relevance | Numeric (1-10) | How relevant was the answer? | -| Grammar and style | Text | Free-form correction notes | -| Prompt A vs. B comparison | Star rating | Rate each variant 1-5 stars | + + + Stand up an active queue and start collecting judgement + + + Work through a queue as an annotator + + + Score a single item on the spot, no queue involved + + -## Get started +## Concepts - - Create a label, set up a queue, and annotate your first item in 5 minutes. + + The object model end to end: how labels, queues, items, and scores connect - - Understand the five label types and when to use each one. + + The answer types and how to pick one - - Learn how queues organize work with assignment strategies and review workflows. + + Roles, statuses, and how an item gets to complete - - Dive into the unified Score model that powers all annotation data. + + What a score record carries, and when a new one is created instead of an edit diff --git a/src/pages/docs/annotations/quickstart.mdx b/src/pages/docs/annotations/quickstart.mdx deleted file mode 100644 index fcdfe958..00000000 --- a/src/pages/docs/annotations/quickstart.mdx +++ /dev/null @@ -1,94 +0,0 @@ ---- -title: "Annotations Quickstart: Label & Queue" -description: "Create an annotation label, set up a queue, add traces, and annotate your first item in 5 minutes with this Future AGI walkthrough." ---- - -## What you will do - -In this walkthrough you will create an annotation label, set up a queue, add traces to it, and annotate your first item. The entire flow takes about 5 minutes. - - - - Navigate to **Annotations** in the left sidebar, then open the **Labels** tab. Click **Create Label**. - - ![Labels page](/images/docs/annotations/labels-list.png) - - Fill in the form: - - | Field | Value | - |-------|-------| - | Name | `Sentiment` | - | Type | Categorical | - | Options | `Positive`, `Negative`, `Neutral` | - | Allow Notes | Enabled | - - Click **Create** to save. - - ![Create label](/images/docs/annotations/create-label-categorical.png) - - - - Switch to the **Queues** tab and click **Create Queue**. - - | Field | Value | - |-------|-------| - | Name | `Review Queue` | - | Labels | Select the `Sentiment` label you just created | - | Assignment Strategy | Round Robin | - | Annotators | Add yourself | - | Annotations Required | 1 | - - Click **Create** to save the queue. - - ![Create queue](/images/docs/annotations/create-queue.png) - - - - Go to your **Observe** project and open the **LLM Tracing** view. Select one or more traces using the checkboxes, then click the **Add to Queue** button in the toolbar. - - In the dialog, choose **Review Queue** and confirm. The selected traces are now queue items with a **Pending** status. - - - - Go back to **Annotations > Queues** and click on **Review Queue** to open its detail page. Click **Start Annotating**. - - The annotation workspace loads the first pending item. You will see: - - - The trace content on the left. - - The annotation panel on the right with your `Sentiment` label. - - Select an option (e.g. **Positive**), optionally add a note, and click **Submit**. - - ![Annotation workspace](/images/docs/annotations/annotate-workspace.png) - - The workspace automatically advances to the next item. You can also click **Skip** to move past an item you cannot annotate. - - - - Click the **Analytics** tab on the queue detail page to see completion rates, annotator activity, and label distribution. - - ![Analytics](/images/docs/annotations/queue-detail-analytics.png) - - - - -**Keyboard shortcuts** speed up annotation significantly: - -- **Ctrl+Enter** (or Cmd+Enter) -- Submit the current annotation -- **1-9** -- Select a categorical option by its position -- **S** -- Skip the current item - - -## Next Steps - - - - Explore all five label types and their configuration options. - - - Configure assignment strategies, multi-annotator requirements, and review workflows. - - - Understand how annotation data is stored and queried via the Score model. - - diff --git a/src/pages/docs/annotations/reference/label-types-and-values.mdx b/src/pages/docs/annotations/reference/label-types-and-values.mdx new file mode 100644 index 00000000..8cf02e19 --- /dev/null +++ b/src/pages/docs/annotations/reference/label-types-and-values.mdx @@ -0,0 +1,81 @@ +--- +title: "Label types & values" +description: "Settings, validation rules, and the score value shape for each label type" +--- + +A [label](/docs/annotations/concepts/labels) has one of five types. The type fixes the settings it needs and the control an annotator sees. Every submitted answer is written into the label's [score](/docs/annotations/concepts/scores) as JSON, in `Score.value`, and this page shows the shape that value takes for each type, plus the checks re-applied when a value is submitted. For which type to pick, see Labels; for the steps to create one, see [Create a label](/docs/annotations/guides/create-label). + +Every setting listed below is required by the backend; there's no default value for any of them. The prefills shown in each table are what the create drawer fills in for you, not defaults the API falls back to. + + +Every type can also turn on `allow_notes`, which lets the annotator attach a free-text note alongside their answer. The note is stored in the score's `notes` field, separate from `Score.value`. + + +## Categorical + +| Setting | What it does | Constraint | Create drawer prefills | +|---|---|---|---| +| `options` | The list of options the annotator picks from, each an object with a `label` field, for example `[{"label": "Good"}, {"label": "Needs work"}]` | Two or more, each non-empty, and unique once case differences are ignored | Required, no default | +| `multi_choice` | Whether the annotator can pick more than one option | None | `false` (single choice) | + +Creating a categorical label also requires additional auto-annotation settings: `rule_prompt`, `auto_annotate`, and `strategy`. + +The annotator picks from the options you defined, one or several depending on `multi_choice`. The stored value is always an object with a `selected` array of the picked option labels: a single-select answer holds one label, for example `{"selected": ["Good"]}`; a multi-select answer holds more than one, for example `{"selected": ["Good", "Needs work"]}`. + +## Numeric + +| Setting | What it does | Constraint | Create drawer prefills | +|---|---|---|---| +| `min` | The lowest value on the range | 0 or greater | `0` | +| `max` | The highest value on the range | 0 or greater, and greater than `min` | `10` | +| `step_size` | The increment between values the annotator can land on | Greater than 0 | `1` | +| `display_type` | Whether the annotator sees a slider or a row of buttons | `slider` or `button` | `slider` | + +The annotator sees a slider or a row of buttons, depending on `display_type`, stepping from `min` to `max` in `step_size` increments. The chosen value is stored as `{"value": 7.5}`. + +## Text + +| Setting | What it does | Constraint | Create drawer prefills | +|---|---|---|---| +| `placeholder` | Placeholder text shown in the empty field | None | `Enter your feedback...` | +| `min_length` / `max_length` | The minimum and maximum length allowed for the submitted text | `min_length` must be less than `max_length` | `0` / `500` | + +The annotator gets a free-text field showing `placeholder` when empty. The entered value is stored as `{"text": "Needs a citation for the second claim"}`. + +## Star Rating + +| Setting | What it does | Constraint | Create drawer prefills | +|---|---|---|---| +| `no_of_stars` | How many stars the annotator sees | Greater than 0; the create drawer caps it at 10, though the backend has no upper bound | `5` | + +The annotator sees a row of `no_of_stars` stars to tap. The number of stars picked is stored as `{"rating": 4}`. + +## Thumbs Up/Down + +No settings beyond the type itself. + +The annotator sees a thumbs up / thumbs down toggle. The pick is stored as `{"value": "up"}` or `{"value": "down"}`. + +## Checks applied on submit + +Every value is checked again against the label's settings when it's submitted, not just when the label is created: + +- **Categorical**: every selected option must be one you defined, and if `multi_choice` is off, only one option can be selected +- **Numeric**: the value must fall within `[min, max]`, and must land on a `step_size` increment unless it's exactly `max` +- **Text**: the value's length must fall within `[min_length, max_length]` +- **Star Rating**: the value must be a whole number between 1 and `no_of_stars` +- **Thumbs Up/Down**: the value must be up or down + +## Keep exploring + + + + The mental model behind types and options + + + Configure these settings on a real label + + + Where the submitted value ends up + + diff --git a/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx b/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx new file mode 100644 index 00000000..3deb8bf0 --- /dev/null +++ b/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx @@ -0,0 +1,105 @@ +--- +title: "Queue settings & limits" +description: "Every default value, permission, and cap that governs how a queue behaves" +--- + +This is the reference for every field, status, role, and limit that shapes a [queue](/docs/annotations/concepts/queues-and-items) in the annotation workspace. Field names below are the API/SDK payload names; where a field is also a control on the queue's form in the app, both set the same value. + +## Queue fields and their defaults + +| Field | What it holds | Default | +|---|---|---| +| `name` | The queue's name. Must be unique among non-archived queues in its org and scope | required, no default | +| `description` | Free text describing the queue's purpose | empty | +| `instructions` | Markdown guidelines shown to annotators | empty | +| `status` | Workflow state: `draft`, `active`, `paused`, or `completed` (see transitions below) | `draft` | +| `assignment_strategy` | How items get handed to annotators: manual, round robin, or load balanced. Round robin and load balanced are marked "Coming soon" in the app today, only manual is selectable | `manual` | +| `annotations_required` | How many independent annotators must complete an item before it's done | `1` | +| `reservation_timeout_minutes` | How long an opened item stays locked to the annotator who opened it before it's released back to the queue. Options are 15, 30, 60, or 240 minutes | `60` | +| `requires_review` | Whether a completed item needs [reviewer approval](/docs/annotations/guides/review-submissions) before it counts as done. Needs an entitlement, see [Limits, caps, and gated features](#limits-caps-and-gated-features) | `false` | +| `auto_assign` | Whether every queue member can annotate any item without being assigned to it first | `false` | +| `is_default` | Whether this is the queue Future AGI creates automatically for a project, dataset, or agent definition | `false` | +| `project` / `dataset` / `agent_definition` | Which one, if any, the queue is scoped to. A queue is scoped to at most one of the three, or to none for an org-level queue | none | + +A queue's name only has to be unique among **non-archived** queues in its scope, so archiving a queue frees up its name for reuse. The same logic caps default queues: only one **non-archived** default queue can exist per project, per dataset, and per agent definition at a time. + +Each [label](/docs/annotations/reference/label-types-and-values) you attach to a queue carries two settings of its own: `order`, which controls where it appears in the annotation workspace, and `required`, which forces the annotator to fill it in before submitting. Marking a label required needs an entitlement, covered in [Limits, caps, and gated features](#limits-caps-and-gated-features). + +## Queue statuses and the exact permitted transitions + +| Status | Can move to | +|---|---| +| Draft | Active | +| Active | Paused, Completed | +| Paused | Active, Completed | +| Completed | Active, Paused | + +There's no path back to Draft once a queue leaves it, and Completed isn't a dead end: reopening it by moving it to Active or Paused is a normal transition, not a special case. + +### Archive, restore, and hard delete + +Archiving a queue takes it out of the active list. It stops accepting new work, but it can be restored later: everything about it (its items, labels, and annotators) comes back as it was. + +Hard delete is different: it's permanent. It removes the queue and everything attached to it for good, with no way to bring it back. To hard delete a queue, you have to pass `force=true` and type the queue's exact name to confirm, so it can't fire from a stray click or a typo. + +## Item statuses and the six source types + +| Status | Meaning | +|---|---| +| Pending | Waiting for an annotator to pick it up | +| In Progress | An annotator has it open, reserved to them for the queue's `reservation_timeout_minutes` so nobody else can grab it in the meantime. On a queue that requires review, In Progress also covers a submitted item awaiting reviewer approval: its reservation is cleared and it's no longer open to anyone until a reviewer acts on it | +| Completed | All required annotations have been submitted for it (and approved, if the queue requires review) | +| Skipped | An annotator passed on it. It stays available for someone else to pick up | + +An item can come from six sources: + +| Source type | What it points to | +|---|---| +| Dataset row | A row from a dataset | +| Trace | A full trace | +| Span | A single span inside a trace | +| Prototype | A prototype run | +| Simulation | A simulation | +| Session | A trace session | + +## Roles and what each role can do + +| Role | Can do | +|---|---| +| Annotator | Submit annotations on items in the queue | +| Reviewer | Approve or send back submitted annotations, when the queue requires review | +| Manager | Configure the queue: its settings, labels, and annotators | + +Only annotators and managers can actually submit an annotation. Holding the reviewer role by itself doesn't grant that. + +Whoever creates a queue becomes its first manager automatically. Org admins and workspace admins act as managers on every queue in their scope too, without ever being added as a member. + +## Limits, caps, and gated features + +| Limit | Value | +|---|---| +| Items per [Add Items](/docs/annotations/guides/explore-queue/add-items) call | 1,000 | +| Filter-based selection ceiling | 10,000 items | +| Items per synchronous [export](/docs/annotations/guides/export-annotations) | 1,000 | +| Mentions per comment | 50 | +| Emoji reaction length | 16 characters | + + +How many queues your org can have is capped by your plan, not by the product itself, so the number depends on your plan. + + +Two more things need an entitlement: turning on **Requires Review** for a queue, and marking a per-label `required` flag. Both fail with an upgrade prompt if your plan doesn't include them. + +## Keep exploring + + + + The mental model behind these fields + + + Configure these settings on a real queue + + + The label settings this page doesn't cover + + diff --git a/src/pages/docs/annotations/reference/sdk-api.mdx b/src/pages/docs/annotations/reference/sdk-api.mdx new file mode 100644 index 00000000..748df193 --- /dev/null +++ b/src/pages/docs/annotations/reference/sdk-api.mdx @@ -0,0 +1,86 @@ +--- +title: "SDK & API" +description: "Which surface to reach for: the dashboard, the Python SDK, or the REST API" +--- + +## Three ways to work with annotations + +The dashboard is where you set up and run a campaign: build a [queue](/docs/annotations/concepts/queues-and-items), attach [labels](/docs/annotations/concepts/labels), add annotators, and watch it through to completion. + +This page covers the Python SDK and the REST API. The Python SDK's `fi.queues.AnnotationQueue` client covers the queue lifecycle end to end, from a script. It: + +- creates queues +- creates labels +- adds and assigns items +- submits annotations +- reads progress and analytics +- exports + +The REST API covers the same ground, plus every other endpoint the platform exposes. Both surfaces can also score a source directly, without a queue involved at all: a trace, span, session, dataset row, call execution, or prototype run you want to annotate without the queue workflow around it, via `create_score()` in Python or [Create Score](/docs/api/annotations/scores/create-score) over REST. + +## Install and authenticate + +```bash +pip install futureagi +``` + +```python +from fi.queues import AnnotationQueue + +client = AnnotationQueue( + fi_api_key="YOUR_API_KEY", + fi_secret_key="YOUR_SECRET_KEY", +) +``` + +You can also set `FI_API_KEY` and `FI_SECRET_KEY` as environment variables and drop both arguments; the client picks them up automatically. Find both under **Settings → API Keys** in the platform. + +## An end-to-end example + +Creating `Support quality review`, pushing two traces into it, checking progress, then pulling the completed results back out: + +```python +queue = client.create(name="Support quality review", instructions="Rate response quality 1-5") + +client.add_items(queue.id, items=[ + {"source_type": "trace", "source_id": "trace_abc123"}, + {"source_type": "trace", "source_id": "trace_def456"}, +]) + +progress = client.get_progress(queue.id) +print(f"{progress.completed} of {progress.total} done") + +results = client.export(queue.id, export_format="json", status="completed") +``` + +## Job to method to endpoint + +Each job below has a Python method and a REST endpoint that do the same thing. Full parameter tables live on the linked SDK pages, not here. + +| Job | Python SDK | REST API | +|---|---|---| +| Create a label | [`create_label()`](/docs/sdk/annotation-queues/labels) | [Create Label](/docs/api/annotations/labels/create-label) | +| Create a queue | [`create()`](/docs/sdk/annotation-queues/queues) | [Create Queue](/docs/api/annotations/queues/create-queue) | +| Add items | [`add_items()`](/docs/sdk/annotation-queues/items) | [Add Items](/docs/api/annotations/items/add-items) | +| Submit annotations for a queue item | [`submit_annotations()`](/docs/sdk/annotation-queues/annotations) | [Submit Annotations](/docs/api/annotations/items/submit-annotations) | +| Score a source directly | [`create_score()`](/docs/sdk/annotation-queues/scores) | [Create Score](/docs/api/annotations/scores/create-score) | +| Read progress | [`get_progress()`](/docs/sdk/annotation-queues/analytics) | [Get Progress](/docs/api/annotations/queues/get-progress) | +| Export | [`export()`](/docs/sdk/annotation-queues/export) | [Export](/docs/api/annotations/queues/export) | + + +The dashboard's caps apply to the SDK and the REST API too, not just the UI: up to 1,000 items per `add_items()` call, and up to 1,000 items per synchronous `export()` call. Go over either and the call errors instead of hanging. See [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits). + + +## Keep exploring + + + + The full method reference, one page per concept + + + Every field, status, role, and cap a queue runs under + + + Every REST endpoint across the platform + + diff --git a/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx b/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx deleted file mode 100644 index 5b84537b..00000000 --- a/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx +++ /dev/null @@ -1,448 +0,0 @@ ---- -title: "Annotation Queues via Python SDK" -description: "Create queues, manage labels, add items, submit annotations, track progress, and export results programmatically with the Future AGI Python SDK." ---- - -Annotation queues let you organize traces, sessions, datasets, and simulation outputs for structured human review. Using the SDK, you can: - -- Create and configure annotation queues programmatically -- Create and manage annotation labels (categorical, text, numeric, star, thumbs up/down) -- Add items from multiple sources (traces, spans, sessions, dataset rows, simulations, prototype runs) -- Submit or import annotations in bulk -- Track progress and inter-annotator agreement -- Export annotated data to datasets - - -All methods that accept `queue_id` also accept `queue_name` as an alternative. Similarly, methods that accept `label_id` also accept `label_name`. The SDK resolves names to IDs automatically. - - ---- - -## Installation - -```bash -pip install futureagi -``` - -## Authentication - -You can find your API key and secret key under **Build > Keys** in the sidebar. - -![API Keys](/images/annotation-queue/apikey.png) - - -```python -from fi.queues import AnnotationQueue - -client = AnnotationQueue( - fi_api_key="YOUR_API_KEY", - fi_secret_key="YOUR_SECRET_KEY", -) -``` - ---- - -## Creating Labels - -Create annotation labels to define what annotators should evaluate. Each label has a type that determines the kind of input annotators provide: - -```python -# Categorical label (multiple choice) -sentiment_label = client.create_label( - name="Sentiment", - type="categorical", - settings={ - "rule_prompt": "Classify the sentiment of the response", - "multi_choice": False, - "options": [ - {"label": "Positive"}, - {"label": "Negative"}, - {"label": "Neutral"}, - ], - "auto_annotate": False, - "strategy": None, - }, -) - -# Numeric label (slider or buttons) -quality_label = client.create_label( - name="Quality Score", - type="numeric", - settings={ - "min": 1, - "max": 10, - "step_size": 1, - "display_type": "slider", - }, -) - -# Thumbs up/down label -thumbs_label = client.create_label( - name="Helpful", - type="thumbs_up_down", - settings={}, -) -``` - -You can also list and retrieve existing labels by ID or name: - -```python -# List all labels -labels = client.list_labels() - -# List labels scoped to a project -project_labels = client.list_labels(project_id="your_project_id") - -# Get a specific label by ID or name -label = client.get_label(label_id="your_label_id") -label = client.get_label(label_name="Sentiment") - -# Delete a label by ID or name -client.delete_label(label_id="your_label_id") -client.delete_label(label_name="Sentiment") -``` - ---- - -## Creating an Annotation Queue - -You can find all your annotation queues under **Observe > Annotations** in the sidebar. - -![Annotation Queues list](/images/annotation-queue/annotationqueue1.png) - - -Create a queue with instructions and configuration for your reviewers: - -```python -queue = client.create( - name="Sentiment Review", - description="Review and label sentiment of customer interactions", - instructions="Rate each trace as positive, negative, or neutral. Consider the overall tone of the conversation.", - assignment_strategy="load_balanced", - annotations_required=2, - reservation_timeout_minutes=60, - requires_review=True, -) -print(f"Created queue: {queue.id} (status: {queue.status})") -``` - -**Assignment strategies:** -- `"manual"` — Explicitly assign items to annotators -- `"round_robin"` — Distribute items evenly across annotators -- `"load_balanced"` — Assign to the annotator with the fewest pending items - ---- - -## Attaching Labels to a Queue - -Once labels are created, attach them to a queue using IDs or names: - -```python -# Using IDs (from create_label return values) -client.add_label(queue.id, label_id=sentiment_label.id) -client.add_label(queue.id, label_id=quality_label.id) - -# Or using names -client.add_label(queue_name="Sentiment Review", label_name="Sentiment") -client.add_label(queue_name="Sentiment Review", label_name="Quality Score") -``` - ---- - -## Activating the Queue - -Click on a queue to view its settings, including the queue name, description, instructions, and attached labels. - -![Queue detail](/images/annotation-queue/annotationqueuedetail1.png) - - -Queues start in `draft` status. Activate when ready for annotation: - -```python -# Using ID -queue = client.activate(queue.id) - -# Or using name -queue = client.activate(queue_name="Sentiment Review") -print(f"Queue status: {queue.status}") # "active" -``` - ---- - -## Adding Items - -Add items from various sources to the queue: - -```python -result = client.add_items(queue.id, items=[ - {"source_type": "trace", "source_id": "trace_uuid_1"}, - {"source_type": "trace", "source_id": "trace_uuid_2"}, - {"source_type": "observation_span", "source_id": "span_uuid_1"}, - {"source_type": "dataset_row", "source_id": "row_uuid_1"}, - {"source_type": "trace_session", "source_id": "session_uuid_1"}, - {"source_type": "call_execution", "source_id": "simulation_uuid_1"}, - {"source_type": "prototype_run", "source_id": "prototype_run_uuid_1"}, -]) -print(f"Added: {result.added}, Duplicates: {result.duplicates}") -``` - -**Supported source types:** `trace`, `observation_span`, `trace_session`, `call_execution`, `prototype_run`, `dataset_row` - ---- - -## Listing and Filtering Items - -```python -# List all pending items -pending_items = client.list_items(queue.id, status="pending") - -# List items assigned to a specific user -assigned_items = client.list_items(queue.id, assigned_to="user_uuid") - -# Paginate through items -page_2 = client.list_items(queue.id, page=2, page_size=20) -``` - ---- - -## Assigning Items - -Manually assign items to annotators: - -```python -# Assign items to a user -client.assign_items( - queue.id, - item_ids=[items[0].id, items[1].id], - user_id="annotator_user_id", -) - -# Unassign items -client.assign_items( - queue.id, - item_ids=[items[0].id], - user_id=None, -) -``` - ---- - -## Submitting Annotations - -In the UI, annotators see each item's content alongside the configured labels and can submit their annotations directly. - -![Queue item](/images/annotation-queue/queueitem1.png) - - -Submit annotations as the authenticated user: - -```python -client.submit_annotations( - queue.id, - item_id=items[0].id, - annotations=[ - {"label_id": "sentiment_label_id", "value": "positive"}, - {"label_id": "confidence_label_id", "value": 0.95}, - ], - notes="Clear positive sentiment throughout the conversation", -) -``` - ---- - -## Importing Annotations Programmatically - -Import annotations from an external source or automated pipeline: - -```python -result = client.import_annotations( - queue.id, - item_id=items[0].id, - annotations=[ - {"label_id": "sentiment_label_id", "value": "positive"}, - {"label_id": "confidence_label_id", "value": 0.92}, - ], - annotator_id="external_annotator_user_id", # optional -) -print(f"Imported: {result.imported}") -``` - ---- - -## Completing and Skipping Items - -```python -# Mark item as completed -client.complete_item(queue.id, item_id=items[0].id) - -# Skip an item -client.skip_item(queue.id, item_id=items[1].id) -``` - ---- - -## Tracking Progress - -```python -progress = client.get_progress(queue.id) -print(f"Total: {progress.total}") -print(f"Completed: {progress.completed}") -print(f"Pending: {progress.pending}") -print(f"Progress: {progress.progress_pct}%") -``` - ---- - -## Analytics and Agreement - -The Analytics tab shows throughput, status breakdown, label distribution, and annotator performance. - -![Queue analytics](/images/annotation-queue/queueanalytics.png) - - -```python -# Get throughput and annotator performance -analytics = client.get_analytics(queue.id) -print(f"Status breakdown: {analytics.status_breakdown}") -print(f"Total completed: {analytics.throughput['total_completed']}") -print(f"Avg per day: {analytics.throughput['avg_per_day']}") - -# Daily throughput (last 30 days) -for day in analytics.throughput["daily"]: - print(f" {day['date']}: {day['count']} completed") - -# Get inter-annotator agreement -agreement = client.get_agreement(queue.id) -print(f"Overall agreement: {agreement.overall_agreement}") -``` - ---- - -## Exporting Results - -### Export as JSON or CSV - -```python -# Export completed annotations as JSON -data = client.export(queue.id, export_format="json", status="completed") - -# Export as CSV -csv_data = client.export(queue.id, export_format="csv", status="completed") -``` - -### Export to a Dataset - -```python -# Create a new dataset from annotations -result = client.export_to_dataset(queue.id, dataset_name="Sentiment Labels") -print(f"Created dataset '{result.dataset_name}' with {result.rows_created} rows") - -# Or append to an existing dataset -result = client.export_to_dataset(queue.id, dataset_id="existing_dataset_uuid") -``` - ---- - -## Using Scores Without a Queue - -You can also annotate any source entity directly using scores, without creating a queue: - -```python -# Create a single score (by label ID or name) -score = client.create_score( - source_type="trace", - source_id="trace_uuid_1", - label_name="Quality Score", - value="good", - score_source="api", - notes="Automated quality check", -) - -# Create multiple scores at once -client.create_scores( - source_type="trace", - source_id="trace_uuid_1", - scores=[ - {"label_id": "quality_label_id", "value": "good"}, - {"label_id": "relevance_label_id", "value": 4.5}, - ], -) - -# Retrieve scores -scores = client.get_scores(source_type="trace", source_id="trace_uuid_1") -for s in scores: - print(f"{s.label_name}: {s.value} (by {s.annotator_name})") -``` - ---- - -## Completing a Queue - -When all items have been reviewed: - -```python -queue = client.complete_queue(queue.id) -print(f"Queue status: {queue.status}") # "completed" -``` - - -Completing a queue does **not** automatically disable its automation rules. If you have active rules, they may continue adding items, which will re-activate the queue. Disable or delete automation rules manually before completing the queue. - - ---- - -## Complete Example - -```python -from fi.queues import AnnotationQueue - -client = AnnotationQueue( - fi_api_key="YOUR_API_KEY", - fi_secret_key="YOUR_SECRET_KEY", -) - -# 1. Create and configure the queue -queue = client.create( - name="Trace Quality Review", - instructions="Rate the quality of each AI response on a scale of 1-5", - assignment_strategy="round_robin", - annotations_required=2, -) - -# 2. Create a label, attach it, and activate -label = client.create_label( - name="Quality", - type="numeric", - settings={"min": 1, "max": 5, "step_size": 1, "display_type": "slider"}, -) -client.add_label(queue.id, label.id) -queue = client.activate(queue.id) - -# 3. Add items (using queue name works too) -result = client.add_items(queue_name="Trace Quality Review", items=[ - {"source_type": "trace", "source_id": "trace_1"}, - {"source_type": "trace", "source_id": "trace_2"}, - {"source_type": "trace", "source_id": "trace_3"}, -]) -print(f"Added {result.added} items") - -# 4. List and annotate items -items = client.list_items(queue.id, status="pending") -for item in items: - client.submit_annotations( - queue.id, - item.id, - annotations=[{"label_id": label.id, "value": 4}], - ) - client.complete_item(queue.id, item.id) - -# 5. Check progress and export -progress = client.get_progress(queue_name="Trace Quality Review") -print(f"Completed: {progress.completed}/{progress.total}") - -export_result = client.export_to_dataset(queue.id, dataset_name="Quality Reviews") -print(f"Exported to dataset: {export_result.dataset_name}") - -# 6. Complete the queue -client.complete_queue(queue.id) -``` diff --git a/src/pages/docs/annotations/sdk/javascript.mdx b/src/pages/docs/annotations/sdk/javascript.mdx deleted file mode 100644 index 61e2bc16..00000000 --- a/src/pages/docs/annotations/sdk/javascript.mdx +++ /dev/null @@ -1,303 +0,0 @@ ---- -title: "Annotations JavaScript & TypeScript SDK" -description: "Log annotations, manage queues, submit scores, and export results using the FutureAGI JavaScript/TypeScript SDK's Annotation and AnnotationQueue classes." ---- - -# JavaScript SDK - -The FutureAGI JavaScript/TypeScript SDK provides two primary classes: `Annotation` for logging annotations via a DataFrame-style interface, and `AnnotationQueue` for full queue lifecycle management. - -## Installation - - - -```bash npm -npm install @future-agi/sdk -``` - -```bash yarn -yarn add @future-agi/sdk -``` - -```bash pnpm -pnpm add @future-agi/sdk -``` - - - ---- - -## Annotation Class -- Log Annotations - -### Initialize the client - -```typescript -import { Annotation } from '@future-agi/sdk'; - -const client = new Annotation({ - fiApiKey: 'YOUR_API_KEY', - fiSecretKey: 'YOUR_SECRET_KEY', -}); -``` - -### Log annotations - -Log annotations using DataFrame-style records. Each record is an object with column keys following the same naming convention as the [Python SDK](/docs/annotations/sdk/python). - -```typescript -const response = await client.logAnnotations([ - { - 'context.span_id': 'span_abc123', - 'annotation.quality.text': 'Excellent response', - 'annotation.sentiment.label': 'positive', - 'annotation.accuracy.score': 9.0, - 'annotation.rating.rating': 5, - 'annotation.helpful.thumbs': true, - 'annotation.notes': 'Top quality', - }, - { - 'context.span_id': 'span_def456', - 'annotation.quality.text': 'Needs improvement', - 'annotation.sentiment.label': 'negative', - 'annotation.accuracy.score': 3.5, - 'annotation.rating.rating': 2, - 'annotation.helpful.thumbs': false, - 'annotation.notes': 'Hallucinated facts', - }, -], { projectName: 'My Project' }); - -console.log(`Created: ${response.annotationsCreated}, Errors: ${response.errorsCount}`); -``` - - -For the full column naming convention table, see the [Python SDK -- Column naming convention](/docs/annotations/sdk/python#column-naming-convention). The format is identical across both SDKs. - - -### Get labels - -```typescript -const labels = await client.getLabels({ projectId: 'proj_123' }); - -labels.forEach(l => console.log(`${l.name} (${l.type}): ${l.id}`)); -``` - -### List projects - -```typescript -const projects = await client.listProjects({ projectType: 'observe' }); - -projects.forEach(p => console.log(`${p.name}: ${p.id}`)); -``` - ---- - -## AnnotationQueue Class -- Full Queue Management - -The `AnnotationQueue` class provides complete programmatic control over the annotation queue lifecycle: creating queues, adding items, assigning work, submitting annotations, and exporting results. - -### Initialize the client - -```typescript -import { AnnotationQueue } from '@future-agi/sdk'; - -const queues = new AnnotationQueue({ - fiApiKey: 'YOUR_API_KEY', - fiSecretKey: 'YOUR_SECRET_KEY', -}); -``` - -### Create a queue - -```typescript -const queue = await queues.create({ - name: 'Review Queue', - description: 'Quality review of traces', - instructions: 'Rate response quality on all labels', - assignmentStrategy: 'round_robin', - annotationsRequired: 2, - reservationTimeoutMinutes: 30, - requiresReview: false, -}); -``` - -### Add items to a queue - -```typescript -const result = await queues.addItems(queue.id, [ - { sourceType: 'trace', sourceId: 'trace_abc' }, - { sourceType: 'observation_span', sourceId: 'span_def' }, - { sourceType: 'dataset_row', sourceId: 'row_ghi' }, -]); - -console.log(`Added: ${result.added}, Duplicates: ${result.duplicates}`); -``` - -#### Valid source types - -| Source Type | Description | -|-------------|-------------| -| `trace` | An LLM trace | -| `observation_span` | A specific span in a trace | -| `trace_session` | A conversation session | -| `dataset_row` | A dataset row | -| `call_execution` | A simulation call | -| `prototype_run` | A prototype run | - -### Submit annotations - -```typescript -await queues.submitAnnotations(queue.id, itemId, [ - { labelId: 'label_123', value: 'positive', scoreSource: 'human' }, - { labelId: 'label_456', value: 4.5, scoreSource: 'human' }, -], { notes: 'High quality response' }); -``` - -### Create scores directly (without queue) - -You can create scores against any source without going through a queue workflow. - -```typescript -const score = await queues.createScore({ - sourceType: 'trace', - sourceId: 'trace_abc', - labelId: 'label_123', - value: { text: 'Good response' }, - scoreSource: 'human', - notes: 'Quick feedback', -}); -``` - -### Bulk create scores - -```typescript -await queues.createScores({ - sourceType: 'trace', - sourceId: 'trace_abc', - scores: [ - { labelId: 'label_123', value: 'positive' }, - { labelId: 'label_456', value: 4.5 }, - ], - notes: 'Batch annotation', -}); -``` - -### Queue lifecycle - -```typescript -// Activate a draft queue -await queues.activate(queue.id); - -// Mark a queue as completed -await queues.completeQueue(queue.id); - -// Add or remove labels from a queue -await queues.addLabel(queue.id, 'label_789'); -await queues.removeLabel(queue.id, 'label_789'); - -// List items with optional status filter -const items = await queues.listItems(queue.id, { status: 'pending' }); - -// Assign items to a specific user -await queues.assignItems(queue.id, ['item_1', 'item_2'], 'user_123'); - -// Complete or skip items -await queues.completeItem(queue.id, 'item_1'); -await queues.skipItem(queue.id, 'item_2'); -``` - -### Progress and analytics - -```typescript -const progress = await queues.getProgress(queue.id); -console.log(`${progress.completed}/${progress.total} (${progress.progressPct}%)`); - -const analytics = await queues.getAnalytics(queue.id); - -const agreement = await queues.getAgreement(queue.id); -``` - -### Export - - - -```typescript JSON export -const data = await queues.export(queue.id, { - format: 'json', - status: 'completed', -}); -``` - -```typescript Export to dataset -const dataset = await queues.exportToDataset(queue.id, { - datasetName: 'Annotated Traces Q1', - statusFilter: 'completed', -}); - -console.log(`Created dataset ${dataset.datasetId} with ${dataset.rowsCreated} rows`); -``` - - - ---- - -## Complete Method Reference - -### AnnotationQueue methods - -| Method | Description | -|--------|-------------| -| `create(config)` | Create a new queue | -| `list(options)` | List queues | -| `get(queueId)` | Get queue details | -| `update(queueId, updates)` | Update queue configuration | -| `delete(queueId)` | Delete a queue | -| `activate(queueId)` | Set queue status to active | -| `completeQueue(queueId)` | Set queue status to completed | -| `addLabel(queueId, labelId)` | Add a label to a queue | -| `removeLabel(queueId, labelId)` | Remove a label from a queue | -| `addItems(queueId, items)` | Add source items to a queue | -| `listItems(queueId, options)` | List queue items with optional filters | -| `removeItems(queueId, itemIds)` | Remove items from a queue | -| `assignItems(queueId, itemIds, userId)` | Assign items to a user | -| `submitAnnotations(queueId, itemId, annotations)` | Submit annotations for an item | -| `getAnnotations(queueId, itemId)` | Get annotations for an item | -| `completeItem(queueId, itemId)` | Mark an item as completed | -| `skipItem(queueId, itemId)` | Skip an item | -| `createScore(options)` | Create a single score (no queue required) | -| `createScores(options)` | Bulk create scores (no queue required) | -| `getScores(sourceType, sourceId)` | Get scores for a source | -| `getProgress(queueId)` | Get queue completion progress | -| `getAnalytics(queueId)` | Get queue analytics and metrics | -| `getAgreement(queueId)` | Get inter-annotator agreement metrics | -| `export(queueId, options)` | Export annotations as JSON or CSV | -| `exportToDataset(queueId, options)` | Export annotations to a FutureAGI dataset | - ---- - -## Best Practices - -- **Use `logAnnotations()` for bulk SDK-based annotation** -- The DataFrame-style format is the fastest way to annotate many spans at once. -- **Use `AnnotationQueue` for programmatic queue management** -- Create, assign, and complete queues entirely from code. -- **Use `createScore()` / `createScores()` for direct score creation** -- Bypass the queue workflow when you need to attach scores to traces directly. -- **Always handle errors** -- Check for partial failures in bulk operations. Both `logAnnotations` and `addItems` can succeed for some records and fail for others. -- **Use TypeScript** -- All SDK methods are fully typed. TypeScript catches column name typos and invalid configurations at compile time. - - -Bulk operations (`logAnnotations`, `addItems`, `createScores`) may partially succeed. Always inspect the response for per-record errors before assuming all records were processed. - - ---- - -## Next steps - - - - DataFrame-based annotation logging with the Python SDK. - - - Query and manage annotation scores via the REST API. - - - REST API reference for queue CRUD operations. - - diff --git a/src/pages/docs/annotations/sdk/python.mdx b/src/pages/docs/annotations/sdk/python.mdx deleted file mode 100644 index 37745faa..00000000 --- a/src/pages/docs/annotations/sdk/python.mdx +++ /dev/null @@ -1,160 +0,0 @@ ---- -title: "Annotations Python SDK: Log & Manage" -description: "Log annotations via DataFrame, retrieve labels, list projects, and submit human feedback to traces using the FutureAGI Python SDK." ---- - -# Python SDK - -The FutureAGI Python SDK provides a simple, DataFrame-based interface for logging annotations against your traces. Install the package, authenticate, and start annotating in minutes. - -## Installation - - - -```bash pip -pip install futureagi -``` - -```bash pip3 -pip3 install futureagi -``` - - - -## Authentication - -```python -from fi.annotations import Annotation - -client = Annotation( - fi_api_key="YOUR_API_KEY", - fi_secret_key="YOUR_SECRET_KEY", -) -``` - - -You can also set `FI_API_KEY` and `FI_SECRET_KEY` as environment variables. The client picks them up automatically when no arguments are passed. - - ---- - -## Log Annotations - -The `log_annotations()` method accepts a pandas DataFrame where each row represents one annotation record. Columns follow the naming convention `annotation..`. - -### Column naming convention - -| Column Pattern | Label Type | Example Value | -|----------------|------------|---------------| -| `annotation..text` | Text | `"good response"` | -| `annotation..label` | Categorical | `"positive"` | -| `annotation..score` | Numeric | `8.5` | -| `annotation..rating` | Star (1-5) | `4` | -| `annotation..thumbs` | Thumbs Up/Down | `True` | -| `annotation.notes` | Notes (shared) | `"Great response!"` | -| `context.span_id` | (required) Span ID | `"span_abc123"` | - - -Every row **must** include a `context.span_id` column. This links the annotation to a specific span in your Observe project. - - -### Full example - -```python -import pandas as pd -from fi.annotations import Annotation - -client = Annotation( - fi_api_key="YOUR_API_KEY", - fi_secret_key="YOUR_SECRET_KEY", -) - -df = pd.DataFrame({ - "context.span_id": ["span_abc123", "span_def456"], - "annotation.quality.text": ["Excellent response", "Needs improvement"], - "annotation.sentiment.label": ["positive", "negative"], - "annotation.accuracy.score": [9.0, 3.5], - "annotation.rating.rating": [5, 2], - "annotation.helpful.thumbs": [True, False], - "annotation.notes": ["Top quality", "Hallucinated facts"], -}) - -response = client.log_annotations(df, project_name="My Project") -print(f"Created: {response.annotations_created}, Errors: {response.errors_count}") -``` - -### Response object - -| Field | Type | Description | -|-------|------|-------------| -| `message` | `str` | Summary message | -| `annotations_created` | `int` | New annotations created | -| `annotations_updated` | `int` | Existing annotations updated | -| `notes_created` | `int` | Notes created | -| `succeeded_count` | `int` | Successful records | -| `errors_count` | `int` | Failed records | -| `errors` | `list` | Error details per failed record | - ---- - -## Get Labels - -Retrieve all annotation labels configured for a project. Use the returned label IDs when constructing your DataFrame columns. - -```python -labels = client.get_labels(project_id="proj_123") - -for label in labels: - print(f"{label.name} ({label.type}): {label.id}") -``` - ---- - -## List Projects - -List all projects accessible to your API key. Filter by project type to find your Observe projects. - -```python -projects = client.list_projects(project_type="observe") - -for p in projects: - print(f"{p.name}: {p.id}") -``` - ---- - -## Annotation Queues - - -For queue management -- creating queues, adding items, submitting annotations, and exporting results -- use the REST API directly or the [JavaScript SDK](/docs/annotations/sdk/javascript) which provides full queue support. See the [Queues API reference](/docs/api/annotations/queues/create-queue) for details. - - ---- - -## Best Practices - -- **Batch annotations** -- Group 100--500 records per DataFrame for optimal throughput. -- **Consistent span IDs** -- Ensure span IDs match traces in your Observe project. Invalid IDs result in per-row errors. -- **Idempotent notes** -- Duplicate notes for the same span are silently skipped. -- **Error handling** -- Always check `response.errors_count` and inspect `response.errors` for partial failures. -- **Label IDs** -- Use `get_labels()` to fetch label names and IDs before constructing your DataFrame. - - -Annotations are immutable once submitted. Double-check your DataFrame before calling `log_annotations()`. - - ---- - -## Next steps - - - - Full queue management, scores, and annotation support in JavaScript/TypeScript. - - - Query and manage annotation scores via the REST API. - - - Upload annotations in bulk using the REST API directly. - - diff --git a/src/pages/docs/annotations/troubleshooting.mdx b/src/pages/docs/annotations/troubleshooting.mdx new file mode 100644 index 00000000..1d16e0d0 --- /dev/null +++ b/src/pages/docs/annotations/troubleshooting.mdx @@ -0,0 +1,71 @@ +--- +title: "Annotation FAQ & fixes" +description: "Common annotation questions, and fixes for the errors you hit most" +--- + +## In this page + +The questions people ask most about annotation, and the errors they run into, with a direct fix for each. Hit an error? Jump straight to [Common errors and fixes](#common-errors-and-fixes). If your answer isn't here, reach out via [support](https://futureagi.com/contact-us). + +## Common errors and fixes + +| Symptom | Cause | Fix | +|---|---|---| +| The submit button won't enable | Every label attached to the queue needs an answer before you can submit | Answer every label, submit stays disabled until all have values; pressing Ctrl+Enter while any are empty lists which ones are still open | +| An item says it's reserved by someone else | Another annotator already has the item open | Skip to Next Item, or come back once the other annotator submits or skips it | +| You can't annotate because the queue isn't active | The queue is in draft or paused | Ask a queue manager to switch it to active from the Settings tab | +| An item is assigned to someone else | Auto-assign is off, and the item was assigned to another annotator | Ask a queue manager to reassign it to you | +| Skip is refused on an item | The item is already completed, or it's pending review on a queue that requires review | Completed items can't be skipped; an item waiting on review has to clear review first | +| An Add Items call is rejected | The payload has more than 1,000 items, so the API returns HTTP 413 | Split the items into batches of 1,000 or fewer, see [Add items](/docs/annotations/guides/explore-queue/add-items) | +| A filter-mode selection is rejected | The filter resolves to more than 10,000 items | Narrow the filter, or add items in smaller batches | +| An export of a large queue fails | A synchronous export refuses queues with more than 1,000 items outright rather than truncating them, returning HTTP 413 | Narrow the filter so it resolves to 1,000 items or fewer, see [Export annotations](/docs/annotations/guides/export-annotations) | +| A numeric or text value is rejected on submit | The value falls outside the label's configured min, max, step size, or length | Match the value to the label's settings, see [Label types & values](/docs/annotations/reference/label-types-and-values) | +| A queue name is rejected as already taken | Another queue in the same scope already uses that name | Choose a different name | +| A hard delete refuses to go through | Hard delete needs the queue's exact name typed as confirmation, plus a force flag on the API | Type the queue's exact name to confirm, the Delete forever button in the dialog stays disabled until it matches; via the API, pass `force=true` with the exact name | + +## Roles and permissions + +**Who can annotate, and who can review?** + +Annotating a queue needs the annotator or manager role on it; without one of those roles, submitting is refused. Org and workspace admins get manager-level access automatically, without being added to the queue explicitly. Reviewing has its own role, see [Review submissions](/docs/annotations/guides/review-submissions) for how it works, and [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the full roles table. + +**Why don't I see the Settings or Rules tab?** + +Both are manager-only surfaces. If you're not a manager on the queue, and not an org or workspace admin, they stay hidden. + +## Scores + +**Why does the same trace show two scores from the same person?** + +Score the same trace from two different queues and you get two independent [scores](/docs/annotations/concepts/scores), not one overwritten value. + +**If I edit a score, do I lose the old value?** + +No. Changing a score's value appends to its history instead of overwriting it. Previous values show in the annotation history panel on the item, listed as Previous 1, Previous 2, and so on. + +## Queues + +**Why is the Agreement tab empty?** + +Agreement measures how much annotators agree, so it has nothing to compare until more than one independent submission lands on the same items. See [Track progress & agreement](/docs/annotations/guides/explore-queue/progress-and-agreement) for what the queue needs to populate it. + +**What happens to items when a queue is archived?** + +The items stay in the queue. Archiving switches the queues list to Archived, and any rules attached to it pause. Restore it from the Archived view and it comes back in the status it had when you archived it. + +## Keep exploring + + + + The operational model behind a queue and its items + + + Why a score outlives the queue item that created it + + + The settings and validation rules behind every label type + + + Fields, statuses, roles, and the hard caps on a queue + + diff --git a/src/pages/docs/cookbook/decrease-hallucination.mdx b/src/pages/docs/cookbook/decrease-hallucination.mdx index 3727b475..8d3fe329 100644 --- a/src/pages/docs/cookbook/decrease-hallucination.mdx +++ b/src/pages/docs/cookbook/decrease-hallucination.mdx @@ -238,7 +238,7 @@ To quantify performance of each combination of RAG setup, a set of evals accordi - **`criteria`**: Description of the criteria for evaluation - Returns a percentage score, where a high-score Indicate that the context is relevant or sufficient to produce an accurate and coherent output. - Click [here](/docs/prototype/features/evals) to learn more about the evals provided by Future AGI + Click [here](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI The `eval_tags` list contains multiple instances of `EvalTag`. Each `EvalTag` represents a specific evaluation configuration to be applied during runtime, encapsulating all necessary parameters for the evaluation process. @@ -253,13 +253,13 @@ Parameters of `EvalTag` : - For Context Adherence Eval, `EvalName.CONTEXT_ADHERENCE`, - For Context Retrieval Quality,`EvalName.EVAL_CONTEXT_RETRIEVAL_QUALITY` - Click [here](/docs/prototype/features/evals) to get complete list of evals provided by Future AGI + Click [here](/docs/evaluation/builtin) to get complete list of evals provided by Future AGI - **`config`**: Dictionary for providing specific configurations for the evaluation. An empty dictionary `{}` means that default configuration parameters will be used. - Click [here](/docs/prototype/features/evals) to learn more about what config is required for corresponding evals + Click [here](/docs/evaluation/builtin) to learn more about what config is required for corresponding evals - **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation. - Click [here](/docs/prototype/features/evals) to learn more about what inputs are required for corresponding evals + Click [here](/docs/evaluation/builtin) to learn more about what inputs are required for corresponding evals - **`custom_eval_name`**: A user-defined name for the specific evaluation instance. **7.2 Setting Up Trace Provider** diff --git a/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx b/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx index e3d64954..cc4affb0 100644 --- a/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx +++ b/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx @@ -7,7 +7,7 @@ description: "Build a Google ADK multi-agent system with tracing, then use Futur This cookbook walks through a complete example: build a multi-agent system with Google ADK, instrument it with tracing, and use [Error Feed](/docs/error-feed) to automatically analyze agent performance. By the end, you'll have traces flowing into Observe with Error Feed scores and recommendations visible on each trace. -For a framework-agnostic guide on reading Error Feed results, see [Issue Overview](/docs/error-feed/features/issue-overview). +For a framework-agnostic guide on reading Error Feed results, see [Investigate an issue](/docs/error-feed/guides/investigate-an-issue). --- @@ -236,12 +236,12 @@ Click on a trace to open the trace tree. Error Feed insights appear in a collaps ![Error Feed insights expanded](/images/docs/agent-compass-quickstart/agent_compass_expanded.png) -For details on how to read scores, insights, clusters, and recommendations, see [Issue Overview](/docs/error-feed/features/issue-overview). +For details on how to read scores, insights, clusters, and recommendations, see [Investigate an issue](/docs/error-feed/guides/investigate-an-issue). --- ## Next Steps -- [Issue Overview](/docs/error-feed/features/issue-overview): Understand scores, clusters, and recommendations -- [Error Taxonomy](/docs/error-feed/concepts/taxonomy): Explore all error categories +- [Investigate an issue](/docs/error-feed/guides/investigate-an-issue): Understand scores, clusters, and recommendations +- [Error taxonomy](/docs/error-feed/reference/error-taxonomy): Explore all error categories - [Set Up Observability](/docs/quickstart/setup-observability): Send traces from other frameworks diff --git a/src/pages/docs/cookbook/eval-metrics-optimization.mdx b/src/pages/docs/cookbook/eval-metrics-optimization.mdx index f5eede53..400c7eae 100644 --- a/src/pages/docs/cookbook/eval-metrics-optimization.mdx +++ b/src/pages/docs/cookbook/eval-metrics-optimization.mdx @@ -165,7 +165,7 @@ data_mapper = BasicDataMapper(key_map={"response": "generated_output"}) See a complete end-to-end example of running an optimization. diff --git a/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx b/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx index 1530db76..eb4c0ba2 100644 --- a/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx +++ b/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx @@ -191,7 +191,7 @@ You went from a noisy traced project to a fixed agent and a reusable regression Curate balanced golden datasets from real traces with `/build-dataset` - + All built-in slash commands and how to write your own
diff --git a/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx b/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx index 81735884..de8f7104 100644 --- a/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx +++ b/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx @@ -180,7 +180,7 @@ Production traces, curated and ground-truthed in one Falcon AI conversation, bec From a single bad trace to a paste-ready prompt fix in minutes - + All built-in slash commands and how to write your own
diff --git a/src/pages/docs/cookbook/langchain-langgraph.mdx b/src/pages/docs/cookbook/langchain-langgraph.mdx index 52f1c4dc..2fc52605 100644 --- a/src/pages/docs/cookbook/langchain-langgraph.mdx +++ b/src/pages/docs/cookbook/langchain-langgraph.mdx @@ -116,7 +116,7 @@ Instrumentation of such project requires 3 steps: - **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation. - **`custom_eval_name`**: A user-defined name for the specific evaluation instance. - > Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about the evals provided by Future AGI + > Click [**here**](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI > 2. **Setting Up Trace Provider:** diff --git a/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx b/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx index 3e07374e..1c89fe9e 100644 --- a/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx +++ b/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx @@ -52,7 +52,7 @@ Add to `~/.cursor/mcp.json`: } ``` -Or use the [one-click install link](/docs/quickstart/setup-mcp-server) on the setup page. +Or use the [one-click install link](/docs/falcon-ai/guides/use-the-mcp-server) on the setup page. @@ -180,7 +180,7 @@ You connected Future AGI's MCP server to your IDE, asked natural-language questi ## Explore further - + Full setup reference, OAuth scopes, and supported tool groups diff --git a/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx b/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx index d8361104..83e73baf 100644 --- a/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx +++ b/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx @@ -210,7 +210,7 @@ You can now create annotation views, define labels, assign annotators, and log a ## Next steps - + Full annotation reference diff --git a/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx b/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx index 6f02a5d8..f774249a 100644 --- a/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx +++ b/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx @@ -178,7 +178,7 @@ You can add `elif` branches between `if` and `else` for more granular routing; e -For all six dynamic column types in detail, see [Create Dynamic Column](/docs/dataset/concept/dynamic-column). +For all six dynamic column types in detail, see [Create Dynamic Column](/docs/dataset/concepts/static-and-dynamic-columns). @@ -208,7 +208,7 @@ You can now enrich any dataset with AI-generated columns, vector-retrieved conte A/B test prompts - + Full column type reference diff --git a/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx b/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx index 2be00991..2e1cb5d8 100644 --- a/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx +++ b/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx @@ -286,12 +286,12 @@ This guide uses **MetaPrompt** and **Bayesian Search**, but FutureAGI offers six | Optimizer | Best for | How it works | |---|---|---| -| [**Meta-Prompt**](/docs/optimization/optimizers/meta-prompt) | General prompt improvement | A teacher LLM iteratively rewrites the prompt based on eval feedback | -| [**Bayesian Search**](/docs/optimization/optimizers/bayesian-search) | Few-shot example selection | Uses Bayesian optimization to find the best number and combination of examples | -| [**ProTeGi**](/docs/optimization/optimizers/protegi) | Targeted prompt editing | Generates localized edits to specific parts of the prompt, then tests each | -| [**GEPA**](/docs/optimization/optimizers/gepa) | Exploring diverse prompt styles | Evolutionary approach — breeds, mutates, and selects prompts over generations | -| [**PromptWizard**](/docs/optimization/optimizers/promptwizard) | Multi-stage refinement | Combines critique, refinement, and example synthesis in a structured pipeline | -| [**Random Search**](/docs/optimization/optimizers/random-search) | Quick baseline comparison | Generates random prompt variants and picks the best — useful as a sanity check | +| [**Meta-Prompt**](/docs/optimization/reference/optimizers/meta-prompt) | General prompt improvement | A teacher LLM iteratively rewrites the prompt based on eval feedback | +| [**Bayesian Search**](/docs/optimization/reference/optimizers/bayesian-search) | Few-shot example selection | Uses Bayesian optimization to find the best number and combination of examples | +| [**ProTeGi**](/docs/optimization/reference/optimizers/protegi) | Targeted prompt editing | Generates localized edits to specific parts of the prompt, then tests each | +| [**GEPA**](/docs/optimization/reference/optimizers/gepa) | Exploring diverse prompt styles | Evolutionary approach — breeds, mutates, and selects prompts over generations | +| [**PromptWizard**](/docs/optimization/reference/optimizers/promptwizard) | Multi-stage refinement | Combines critique, refinement, and example synthesis in a structured pipeline | +| [**Random Search**](/docs/optimization/reference/optimizers/random-search) | Quick baseline comparison | Generates random prompt variants and picks the best — useful as a sanity check | Not sure which to pick? Start with **Meta-Prompt** for instruction tuning or **Bayesian Search** for few-shot tasks. See the [Optimizers Overview](/docs/optimization) for a detailed comparison and decision tree. @@ -319,7 +319,7 @@ You can now automatically optimize any prompt using MetaPromptOptimizer or Bayes Version and serve prompts - + Run optimization from the Future AGI UI diff --git a/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx b/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx index 47c49661..6a70b746 100644 --- a/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx +++ b/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx @@ -74,7 +74,7 @@ print(result["messages"]) # "What are your business hours?" ``` -`failed_rule` and `reasons` are always **lists** — even when only one rule triggers. For full details on all return keys, see [Protect API Reference](/docs/protect/concepts/concept). +`failed_rule` and `reasons` are always **lists** — even when only one rule triggers. For full details on all return keys, see [Protect API Reference](/docs/protect/concepts/understanding-protect). @@ -137,7 +137,7 @@ print(result["failed_rule"]) # ["security", "data_privacy_compliance"] print(result["reasons"][0]) # "Detected instruction override attempt..." ``` -The four available metrics are `content_moderation`, `security`, `data_privacy_compliance`, and `bias_detection`. See [Protect How-To](/docs/protect/features/run-protect) for what each metric catches. +The four available metrics are `content_moderation`, `security`, `data_privacy_compliance`, and `bias_detection`. See [Protect How-To](/docs/protect/guides/run-protect-from-the-sdk) for what each metric catches. @@ -232,7 +232,7 @@ print(result["status"]) # "passed" ``` -Use standard Protect for accuracy-critical flows (user-facing chatbots, compliance). Use Protect Flash for high-volume pipelines (batch screening, log analysis). See [Protect vs Protect Flash](/docs/protect/concepts/concept) for a detailed comparison. +Use standard Protect for accuracy-critical flows (user-facing chatbots, compliance). Use Protect Flash for high-volume pipelines (batch screening, log analysis). See [Protect vs Protect Flash](/docs/protect/concepts/understanding-protect) for a detailed comparison. @@ -253,10 +253,10 @@ You can now screen user inputs and AI outputs for prompt injection, PII, toxicit ## Next steps - + All safety metrics - + How Protect works diff --git a/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx b/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx deleted file mode 100644 index 9b29ffa5..00000000 --- a/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx +++ /dev/null @@ -1,316 +0,0 @@ ---- -title: "Prototype and Iterate on LLM Applications" -description: "Register a Prototype project with automatic span evaluation, iterate with versioned prompts, and compare versions before deploying to production." ---- - - -Register a Prototype project with automatic span evaluation, iterate with versioned prompts, compare versions side by side, and choose a winner before deploying to production. - - -
-Open in Colab -GitHub -
- -| Time | Difficulty | Package | -|------|-----------|---------| -| 15 min | Intermediate | `fi-instrumentation-otel` | - - -- FutureAGI account → [app.futureagi.com](https://app.futureagi.com) -- API keys: `FI_API_KEY` and `FI_SECRET_KEY` (see [Get your API keys](/docs/admin-settings)) -- Python 3.9+ -- OpenAI API key - - -## Install - -```bash -pip install fi-instrumentation-otel traceAI-openai openai -``` - -```bash -export FI_API_KEY="your-api-key" -export FI_SECRET_KEY="your-secret-key" -export OPENAI_API_KEY="your-openai-api-key" -``` - ---- - -## What is Prototype? - -Prototype lets you test different LLM configurations, prompts, and parameters in a controlled environment before deploying to production. Each run is a **version**: you compare versions side by side on evaluation scores, cost, and latency, then choose a winner. - -## Tutorial - - - - -`register()` creates a tracer provider connected to FutureAGI. Setting `project_type=ProjectType.EXPERIMENT` creates a Prototype project. The `project_version_name` tags all traces from this run as a distinct version you can compare later. - -`EvalTag` objects define which evaluations run automatically on every matching span, with no manual eval calls needed. - -```python -from fi_instrumentation import register -from fi_instrumentation.fi_types import ( - ProjectType, - EvalName, - EvalTag, - EvalTagType, - EvalSpanKind, - ModelChoices, -) - -trace_provider = register( - project_type=ProjectType.EXPERIMENT, - project_name="support-bot-prototype", - project_version_name="v1-baseline", - eval_tags=[ - EvalTag( - eval_name=EvalName.COMPLETENESS, - type=EvalTagType.OBSERVATION_SPAN, - value=EvalSpanKind.LLM, - model=ModelChoices.TURING_FLASH, - custom_eval_name="completeness_check", - mapping={ - "input": "llm.input_messages.1.message.content", - "output": "llm.output_messages.0.message.content", - }, - ), - EvalTag( - eval_name=EvalName.SUMMARY_QUALITY, - type=EvalTagType.OBSERVATION_SPAN, - value=EvalSpanKind.LLM, - model=ModelChoices.TURING_FLASH, - custom_eval_name="response_quality", - mapping={ - "input": "llm.input_messages.1.message.content", - "output": "llm.output_messages.0.message.content", - }, - ), - ], -) -``` - -Each `EvalTag` has: -- `eval_name`: the built-in evaluation to run (e.g. `EvalName.COMPLETENESS`, `EvalName.SUMMARY_QUALITY`) -- `type`: where to apply the eval (`EvalTagType.OBSERVATION_SPAN`) -- `value`: which span kind to evaluate (`EvalSpanKind.LLM`) -- `mapping`: maps eval input keys to span attribute paths -- `model`: the FutureAGI eval model to use -- `custom_eval_name`: a label for this eval tag (must be unique per project) - - - - -Patch the OpenAI client with `OpenAIInstrumentor` so every API call is automatically traced and evaluated against your `EvalTag` configuration. - -```python -from traceai_openai import OpenAIInstrumentor -from openai import OpenAI - -OpenAIInstrumentor().instrument(tracer_provider=trace_provider) - -client = OpenAI() - -questions = [ - "How do I reset my password?", - "What is your refund policy?", - "Can I upgrade my plan mid-cycle?", -] - -for q in questions: - response = client.chat.completions.create( - model="gpt-4o-mini", - messages=[ - {"role": "system", "content": "You are a helpful customer support agent. Answer concisely."}, - {"role": "user", "content": q}, - ], - ) - print(f"Q: {q}") - print(f"A: {response.choices[0].message.content}\n") - -trace_provider.force_flush() -``` - -Expected output: -``` -Q: How do I reset my password? -A: Go to the login page, click "Forgot Password," enter your email, and follow the reset link sent to your inbox. - -Q: What is your refund policy? -A: We offer full refunds within 30 days of purchase. After 30 days, refunds are prorated. - -Q: Can I upgrade my plan mid-cycle? -A: Yes, you can upgrade anytime. The price difference is prorated for the remainder of your billing cycle. -``` - - - - -Go to [app.futureagi.com](https://app.futureagi.com), select **Prototype** (left sidebar under BUILD), and click your project **support-bot-prototype** to see version **v1-baseline**. - -The dashboard shows: -- Every traced span with its input, output, token count, and latency -- Evaluation scores from your `EvalTag` configuration (`completeness_check` and `response_quality`) displayed alongside each span - - - - -This is where rapid iteration happens. Register a new version with a different `project_version_name` and run the same queries with an improved prompt. Each version is a separate experiment you can compare. - - -Each call to `register()` creates a new tracer provider. Run Version 2 in a separate script or after the Version 1 script completes — do not call `register()` twice in the same process. - - -```python -from fi_instrumentation import register -from fi_instrumentation.fi_types import ( - ProjectType, - EvalName, - EvalTag, - EvalTagType, - EvalSpanKind, - ModelChoices, -) -from traceai_openai import OpenAIInstrumentor -from openai import OpenAI - -trace_provider_v2 = register( - project_type=ProjectType.EXPERIMENT, - project_name="support-bot-prototype", - project_version_name="v2-detailed", - eval_tags=[ - EvalTag( - eval_name=EvalName.COMPLETENESS, - type=EvalTagType.OBSERVATION_SPAN, - value=EvalSpanKind.LLM, - model=ModelChoices.TURING_FLASH, - custom_eval_name="completeness_check", - mapping={ - "input": "llm.input_messages.1.message.content", - "output": "llm.output_messages.0.message.content", - }, - ), - EvalTag( - eval_name=EvalName.SUMMARY_QUALITY, - type=EvalTagType.OBSERVATION_SPAN, - value=EvalSpanKind.LLM, - model=ModelChoices.TURING_FLASH, - custom_eval_name="response_quality", - mapping={ - "input": "llm.input_messages.1.message.content", - "output": "llm.output_messages.0.message.content", - }, - ), - ], -) - -OpenAIInstrumentor().uninstrument() -OpenAIInstrumentor().instrument(tracer_provider=trace_provider_v2) - -client = OpenAI() - -questions = [ - "How do I reset my password?", - "What is your refund policy?", - "Can I upgrade my plan mid-cycle?", -] - -for q in questions: - response = client.chat.completions.create( - model="gpt-4o-mini", - messages=[ - { - "role": "system", - "content": ( - "You are a knowledgeable customer support agent. " - "Provide detailed, step-by-step answers. " - "Include any relevant edge cases or exceptions. " - "End with a follow-up question to confirm the issue is resolved." - ), - }, - {"role": "user", "content": q}, - ], - ) - print(f"Q: {q}") - print(f"A: {response.choices[0].message.content}\n") - -trace_provider_v2.force_flush() -``` - -Expected output: -``` -Q: How do I reset my password? -A: Here's how to reset your password step by step: -1. Go to our login page at app.example.com -2. Click "Forgot Password" below the sign-in button -3. Enter the email address associated with your account -4. Check your inbox for a reset link (check spam if you don't see it within 5 minutes) -5. Click the link and enter your new password - -Note: The reset link expires after 24 hours. If it expires, repeat the process. - -Is there anything else about your account access I can help with? - -Q: What is your refund policy? -... -``` - - - - -Back in the Prototype dashboard, your project now shows two versions: **v1-baseline** and **v2-detailed**. - -Click any version to see its individual traces and eval scores. The project overview shows aggregate metrics across all versions — average eval scores, latency, token usage, and cost — so you can compare at a glance. - - - - -Once you have compared evaluation scores, latency, and cost across versions, choose a winner. - -1. Go to **Prototype** → click your project -2. Click **Choose Winner** — a **Winner Settings** drawer opens -3. Under **Evaluation Metrics**, adjust the importance slider (0 = Not Important, 10 = Very Important) for each eval — `completeness_check` and `response_quality` -4. Under **System Metrics**, adjust the importance sliders for **Avg Cost** and **Avg Latency** -5. Click **Choose Winner** to rank all versions - -The version with the highest weighted score across your chosen importance values is selected as the winner. - -{/* The recording above (Step 5) also covers the Choose Winner flow. */} - - - - -## What you built - - -You can now register a Prototype project, auto-evaluate spans with EvalTags, iterate with versioned prompts, compare versions, and choose the best one for production. - - -- Registered a Prototype project with `ProjectType.EXPERIMENT` and automatic span evaluation via `EvalTag` -- Ran a baseline OpenAI app (v1) and saw completeness and response quality scores in the dashboard -- Iterated with a new prompt version (v2) using a different `project_version_name` -- Compared both versions on eval scores, latency, and cost in the Prototype dashboard -- Chose the winning version using weighted metric comparison - -## Next steps - - - - Docs and version management - - - EvalTag configurations - - - UI-first prompt comparison - - - Custom spans and metadata - - diff --git a/src/pages/docs/cookbook/text-to-sql.mdx b/src/pages/docs/cookbook/text-to-sql.mdx index 17ffe984..4937a979 100644 --- a/src/pages/docs/cookbook/text-to-sql.mdx +++ b/src/pages/docs/cookbook/text-to-sql.mdx @@ -375,7 +375,7 @@ def setup_database(): - `DETECT_HALLUCINATION`: Identifies instances where the agent generates SQL that references non-existent tables, columns, or relationships that aren't present in the database schema. - `table_checker`: A custom evaluation that verifies whether the SQL queries reference the appropriate tables needed to satisfy the user's request, ensuring optimal join patterns and table selection. - > **Click [here](https://docs.futureagi.com/docs/prototype/evals) to learn more about the evals provided by Future AGI** + > **Click [here](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI** > - The **`eval_tags`** list contains multiple instances of **`EvalTag`**. Each **`EvalTag`** represents a specific evaluation configuration to be applied during runtime, encapsulating all necessary parameters for the evaluation process. - Parameters of **`EvalTag`** : @@ -385,15 +385,15 @@ def setup_database(): - **`EvalSpanKind.TOOL`**: For operations involving tools. - **`eval_name`**: The name of the evaluation to be performed. - > Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to get complete list of evals provided by Future AGI + > Click [**here**](/docs/evaluation/builtin) to get complete list of evals provided by Future AGI > - **`config`**: Dictionary for providing specific configurations for the evaluation. An empty dictionary means that default configuration parameters will be used. - Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about what config is required for corresponding evals + Click [**here**](/docs/evaluation/builtin) to learn more about what config is required for corresponding evals - **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation. - Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about what inputs are required for corresponding evals + Click [**here**](/docs/evaluation/builtin) to learn more about what inputs are required for corresponding evals - **`custom_eval_name`**: A user-defined name for the specific evaluation instance. - `model`: LLM model name required to perform the evaluation. Such as `TURING_LARGE`, which is a proprietary model provided by Future AGI. diff --git a/src/pages/docs/dataset/concept/dynamic-column.mdx b/src/pages/docs/dataset/concept/dynamic-column.mdx deleted file mode 100644 index 8a53ef9d..00000000 --- a/src/pages/docs/dataset/concept/dynamic-column.mdx +++ /dev/null @@ -1,62 +0,0 @@ ---- -title: "Dynamic Columns: Auto-Generated Dataset Values in Future AGI" -description: "Dataset columns auto-generated by running LLM prompts, evaluations, vector retrieval, entity extraction, or custom Python code against every row." ---- - -## About - -A dynamic column is generated automatically by the platform. Instead of entering data yourself, you configure a method (like running an LLM prompt or an evaluation) and the platform computes a value for every row. - -For example, starting with two [static columns](/docs/dataset/concept/static-column): - -| user_query | expected_answer | model_response | is_correct | -|---|---|---|---| -| What is the capital of France? | Paris | Paris | true | -| Who wrote Hamlet? | Shakespeare | William Shakespeare | true | - -Here `model_response` is a dynamic column created by running a prompt against each `user_query`. And `is_correct` is another dynamic column created by running an evaluation that compares `model_response` to `expected_answer`. - -Dynamic columns can be regenerated at any time. If you change the prompt or switch models, you can re-run the column and the values update across all rows. - ---- - -## When to use - -- **Get model outputs**: Run an LLM on every row and store the responses for comparison or evaluation -- **Score outputs**: Run evaluations and store the results (pass/fail, scores, explanations) alongside your data -- **Extract structured data**: Pull entities, JSON keys, or classifications out of unstructured text columns -- **Enrich with external data**: Call APIs or vector databases to add context to each row -- **Transform data**: Apply custom Python logic to compute derived values - ---- - -## Supported Methods - -| Method | What it does | -|---|---| -| Run Prompt | Run an LLM prompt that can reference other columns as variables. [Learn more](/docs/dataset/features/run-prompt) | -| Vector Retrieval | Connect to a vector database and retrieve the top-k chunks for a query | -| Entity Extraction | Extract named entities (people, organizations, locations) from text columns using a model | -| JSON Key Extraction | Parse a JSON column and extract specific keys or nested values | -| Custom Code Execution | Write and run Python code for transformations or complex operations | -| Text Classification | Assign categories or labels to text using a model | -| API Calls | Call an external API endpoint for every row and store the response | -| Conditional Logic | Apply different actions based on conditions (if/else branching across rows) | - ---- - -## How It Works - -1. Choose a dynamic column method from the list above -2. Configure the method (select a model, write a prompt, define the logic) -3. Map input columns (e.g. use `user_query` as the input to your prompt) -4. Run the column. The platform processes all rows in parallel and fills in the values. -5. View the results in your dataset. Re-run anytime to refresh. - ---- - -## Next Steps - -- [Static Columns](/docs/dataset/concept/static-column): Columns with fixed data you provide directly -- [Run Prompt in Dataset](/docs/dataset/features/run-prompt): The most common dynamic column method -- [Experiments](/docs/dataset/features/experiments): Compare dynamic column results across different configurations \ No newline at end of file diff --git a/src/pages/docs/dataset/concept/static-column.mdx b/src/pages/docs/dataset/concept/static-column.mdx deleted file mode 100644 index e5cb4014..00000000 --- a/src/pages/docs/dataset/concept/static-column.mdx +++ /dev/null @@ -1,56 +0,0 @@ ---- -title: "Static Columns: Fixed Dataset Values in Future AGI" -description: "Dataset columns for storing fixed test inputs, expected outputs, labels, and metadata. Supports 9 data types including text, JSON, image, and audio." ---- - -## About - -A static column holds data that you provide directly. This includes inputs, expected outputs, labels, categories, or any fixed values. Unlike [dynamic columns](/docs/dataset/concept/dynamic-column), static columns don't run any computation. They only change when you update them manually or through the SDK. - -For example, in this dataset the first three columns are static: - -| user_query | expected_answer | category | model_response | -|---|---|---|---| -| What is the capital of France? | Paris | geography | *(dynamic)* | -| Summarize this article | A concise summary of... | summarization | *(dynamic)* | - -You add `user_query`, `expected_answer`, and `category` yourself. The `model_response` column would be a [dynamic column](/docs/dataset/concept/dynamic-column) generated by running a prompt. - ---- - -## When to use - -- **Test inputs and expected outputs**: Store the queries and ground truth answers for evaluation -- **Labels and categories**: Tag rows with classifications (e.g. "easy", "hard", "geography", "math") -- **Default values**: Pre-fill rows with consistent starting data when setting up a dataset -- **Metadata**: Store context like source, timestamp, or user ID alongside your test data - ---- - -## Supported Data Types - -| Type | Description | -|---|---| -| `text` | Strings and free-form text | -| `integer` | Whole numbers | -| `float` | Decimal numbers | -| `boolean` | True or false | -| `array` | Lists of values | -| `json` | Structured JSON objects | -| `image` | Image file references | -| `audio` | Audio file references | -| `datetime` | Date and time values | - ---- - -## How to Add a Static Column - -You can add static columns through the UI or when creating a dataset via the SDK. See [Add Columns to Dataset](/docs/dataset/features/add-columns) for step-by-step instructions. - ---- - -## Next Steps - -- [Dynamic Columns](/docs/dataset/concept/dynamic-column): Columns generated by prompts, evaluations, or models -- [Add Columns](/docs/dataset/features/add-columns): Add new columns to an existing dataset -- [Create a Dataset](/docs/dataset/features/create): Start a new dataset from scratch \ No newline at end of file diff --git a/src/pages/docs/dataset/concept/synthetic-data.mdx b/src/pages/docs/dataset/concept/synthetic-data.mdx deleted file mode 100644 index 5f28844e..00000000 --- a/src/pages/docs/dataset/concept/synthetic-data.mdx +++ /dev/null @@ -1,63 +0,0 @@ ---- -title: "Synthetic Data Generation for AI Testing in Future AGI" -description: "Generate schema-driven test datasets without using real user data. Define column types, constraints, and descriptions, then generate rows using Future AGI." ---- - -## About - -Synthetic data is artificially generated data that follows real-world patterns without using actual user data. In Future AGI, you define a schema (columns, types, descriptions, and constraints) and the platform generates rows that match your specification. - -For example, defining this schema: - -| Column | Type | Description | -|---|---|---| -| customer_query | text | A realistic customer support question | -| sentiment | text | One of: positive, negative, neutral | -| priority | integer | 1 (low) to 5 (urgent) | - -Produces rows like: - -| customer_query | sentiment | priority | -|---|---|---| -| I haven’t received my order and it’s been two weeks | negative | 4 | -| Can I change the shipping address on my recent order? | neutral | 2 | -| Your product is fantastic, just wanted to say thanks! | positive | 1 | - -The generated data follows the constraints you set (sentiment is always one of three values, priority stays in range) while producing varied, realistic content. - ---- - -## When to use - -- **No real data available**: You’re building a new feature and don’t have production data yet -- **Privacy constraints**: Real data contains PII or sensitive information that can’t be used for testing -- **Edge case testing**: You need specific scenarios (angry customers, rare errors, multilingual queries) that are hard to find in real data -- **Scale testing**: You need thousands of rows to stress-test evaluations or prompts -- **Balanced datasets**: Real data is skewed (e.g. 95% positive reviews) and you need more balanced distributions - ---- - -## How It Works - -1. Define the schema: column names, data types, and descriptions -2. Set constraints: value ranges, categorical options, patterns -3. Optionally connect a [Knowledge Base](/docs/knowledge-base) to ground generation with your own documents -4. Choose the number of rows to generate -5. The platform generates the dataset. You can review, edit, and use it immediately. - ---- - -## Key Properties - -- **Schema-driven**: You control the structure. Every column has a type, description, and optional constraints that guide generation. -- **Realistic distribution**: Generated data follows natural patterns and distributions, not random values. Descriptions give the generator context to produce relevant content. -- **Safe by default**: Generated data does not contain real PII, credentials, or sensitive information. - ---- - -## Next Steps - -- [Generate Synthetic Data](/docs/quickstart/generate-synthetic-data): Step-by-step quickstart for creating your first synthetic dataset -- [Static Columns](/docs/dataset/concept/static-column): How static columns store the data you provide -- [Dynamic Columns](/docs/dataset/concept/dynamic-column): How to add model outputs and evaluations on top of your synthetic data -- [Knowledge Base](/docs/knowledge-base): Ground synthetic generation with your own documents \ No newline at end of file diff --git a/src/pages/docs/dataset/concept/understanding-dataset.mdx b/src/pages/docs/dataset/concept/understanding-dataset.mdx deleted file mode 100644 index 056c58ab..00000000 --- a/src/pages/docs/dataset/concept/understanding-dataset.mdx +++ /dev/null @@ -1,77 +0,0 @@ ---- -title: "Future AGI Datasets: Structure, Column Types, and Lifecycle" -description: "Each row is one example; each column is an attribute. Datasets are the foundation for running prompts, evals, experiments, and optimizations in Future AGI." ---- - -## About - -A dataset in Future AGI is a table of structured data. Each row is one example (e.g. a user query and its expected answer). Each column is an attribute (e.g. "input", "expected_output", "model_response", "score"). Datasets are the foundation for running prompts, evaluations, experiments, and optimizations. - -Here's what a simple dataset looks like: - -| input | expected_output | model_response | is_correct | -|---|---|---|---| -| What is the capital of France? | Paris | Paris | true | -| Who wrote Hamlet? | Shakespeare | William Shakespeare | true | -| What is 2+2? | 4 | The answer is 4 | true | - -The first two columns (input, expected_output) are [static columns](/docs/dataset/concept/static-column) that you add manually. The last two (model_response, is_correct) are [dynamic columns](/docs/dataset/concept/dynamic-column) generated by running a prompt and an evaluation against each row. - ---- - -## Structure - -Every dataset has three core components: - -- **Rows**: Each row is one data point or test case. You can add rows manually, import from files, generate them synthetically, or pull them from production traces. -- **Columns**: Each column defines an attribute. Columns have a name, a data type (text, number, boolean, JSON, etc.), and are either static (you provide the data) or dynamic (the platform generates it). -- **Metadata**: Each dataset has a name, description, and organization-level permissions that control who can view and edit it. - ---- - -## How to Create a Dataset - -There are several ways to get data into a dataset: - -- **Manual creation**: Define the structure and add rows through the UI or SDK. [Learn more](/docs/dataset/features/create) -- **File import**: Upload CSV, Excel, JSON, or JSONL files. [Learn more](/docs/dataset/features/create) -- **Synthetic generation**: Describe the schema and let the platform generate realistic test data. [Learn more](/docs/dataset/concept/synthetic-data) -- **From HuggingFace**: Import existing datasets from HuggingFace directly. [Learn more](/docs/cookbook/quickstart/huggingface-dataset-import) -- **From production traces**: Convert observed production data from the Observe module into datasets for regression testing. [Learn more](/docs/observe) - ---- - -## Dataset Lifecycle - -### 1. Create - -Start with a schema (columns and types) and populate it with data using any of the methods above. - -### 2. Enrich - -Add more columns to your dataset over time: - -- **Run prompts**: Send each row through an LLM and store the responses as a new column. [Learn more](/docs/dataset/features/run-prompt) -- **Run evaluations**: Score model outputs using 70+ built-in metrics. Results are stored as new columns. [Learn more](/docs/evaluation) -- **Add annotations**: Manually label rows with custom tags and scores. Future AGI also supports auto-annotations that learn from your labels. [Learn more](/docs/dataset/features/annotate) - -### 3. Experiment - -Use the same dataset to compare different prompts, models, or configurations side by side. Each experiment run adds new columns so you can see results next to each other. [Learn more](/docs/dataset/features/experiments) - -### 4. Maintain - -Datasets evolve over time. You can: - -- Add or remove columns without disrupting existing data -- Add new rows as you discover edge cases -- Archive or delete old datasets to keep your workspace clean - ---- - -## Next Steps - -- [Static Columns](/docs/dataset/concept/static-column): Data you add directly to your dataset -- [Dynamic Columns](/docs/dataset/concept/dynamic-column): Data generated by prompts, evaluations, or models -- [Synthetic Data](/docs/dataset/concept/synthetic-data): Generate realistic test data from a schema -- [Create a Dataset](/docs/dataset/features/create): Get started with your first dataset \ No newline at end of file diff --git a/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx b/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx new file mode 100644 index 00000000..848ba38c --- /dev/null +++ b/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx @@ -0,0 +1,62 @@ +--- +title: "Static & Dynamic Columns" +description: "Whether a column holds values you set yourself, or values a producer computes for you" +--- + +## Where a column's values come from + +Every [column](/docs/dataset/concepts/understanding-datasets) is either **static** or **dynamic**, and that's a property of the column itself, not of any single row inside it. The difference comes down to where its values come from. + +A static column holds values you supply. You type them in, paste them, or set them through the SDK, and a cell only changes when you go back and edit it. + +A dynamic column doesn't hold values you typed, it holds something that fills them for you: [a prompt you run over every row](/docs/dataset/guides/run-a-prompt-on-every-row), an evaluation, an API call, and more. Every cell in the column is whatever that produced for that row, not something you set by hand. The full set of things a dynamic column can run is cataloged in [Dynamic column methods](/docs/dataset/reference/dynamic-column-methods). + +Take a dataset with four columns: + +| user_query | expected_answer | model_response | is_correct | +|---|---|---|---| +| What is the capital of France? | Paris | Paris | true | +| Who wrote Hamlet? | Shakespeare | William Shakespeare | true | + +`user_query` and `expected_answer` are static, you wrote them in. `model_response` is dynamic: a prompt behind the column answers `user_query` for every row. `is_correct` is dynamic too: an evaluation behind it compares `model_response` against `expected_answer`. + +## Mental model: producer or no producer + + ST["Static
you fill it"] + COL --> DY["Dynamic
something fills it"] + ST -->|"you set it"| CS1["Cell · row 1"] + ST -->|"you set it"| CS2["Cell · row 2"] + DY -->|"runs"| PR["A prompt, or an evaluation"] + PR -->|"fills"| CD1["Cell · row 1"] + PR -->|"fills"| CD2["Cell · row 2"]`} /> + +## What follows from having a producer behind the column + +Several consequences fall directly out of that difference. + +**It carries a status while the producer runs.** A static column has no run to track, so it has no status to show. A dynamic column does: while its producer is working, the column sits in a running state, and if the producer fails, the column shows failed. That status is the tell for whether you're looking at a value you can trust yet. + +**It can be re-run, and every row changes at once.** The producer behind a dynamic column doesn't disappear after the first run. Change the prompt, switch the model, fix the eval config, then re-run the column, and every cell it owns recomputes together. Editing a static column, by contrast, is you overwriting one cell at a time; nothing else moves. + +**Deleting it takes the producer with it.** A dynamic column isn't just the column, it's the column plus the producer generating it. Delete the column and its producer goes too, along with anything else that was derived from it. Deleting a static column removes only the column and the values sitting in it, there's no producer behind it to clean up. + +## Why it matters + +Before you touch a column, it's worth knowing which kind you're looking at. The consequences above all come from the same root: touch a dynamic column and you're really touching the prompt or evaluation behind it, not just the cell or the column in front of you. + +## Keep exploring + + + + Create a static or dynamic column in a dataset + + + The most common dynamic column, walked end to end + + + Methods you can point a dynamic column at + + diff --git a/src/pages/docs/dataset/concepts/synthetic-data.mdx b/src/pages/docs/dataset/concepts/synthetic-data.mdx new file mode 100644 index 00000000..294c564a --- /dev/null +++ b/src/pages/docs/dataset/concepts/synthetic-data.mdx @@ -0,0 +1,102 @@ +--- +title: "Synthetic Data" +description: "Turning a column schema into realistic dataset rows, without real production data" +--- + +## What synthetic data is + +**Synthetic data** is a [dataset's](/docs/dataset/concepts/understanding-datasets) rows generated from a schema you define, instead of rows you upload or bring in yourself. You describe the [columns](/docs/dataset/concepts/static-and-dynamic-columns) you want, their names, types, and constraints, and Future AGI generates rows that match. + +Define this schema for a customer-support dataset: + +| Column | Type | Constraints | +|---|---|---| +| customer_query | text | Value: a realistic customer support question | +| sentiment | text | Categorical values: positive, negative, neutral | +| priority | integer | Value: 1 (low) to 5 (urgent) | + +Generation produces rows like: + +| customer_query | sentiment | priority | +|---|---|---| +| I haven't received my order and it's been two weeks | negative | 4 | +| Can I change the shipping address on my recent order? | neutral | 2 | +| Your product is fantastic, just wanted to say thanks! | positive | 1 | + +Each column has its own Property editor for exactly this: Min Length and Max Length on most column types, Value set to Categorical for a list of allowed values, plus custom properties for anything else. Those column properties, not the column's description, are where allowed values and ranges live. + + +Column properties steer the generator toward matching rows, but they aren't hard validation on the result. Skim the generated rows before you rely on them. + + +To generate your first synthetic dataset hands-on, follow the [Generate synthetic data quickstart](/docs/quickstart/generate-synthetic-data). + +## The generation config + +Every synthetic dataset saves what you defined as the dataset's own generation config: + +- **Columns**: the schema you defined, with each column's name, type, and description +- **Row count**: how many rows to generate +- **Description**: what the dataset as a whole should contain +- **Objective**: how you plan to use the dataset, so generation can match that goal +- **Pattern**: an example or format you want the generated rows to follow +- An optional [Knowledge Base](/docs/knowledge-base) link + +That's what lets you reopen a synthetic dataset later, change a column or the row count, and regenerate without rebuilding the schema from scratch. Here's how the config, the job, and the dataset's rows and state fit together: + +|read by| JOB + JOB -->|fills| ROWS + JOB -->|drives| STATE + KB -.->|grounds| JOB`} /> + +When a Knowledge Base is connected, generation grounds rows in its content instead of relying on the schema alone. + +### Editing vs regenerating + +| Action | What happens | +|---|---| +| Edit the config and save | Adds the columns and rows you added, drops the columns you removed, drops rows if you lowered the row count, and leaves every other column's data as it is | +| Regenerate | Rebuilds all rows and columns from the config (destructive, see the warning below) | + + +Regenerating wipes the dataset's current rows and columns and rebuilds them fresh from the config. If you only meant to add a column or a few rows, edit and save instead. + + +The saved config is what makes either possible: you're never redefining the schema by hand. + +## When to use synthetic data + +Reach for synthetic data whenever real rows are unavailable, risky to use, or lopsided for what you're testing: + +- **No real data yet**: you're building a new feature and don't have production rows to test against +- **Privacy limits**: real data carries PII you can't put in a test dataset +- **Edge cases**: you need scenarios that are rare in real traffic, like an angry customer or a multilingual query +- **Scale**: you need thousands of rows to stress-test a prompt or eval +- **Skewed data**: real data leans one way (mostly positive reviews) and you need a more balanced set + +## While it's generating, and when it fails + +Generation doesn't happen instantly. Because it runs as a background job, a synthetic dataset sits in a **Generating** state (or **Regenerating**, if you kicked off a rerun) with a live progress bar while the job works, and a **Configure Synthetic Data** button that reopens the schema. + +If the job errors out, the dataset shows a **Failed** state instead, with the same button to fix the configuration and try again. See [Dataset FAQ & fixes](/docs/dataset/troubleshooting) for what to check first. + +## Keep exploring + + + + Step-by-step guide for creating a dataset, including the synthetic flow + + + How static and dynamic columns store and compute values + + diff --git a/src/pages/docs/dataset/concepts/understanding-datasets.mdx b/src/pages/docs/dataset/concepts/understanding-datasets.mdx new file mode 100644 index 00000000..29c28ac6 --- /dev/null +++ b/src/pages/docs/dataset/concepts/understanding-datasets.mdx @@ -0,0 +1,65 @@ +--- +title: "Understanding Datasets" +description: "The columns, rows, and cells a dataset is built from, and who owns it" +--- + +## What a dataset is made of + +A **dataset** is what you run [prompts](/docs/dataset/guides/run-a-prompt-on-every-row), evals, and [experiments](/docs/dataset/guides/run-an-experiment) against. It owns two collections, columns and rows, plus a few fields on the dataset itself. + +### Columns, rows, and cells + +A **column** defines one attribute every row carries. Whether its values are static or dynamic, who or what fills them in, is covered in [Static & Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns). A **row** is one example. Where a row crosses a column sits a **cell**: the stored value for that column, on that row. + +Take a two-column dataset with columns `input` and `model_response`. Both rows get a cell in each column: four cells total, all belonging to the same dataset. Add a third row and both columns grow a new cell; add a third column and both rows do too, which is what keeps the grid rectangular no matter how many of its columns are dynamic. + + C1["Column · input"] + D --> C2["Column · model_response"] + D --> R1["Row 1"] + D --> R2["Row 2"] + C1 --> X11["Cell · Row 1 × input"] + R1 --> X11 + C1 --> X21["Cell · Row 2 × input"] + R2 --> X21 + C2 --> X12["Cell · Row 1 × model_response"] + R1 --> X12 + C2 --> X22["Cell · Row 2 × model_response"] + R2 --> X22`} /> + +### Row and column order + +Row order isn't insertion order. Every row carries its own row-level `order`, an explicit integer that fixes where it sits top to bottom. Column layout works the same way one level up: the dataset carries a dataset-level `column_order` array that fixes the left-to-right order of columns, and it's pruned automatically when a [column is deleted](/docs/dataset/guides/manage-datasets). + +### How cell values are stored + +A cell's value is always stored as text, whatever [the column's data type](/docs/dataset/reference/limits-and-data-types). A JSON column's value is JSON-stringified before it's stored, and a media column, an image or an audio file, holds a URL string rather than the file itself. + +## Ownership and scoping + +A dataset always belongs to exactly one organization; there's no such thing as a dataset with no owning org. A workspace is optional: a dataset can sit inside one workspace for scoping, or none at all. + +`model_type` fixes what kind of data the dataset is built for: generative text by default, or image, audio, video, and structured types such as classification and ranking. + +## Why it matters + +- You can resort rows for review without disturbing anything else in the dataset; `order` is separate from when a row was created or which cells it holds +- Deleting a column cleans up its position in the layout automatically, so nothing is left pointing at a column that no longer exists +- Anything that reads a cell back, a prompt template, an eval, an export, gets a string and has to parse or fetch it for JSON and media columns +- Access to a dataset follows organization membership first; the optional workspace narrows that further + +## Keep exploring + + + + Where a column's values come from, and what changes when it's dynamic + + + Generate realistic rows from a schema instead of writing them by hand + + + Every way to get a dataset that exists and has data in it + + diff --git a/src/pages/docs/dataset/features/add-columns.mdx b/src/pages/docs/dataset/features/add-columns.mdx deleted file mode 100644 index 759314b6..00000000 --- a/src/pages/docs/dataset/features/add-columns.mdx +++ /dev/null @@ -1,188 +0,0 @@ ---- -title: "Adding Static and Dynamic Columns to a Dataset" -description: "Add static columns for fixed values or dynamic columns whose values are computed from other columns or external operations." ---- - -## About - -Adding a column extends your dataset with a new field. Columns can be of two kinds: - -- **[Static columns](/docs/dataset/concept/static-column)**: Store fixed values (text, numbers, boolean, array, JSON) that you enter or paste. They do not require computation; you edit cells manually. -- **[Dynamic columns](/docs/dataset/concept/dynamic-column)**: Values are computed or fetched when you need them (e.g. from an LLM prompt, vector DB, API, custom code, or from existing columns). You configure the type, test, then create; the system fills the column row by row. - -Both are added via **+ Add Columns** in your dataset. - -## When to use - -- **Store reference data**: Keep fixed labels, scores, or expected outputs alongside generated responses for use in evals. -- **Generate model responses**: Run a prompt on each row and store the output in a new column, ready for evaluation or comparison. -- **Add retrieved context**: Fetch relevant chunks from a vector database per row for RAG evaluation or prompt injection. -- **Classify by category**: Assign topic, sentiment, or intent labels to each row using a model and your predefined categories. -- **Extract from free text**: Pull specific entities or values from an unstructured column into a clean, structured column. - -## How to - -Open your dataset and click **+ Add Columns**. Choose **Static** for fixed values or **Dynamic** for computed columns; under Dynamic, pick the method you need. - - - - - - In your dataset, go to the **Data** tab and click **+ Add Columns**. The Add Columns panel opens. - ![Add Columns](/screenshot/product/dataset/how-to/add-columns-to-dataset/static//1.png) - - - Under **Static Columns**, choose the data type: **Text**, **Float**, **Integer**, **Boolean**, **Array**, or **JSON**. - - - Enter a **Column Name** and ensure **Data Type** matches your choice. Click **Create New Column** to add it. You can then fill or edit cells manually. - ![Create New Column](/screenshot/product/dataset/how-to/add-columns-to-dataset/static/2.png) - - - - - Choose a dynamic column type below. Configure it, use **Test** to preview, then **Create New Column**. - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Run Prompt**. - ![Run Prompt](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/1.png) - - - Give the column a name. Build the prompt with messages; use placeholders like {`{{column_name}}`} to pull values from other columns. - ![Run Prompt](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/2.png) - - - Select model type (LLM, Text-to-Speech, Speech-to-Text, or Image) and the model. Optionally configure parameters and tools. - - - Set concurrency, then click **Test** to preview outputs. Click **Create New Column** to add the column. - - - - [Run Prompt in Dataset](/docs/dataset/features/run-prompt) has the full walkthrough. - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Retrieval**. - - - Name the column. Select the vector database: **Pinecone**, **Qdrant**, or **Weaviate**. - ![Retrieval](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/3.png) - - - Select the **query column**. Add API key/secret. Set **Index Name**, **Namespace**, **Number of Chunks**, and **Query Key**. - ![Retrieval](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/4.png) - - - Set embedding type, model, key to extract, and vector length. Set concurrency, then **Test** and **Create New Column**. - - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **API Call**. - - - Name the column. Choose **Output Type**: string, object, array, or number. - ![API Call](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/5.png) - - - Enter **API URL** and **Method** (GET, POST, PUT, etc.). Add params, headers, and body; use {`{{column_name}}`} to reference column values. - ![API Call](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/6.png) - - - Set concurrency. Click **Test** to verify, then **Create New Column**. - - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Extract JSON Key**. - - - Name the column. Select the dataset column of type JSON that contains the data. - ![Extract JSON](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/7.png) - - - Enter the **JSON key** (path) to extract (e.g. age for a JSON object like {`{"name": "John", "age": 30}`}). Set concurrency, **Test**, then **Create New Column**. - - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Extract Entities**. - - - Name the column. Select the column to extract from and enter **instructions** for what to extract. - ![Extract Entities](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/8.png) - - - Select the model (API key may be required). Set concurrency, **Test**, then **Create New Column**. - - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Classification**. - - - Name the column. Select the column that contains the text to classify. - ![Classification](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/9.png) - - - Click **Add Label** and define categories (e.g. Positive, Negative, Neutral). Choose the model and set concurrency. - - - Click **Test** to preview, then **Create New Column**. - - - - - - - In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Conditional Node**. - - - Name the column. Define **if**, **elif** (optional), and **else** conditions. - ![Conditional](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/10.png) - - - For each branch, choose an operation (Run Prompt, Retrieval, Extract Entities, Extract JSON, Execute Code, Classification, or API Call) and configure it. - ![Conditional](/screenshot/product/dataset/how-to/add-columns-to-dataset/dynamic/11.png) - - - Click **Test** to verify, then **Create New Column**. - - - - - - - -## Next Steps - - - - Add individual records or bulk import data rows to your dataset - - - Test and execute prompts against your dataset entries - - - Design and run controlled experiments to compare approaches - - - Add metadata and annotations to enrich your dataset - - - Create another dataset using SDK, file upload, or synthetic generation - - diff --git a/src/pages/docs/dataset/features/add-rows.mdx b/src/pages/docs/dataset/features/add-rows.mdx deleted file mode 100644 index 6c81c579..00000000 --- a/src/pages/docs/dataset/features/add-rows.mdx +++ /dev/null @@ -1,300 +0,0 @@ ---- -title: "Adding Data Rows to an Existing Dataset in Future AGI" -description: "Add data points to an existing dataset manually, from another dataset, Hugging Face, from production traces, or by generating synthetic rows." ---- - -## About - -Add Rows is how you add more data points (rows) to an existing dataset. Each new row gets one cell per column. You either provide the values, copy them from another dataset or source, or generate them (e.g. synthetic or from traces). The dataset's columns stay as they are; only new rows and their cells are created. - -## When to use - -- **Manual or API data entry**: You have new test cases (e.g. new queries or examples). Add rows with cell values via the UI or API so they become part of the same dataset for run prompt and evals. -- **Copy from another dataset**: You have rows in a different dataset (or an experiment snapshot) and want them in this one. Add rows from that source with a column mapping so the right fields line up. -- **Append from Hugging Face**: You want more examples from a Hugging Face dataset. Add rows from that dataset into the current one so you don't re-import from scratch. -- **Generate more synthetic data**: The dataset was created with synthetic config; you want more rows with the same logic. Add synthetic rows to fill more of the table. -- **Bring in more production data**: You have new traces/spans in the tracer. Add them to an existing dataset so evals and experiments stay on one dataset. - -## How to - -Choose how you want to add rows to your dataset: - - -Learn how to [create a new dataset](/docs/dataset/features/create) first if you don't have one yet. - - - - -Use the SDK to append rows to an existing dataset. - - - In your app or script, open the dataset you want to add rows to (by name or ID). - - - Define new rows with cells (column name + value), then call the add-rows API. - - - - ```python Python - # pip install futureagi - - import os - from fi.datasets import Dataset - from fi.datasets.types import ( - Cell, - Column, - DatasetConfig, - DataTypeChoices, - ModelTypes, - Row, - SourceChoices, - ) - - # Set environment variables - os.environ["FI_API_KEY"] = "" - os.environ["FI_SECRET_KEY"] = "" - - # Get existing dataset - config = DatasetConfig(name="Demo-dataset", model_type=ModelTypes.GENERATIVE_LLM) - dataset = Dataset(dataset_config=config) - dataset = Dataset.get_dataset_config("Demo-dataset") - - # Define columns - columns = [ - Column( - name="user_query", - data_type=DataTypeChoices.TEXT, - source=SourceChoices.OTHERS - ), - Column( - name="response_quality", - data_type=DataTypeChoices.INTEGER, - source=SourceChoices.OTHERS - ), - Column( - name="is_helpful", - data_type=DataTypeChoices.BOOLEAN, - source=SourceChoices.OTHERS - ) - ] - - # Define rows - rows = [ - Row( - order=1, - cells=[ - Cell(column_name="user_query", value="What is machine learning?"), - Cell(column_name="response_quality", value=8), - Cell(column_name="is_helpful", value=True) - ] - ), - Row( - order=2, - cells=[ - Cell(column_name="user_query", value="Explain quantum computing"), - Cell(column_name="response_quality", value=9), - Cell(column_name="is_helpful", value=True) - ] - ) - ] - - try: - # Add rows to dataset - dataset = dataset.add_rows(rows=rows) - print("✓ Data added successfully") - except Exception as e: - print(f"Failed to add data: {e}") - - ``` - - ```typescript Typescript - import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk"; - - process.env["FI_API_KEY"] = ""; - process.env["FI_SECRET_KEY"] = ""; - process.env["FI_BASE_URL"] = "https://api.futureagi.com"; - - async function main() { - try { - const dsName = "Demo-dataset"; - - // 1) Open the dataset (fetch if it exists, create if not) - const dataset = await Dataset.open(dsName); - - // 2) Define rows - const rows = [ - createRow({ - cells: [ - createCell({ columnName: "user_query", value: "What is machine learning?" }), - createCell({ columnName: "response_quality", value: 8 }), - createCell({ columnName: "is_helpful", value: true }), - ], - }), - createRow({ - cells: [ - createCell({ columnName: "user_query", value: "Explain quantum computing" }), - createCell({ columnName: "response_quality", value: 9 }), - createCell({ columnName: "is_helpful", value: true }), - ], - }), - ]; - await dataset.addRows(rows); - console.log("✓ Data added successfully"); - } catch (err) { - console.error("Failed to add data:", err); - } - } - - main(); - ``` - - ```bash Curl - curl --request POST \ - --url https://api.futureagi.com/model-hub/develops//add_rows/ \ - --header 'content-type: application/json' \ - --header 'X-Api-Key: ' \ - --header 'X-Secret-Key: ' \ - --data '{ - "rows": [ - { - "order": 1, - "cells": [ - { - "column_name": "user_query", - "value": "What is machine learning?" - }, - { - "column_name": "response_quality", - "value": 8 - }, - { - "column_name": "is_helpful", - "value": true - } - ] - }, - { - "order": 2, - "cells": [ - { - "column_name": "user_query", - "value": "Explain quantum computing" - }, - { - "column_name": "response_quality", - "value": 9 - }, - { - "column_name": "is_helpful", - "value": true - } - ] - } - ] - }' - ``` - - - Click [here](https://app.futureagi.com/dashboard/keys) to access API Key and Secret Key. - - - - -Add rows using the Add Row option in the dataset view. - - - Open the dataset you want to add rows to from your [dashboard](https://app.futureagi.com/dashboard/develop). - ![add_row_open_dataset](/screenshot/product/dataset/how-to/add-rows-to-dataset/1.png) - - - Click the "Add Row" option to create one or more new rows. New rows appear at the bottom of the table. - ![add_row_action](/screenshot/product/dataset/how-to/add-rows-to-dataset/2.png) - - - Double-click a cell to edit it. Enter values for each column. Repeat for all new rows. - ![add_row_fill_cells](/screenshot/product/dataset/how-to/add-rows-to-dataset/3.png) - - - - -Copy rows from another dataset (or experiment dataset) into this one. - - - From the dataset view, choose the option to add rows from an existing dataset. - ![add_row_from_existing](/screenshot/product/dataset/how-to/add-rows-to-dataset/4.png) - - - Select the source dataset (or experiment dataset). Map each source column to a column in the current dataset. Only mapped columns are copied. - ![add_row_map_columns](/screenshot/product/dataset/how-to/add-rows-to-dataset/5.png) - - | Property | Description | - | -------- | ----------- | - | Source dataset | The dataset or experiment dataset to copy rows from | - | Column mapping | Target column → source column (only mapped columns are copied) | - - - Click "Add" to copy the rows. New rows are appended to the current dataset. - - - - -Append rows from a Hugging Face dataset. - - - From the dataset view, choose to add rows from Hugging Face. - ![add_row_hf_open](/screenshot/product/dataset/how-to/add-rows-to-dataset/6.png) - - - Search and select the Hugging Face dataset. Choose subset, split, and how many rows to import. Map or confirm columns if required. - ![add_row_hf_config](/screenshot/product/dataset/how-to/add-rows-to-dataset/7.png) - - - Start the import. Rows are appended to your dataset and appear in your [dashboard](https://app.futureagi.com/dashboard/develop). - - - - -Upload a file (CSV, JSON, JSONL, or Excel) to append rows to your dataset. - - - From the dataset view, choose the option to add rows by uploading a file. - ![add_row_upload_open](/screenshot/product/dataset/how-to/add-rows-to-dataset/8.png) - - - Select the dataset you want to add rows to, then upload your file. Column names in the file are matched to existing columns; if the file has new column names, new columns are created on the dataset. - - | Property | Description | - | -------- | ----------- | - | Dataset | The dataset to append rows to | - | File | CSV, JSON, JSONL, or Excel file. Column names should match (or will create new columns) | - - - Rows from the file are appended to the dataset. Image and audio values are uploaded to storage. You can run prompt or evals on the updated dataset. - - - - - - -The number of columns will increase automatically to match the number of columns in the new dataset. And the cells will be None by default. - - -## Next Steps - - - - Extend your dataset structure with additional data fields - - - Test and execute prompts against your dataset entries - - - Design and run controlled experiments to compare approaches - - - Add metadata and annotations to enrich your dataset - - - Create another dataset using SDK, file upload, or synthetic generation - - \ No newline at end of file diff --git a/src/pages/docs/dataset/features/annotate.mdx b/src/pages/docs/dataset/features/annotate.mdx deleted file mode 100644 index 522dc73d..00000000 --- a/src/pages/docs/dataset/features/annotate.mdx +++ /dev/null @@ -1,88 +0,0 @@ ---- -title: "Adding Annotations to Dataset Rows in Future AGI" -description: "Annotations are essential for refining datasets, evaluating model outputs, and improving the quality of AI-generated responses." ---- - -## About - -Annotations let you add human labels to dataset rows so you can evaluate model outputs, build training or evaluation data, and improve quality. You create **annotation views** on a dataset: each view defines which columns are shown as context (static fields), which columns hold the content to annotate (response fields), and which **label** (e.g. sentiment, score, free text) is used. Annotators (workspace members you assign) fill in labels per row. For categorical labels, you can optionally use **auto-annotation** to get suggestions based on your existing labels. - -## When to use - -- **Sentiment analysis**: Categorical labels (e.g. Positive, Negative, Neutral) to measure tone of model outputs. -- **Factuality check**: Boolean or text labels to validate whether the output is grounded in the source. -- **Toxicity review**: Categorical labels to flag harmful, biased, or unsafe responses. -- **Relevance scoring**: Numeric (or star) labels to rate how well the response addresses the query. -- **Grammar / style edits**: Text labels to provide corrections or rewritten versions. -- **Prompt comparison**: Categorical or numeric labels to compare responses from different prompt variants. - -## How to - - - - Go to **Datasets** from the dashboard and open the dataset you want to annotate. If you don't have a dataset yet, [create or upload one](/docs/dataset/features/create) first. - ![Select a dataset](/screenshot/product/dataset/how-to/annotate-dataset/1.png) - - - - Inside the dataset view, open the **Annotations** tab or button (near the top or side of the data table). This opens the interface for managing annotation views and labels. - ![Open the annotation interface](/screenshot/product/dataset/how-to/annotate-dataset/2.png) - - - - An annotation view defines *what* you annotate and *how*. Click **Create New View**, give the view a **Name** (e.g. "Sentiment Labels", "Fact Check Ratings"), and save. You will configure static fields, response fields, and the label in a later step. - - - - Labels define the type and possible values for your annotations. Click **Create New Label** if you don't have one. Give the label a **Name** (e.g. "Sentiment", "Accuracy Score") and choose a **Type**: **Categorical** (predefined options, e.g. Positive, Negative, Neutral), **Numeric** (scale with min/max, e.g. 1–5), **Text** (free-form feedback or corrections), **Star** (1–5 stars), or **Thumbs up/down** (pass/fail). Click **Save** to create the label. - ![Define labels](/screenshot/product/dataset/how-to/annotate-dataset/4.png) - **Auto-annotation (Categorical only):** Enable **Auto-Annotation** and the platform learns from your manual labels and suggests labels for unannotated rows. You can accept or override suggestions. - - - - - In the view, connect fields and the label: **Static fields** (columns for context, e.g. user query), **Response fields** (columns to annotate, e.g. model output), **Label** (the label from the previous step). Preview and click **Save**. - - - - In the annotation view settings, open the **Annotators** section and add workspace members who will annotate in this view. - ![Assign annotators](/screenshot/product/dataset/how-to/annotate-dataset/3.png) - - - - Open the annotation view and move through the dataset rows. Click an existing annotation to change it. Changes are saved automatically (or via **Save** if the UI shows it). You can review and override auto-annotation suggestions here as well. - - - -## Annotation Queues - -For structured, multi-annotator annotation campaigns with progress tracking, assignment strategies, and inter-annotator agreement metrics, use **Annotation Queues**. Queues let you organize annotation work across traces, spans, sessions, dataset rows, prototypes, and simulations. - - - - Learn about annotation queues, labels, and the full annotation workflow - - - Get started with annotation queues in 5 minutes - - - -## Next Steps - - - - Add individual records or bulk import data rows to your dataset - - - Extend your dataset structure with additional data fields - - - Test and execute prompts against your dataset entries - - - Design and run controlled experiments to compare approaches - - - Create another dataset using SDK, file upload, or synthetic generation - - diff --git a/src/pages/docs/dataset/features/create.mdx b/src/pages/docs/dataset/features/create.mdx deleted file mode 100644 index badb83d1..00000000 --- a/src/pages/docs/dataset/features/create.mdx +++ /dev/null @@ -1,315 +0,0 @@ ---- -title: "Creating a Dataset in Future AGI from Files, SDK, or Traces" -description: "Create a dataset from CSV, Hugging Face, production traces, or synthetic generation. Use it as the container for prompts, evals, and experiments." ---- - -## About - -Creating a new dataset adds a blank table (or a table filled from a source) under your organization. You get a dataset with a name and optional columns/rows that you can then use for run prompt, evals, experiments, and optimization. The dataset is the container; you can keep editing it after creation. - -## When to use - -- **Evaluate a prompt or model**: You need a set of inputs and (optionally) expected outputs or scores. Creating a dataset gives you that table so you can run prompts and evals on it. -- **Reuse production data**: You have traces/spans from your app and want to turn them into eval data. Creating a dataset from Observe turns selected traces into rows. -- **Import existing data**: You already have test cases in CSV/Excel or on Hugging Face. Creating a dataset from file or Hugging Face imports that data so you don't have to type it in. -- **Generate test data**: You don't have real data yet but know the kind of examples you need. Creating a [synthetic dataset](/docs/dataset/concept/synthetic-data) generates rows for you. -- **Branch from an experiment**: You ran an experiment and want to keep that snapshot as a standalone dataset to edit or reuse. Creating a dataset from that experiment copies it into a new dataset. - -## How to - -Choose how you want to create your dataset: - - - -Use SDK to import your data to Future AGI. - - - Assign a name to your dataset and click on "Next" to proceed. - - ![assign_dataset_name](/screenshot/product/dataset/how-to/create-new-dataset/1.png) - - - Use the code snippet below to add rows to your dataset. - - - - ```python Python - # pip install futureagi - - import os - from fi.datasets import Dataset - from fi.datasets.types import ( - Cell, - Column, - DatasetConfig, - DataTypeChoices, - ModelTypes, - Row, - SourceChoices, - ) - - # Set environment variables - os.environ["FI_API_KEY"] = "" - os.environ["FI_SECRET_KEY"] = "" - - # Get existing dataset - config = DatasetConfig(name="my-dataset", model_type= ModelTypes.GENERATIVE_LLM) - dataset = Dataset(dataset_config=config) - dataset = Dataset.get_dataset_config("my-dataset") - - # Define columns - columns = [ - Column( - name="user_query", - data_type=DataTypeChoices.TEXT, - source=SourceChoices.OTHERS - ), - Column( - name="response_quality", - data_type=DataTypeChoices.INTEGER, - source=SourceChoices.OTHERS - ), - Column( - name="is_helpful", - data_type=DataTypeChoices.BOOLEAN, - source=SourceChoices.OTHERS - ) - ] - - # Define rows - rows = [ - Row( - order=1, - cells=[ - Cell(column_name="user_query", value="What is machine learning?"), - Cell(column_name="response_quality", value=8), - Cell(column_name="is_helpful", value=True) - ] - ), - Row( - order=2, - cells=[ - Cell(column_name="user_query", value="Explain quantum computing"), - Cell(column_name="response_quality", value=9), - Cell(column_name="is_helpful", value=True) - ] - ) - ] - - try: - # Add columns and rows to dataset - dataset = dataset.add_columns(columns=columns) - dataset = dataset.add_rows(rows=rows) - print("✓ Data added successfully") - - except Exception as e: - print(f"Failed to add data: {e}") - ``` - - ```typescript Typescript - import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk"; - - process.env["FI_API_KEY"] = ""; - process.env["FI_SECRET_KEY"] = ""; - - async function main() { - try { - const dsName = "my-dataset"; - - // 1) Open the dataset (fetch if it exists, create if not) - const dataset = await Dataset.open(dsName); - - // 2) Define columns - const columns = [ - { name: "user_query", dataType: DataTypeChoices.TEXT }, - { name: "response_quality", dataType: DataTypeChoices.INTEGER }, - { name: "is_helpful", dataType: DataTypeChoices.BOOLEAN }, - ]; - - // 3) Define rows - const rows = [ - createRow({ - cells: [ - createCell({ columnName: "user_query", value: "What is machine learning?" }), - createCell({ columnName: "response_quality", value: 8 }), - createCell({ columnName: "is_helpful", value: true }), - ], - }), - createRow({ - cells: [ - createCell({ columnName: "user_query", value: "Explain quantum computing" }), - createCell({ columnName: "response_quality", value: 9 }), - createCell({ columnName: "is_helpful", value: true }), - ], - }), - ]; - - // 4) Add columns and rows - await dataset.addColumns(columns); - await dataset.addRows(rows); - console.log("✓ Data added successfully"); - } catch (err) { - console.error("Failed to add data:", err); - } - } - - main(); - - ``` - - ```bash cURL - curl --request POST \ - --url https://api.futureagi.com/model-hub/develops//add_columns/ \ - --header 'X-Api-Key: ' \ - --header 'X-Secret-Key: ' \ - --header 'content-type: application/json' \ - --data '{ - "new_columns_data": [ - { - "name": "user_query", - "data_type": "text" - }, - { - "name": "response_quality", - "data_type": "integer" - }, - { - "name": "is_helpful", - "data_type": "boolean" - } - ] - }' - ``` - - - Click [here](https://app.futureagi.com/dashboard/keys) to access API Key and Secret Key. - - - - - - - - ![upload_file](/screenshot/product/dataset/how-to/create-new-dataset/2.png) - - - - -Synthetically generate data and perform experimentations on it. - - - - Provide basic details about the dataset you want to generate. - - ![add_details](/screenshot/product/dataset/how-to/create-new-dataset/3.png) - - | Property | Description | - | -------- | ------------------------------------- | - | Name | Name of the dataset | - | Knowledge Base (optional) | Select which knowledge base you want to use. | - | Description | Describe the dataset you want to generate | - | Objective (optional) | Use case of the dataset | - | Pattern (optional) | Style, tone or behavioral traits of the generated dataset | - | No. of Rows | Row count of the generated dataset (min 10 rows)| - - - - Define column types and properties - - ![add_column_properties](/screenshot/product/dataset/how-to/create-new-dataset/4.png) - - | Property | Description | - | -------- | ------------------------------------- | - | Column Name | Name of the column | - | Column Type | Choose the type of the column (available types: text, boolean, integer, float, json, array, datetime) | - - - - Now add description for each column. Describe in detail what values you want in this column. - ![add_column_description](/screenshot/product/dataset/how-to/create-new-dataset/5.png) - - - Click on "Create Dataset" button to generate the dataset. Your synthetic dataset will be generated in a few seconds and will be available in your dataset [dashboard](https://app.futureagi.com/dashboard/develop). - - If you are not satisfied with the generated dataset, you can click on "Configure Synthetic Data" button. It will allow you to edit the fields and generate the dataset again. - ![create_dataset](/screenshot/product/dataset/how-to/create-new-dataset/6.png) - ![configure_synthetic_data](/screenshot/product/dataset/how-to/create-new-dataset/7.png) - - - - - -Manually create dataset from scratch. - - - - To proceed with creating dataset manually from scratch, provide the name you want to assign and the number of columns and rows you want. - ![manually](/screenshot/product/dataset/how-to/create-new-dataset/8.png) - This creates an empty dataset with the name you assigned and empty rows and columns. - ![empty_dataset](/screenshot/product/dataset/how-to/create-new-dataset/9.png) - - - You can populate the dataset by double-tapping over the empty cell you want to populate. It will open an editor where you can provide the details you want to fill in that cell. - ![populate_dataset](/screenshot/product/dataset/how-to/create-new-dataset/10.png) - - - - - - - Search for the dataset you want to import from Hugging Face. You can even refine the search by using flters given on left side. - - ![search_hugging_face_dataset](/screenshot/product/dataset/how-to/create-new-dataset/11.png) - - - Once you have selected the dataset you want to import, click on that dataset and it will open a panel where you can select what subset and split you want to import. - - You can also select the number of rows you want to import. By default, it will import all the rows. - ![import_dataset](/screenshot/product/dataset/how-to/create-new-dataset/12.png) - - Click on "Start Experimenting" button and it will start importing the dataset and you will be able to see it in your dataset [dashboard](https://app.futureagi.com/dashboard/develop). - - - - -You can create a subset from an existing dataset. - - - Assign a name to this dataset and choose the existing dataset from the dropdown you want to create a subset from. - ![choose_existing_dataset](/screenshot/product/dataset/how-to/create-new-dataset/13.png) - It allows you to import the dataset in two ways: - - 1. Import Data: It will only import the original columns from the existing dataset. - 2. Import Data and Prompt Configuration: Along with original column, it will also import the prompt columns from that dataset. - - - You can choose what columns you want to use from that existing dataset and also you can assign a new name to the columns you want to use. - ![map_columns](/screenshot/product/dataset/how-to/create-new-dataset/14.png) - - - - Click on "Add" button and it will create a new dataset in your dataset [dashboard](https://app.futureagi.com/dashboard/develop). - - - - - -## Next Steps - - - - Add individual records or bulk import data rows to your dataset - - - Extend your dataset structure with additional data fields - - - Test and execute prompts against your dataset entries - - - Design and run controlled experiments to compare approaches - - - Add metadata and annotations to enrich your dataset - - diff --git a/src/pages/docs/dataset/features/experiments.mdx b/src/pages/docs/dataset/features/experiments.mdx deleted file mode 100644 index cd91bea7..00000000 --- a/src/pages/docs/dataset/features/experiments.mdx +++ /dev/null @@ -1,149 +0,0 @@ ---- -title: "Dataset Experiments: Compare Prompts and Models Side by Side" -description: "Test different prompt and model combinations on the same dataset. Score outputs with built-in evals and compare results side by side." ---- - -## About - -Experiments give you a structured way to answer questions like: *Which prompt performs better? Which model gives the best results? Does my agent beat my prompt for this task?* You import prompts and agents, run them across multiple model and parameter configurations on the same dataset, score the outputs with evals, and compare results side by side so you can make data-driven decisions instead of guessing. - -## When to use - -- **Compare prompts and agents**: Pull prompts from the [Prompt](/docs/prompt) section and agents from the [Agent Playground](/docs/agent-playground) into the same experiment and see which produces better outputs. -- **Compare models and parameters**: Add the same prompt with multiple models, temperatures, or tool configs to compare quality, latency, and cost across configurations. -- **Validate before rollout**: Test a prompt or agent change on a dataset before promoting it to production. -- **Optimize with evals**: Attach built-in or custom evals and use scores to rank prompt/agent-model combinations and pick a winner. -- **Iterate fast**: Stop a long run, edit a single config, or rerun just the failed cells without restarting the whole experiment. - -## How to - -Experiment creation is a guided three-step flow: **Basic Info → Configuration → Evaluations**. Each step validates before you can move forward, and you can jump back to any completed step to edit it. - - - - Open the dataset and click the **Experiments** button in the top-right of the dataset dashboard. - ![Experiments](/screenshot/product/dataset/how-to/experiments-in-dataset/1.png) - - - - Give the experiment a **name** and pick the **experiment type**. - - The name Set up the prompt and model configurations you want to compare. Each configuration becomes a separate column in the experiment grid. is pre-filled with an auto-suggested name based on your dataset. Accept it as-is or overwrite it with your own. Names must be unique within the dataset. - - Pick the experiment type that matches the task you're testing: - - - - Use **LLM** for text generation. You can import prompts *and* agents in the same experiment. - - - Use **TTS** to generate audio from text. Add prompts with different voices, models, and parameters to compare. - - - Use **STT** to transcribe audio. Each prompt configuration must point at a dataset column containing the input audio. - - - Use **Image Generation** to create images from text (or text + image). Compare image models and prompts side by side. - - - - ![Create Experiment](/screenshot/product/dataset/how-to/experiments-in-dataset/2.png) - - - - Set up the prompt and model configurations you want to compare. Each configuration becomes a separate column in the experiment grid. - - - - - For LLM experiments, click **Add Prompt/Agents** to import a prompt or agent. You can mix prompts and agents in the same experiment and score them against the same evals. - - - **Prompts**: pick a prompt from the [Prompt](/docs/prompt) section, select a published version, then attach **one or more models**. Each (prompt, model) pair becomes its own configuration, so adding three models to one prompt creates three columns to compare. For each model you can tune temperature, max tokens, top-p, response format, and tool config. - - **Agents**: pick an agent from the [Agent Playground](/docs/agent-playground) and select a published version. The agent's model, tools, and graph are captured at that version, so the run stays reproducible even if the agent is edited later. You don't pick a model again here. - ![LLM](/screenshot/product/dataset/how-to/experiments-in-dataset/3.png) - - - For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns) and attach one or more **TTS models** (with voice and format settings). Click **+ Add Prompt** to add more prompt entries. Each (prompt, model) pair becomes its own column. Output format is fixed to Audio. - ![TTS](/screenshot/product/dataset/how-to/experiments-in-dataset/4.png) - - - For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns), pick the dataset column containing the input audio, and attach one or more **STT models**. Click **+ Add Prompt** to add more entries to compare transcription quality. - ![STT](/screenshot/product/dataset/how-to/experiments-in-dataset/5.png) - - - For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns) and attach one or more **image models**. Click **+ Add Prompt** to add more entries and compare output quality across models and parameters. - ![Image Generation](/screenshot/product/dataset/how-to/experiments-in-dataset/6.png) - - - Models you've added through Custom Models show up in the model picker for prompt configurations across all experiment types. - - See [Custom Models](/docs/evaluation/guides/custom-models) for how to register a custom or self-hosted model. - - - - - For prompts, you can also configure **tool calling** with **Auto**, **Required**, or **None**, and add tool definitions the model can invoke. - - - - The final step has two parts: an optional **base column** and the **evals** you want to score outputs with. - - **Compare against baseline (optional)**: pick a column from the dataset to compare model outputs against (typically a ground-truth or existing run-prompt column). Skip it if you don't have a reference output yet; you can still run the experiment, attach evals that don't need a baseline, and add a base column later by editing the experiment. - - **Add evaluations**: click **Add Evaluation** and pick from the [built-in eval](/docs/evaluation/builtin) catalog or [create a custom eval](/docs/evaluation/guides/custom-evals). Add as many as you need. Every eval runs on every configuration so the results are directly comparable. - ![Choosing Evals](/screenshot/product/dataset/how-to/experiments-in-dataset/7.png) - - For each eval, map its inputs (e.g. `output`, `input`, `expected`) to the model output or to dataset columns. Mapping is required before the experiment can run. - ![Choosing Evals](/screenshot/product/dataset/how-to/experiments-in-dataset/8.png) - - - - Click **Run** to start. The experiment processes every row across every prompt/agent-model configuration in parallel, running the evals on each output as it arrives. The grid streams results live so you can watch progress without waiting for the whole run to finish. - - - - If you spot a misconfiguration or want to abort, click **Stop** on a running experiment from the Experiments tab. Any in-flight cells are marked as errored, and you can then edit the experiment and rerun without waiting for the full run to complete. - - - - Use **Rerun Experiment** to re-execute the entire experiment after editing prompts, models, evals, or the base column. Editing is granular: only the configurations you actually changed are re-executed, and results from untouched configurations are preserved. - - For more targeted reruns: - - - **Rerun a single cell**: hover any output or eval cell in the grid and click the rerun icon. Useful when one row failed transiently or you've tweaked a single configuration. - - **Rerun a column**: from the column header, choose **Run all cells in the column** or **Run only failed cells in the column**. Failed-only is the fastest way to recover from API hiccups without redoing successful work. - - **Rerun an eval**: re-execute a single eval across all rows after changing its config or mapping, without re-generating any model outputs. - - ![Update](/screenshot/product/dataset/how-to/experiments-in-dataset/9.png) - - - - Open the **Compare** view to see how every configuration performed. Set weights (0-10) for each eval score and for response time, completion tokens, and total tokens. The system normalizes the metrics, computes an overall rating per configuration, and ranks them so the winner is clear. Adjust the weights to match what matters for your use case (e.g. prioritize quality over cost) and the ranking updates in place. - - - -## Tips - -- **Use published versions**: experiments only run published prompt and agent versions. Publish the version you want to test before importing it. -- **Mix prompts and agents**: an **LLM** experiment can contain prompts and agents side by side, scored against the same evals. Useful when you're deciding whether an agent is worth the extra complexity over a prompt. TTS, STT, and Image experiments accept prompts only. -- **Failed-only rerun**: when transient failures (rate limits, network blips) leave a few cells errored, use the failed-only rerun on the column to recover them without redoing successful rows. - -## Next Steps - - - - Add individual records or bulk import data rows to your dataset - - - Extend your dataset structure with additional data fields - - - Test and execute prompts against your dataset entries - - - Add metadata and annotations to enrich your dataset - - - Create another dataset using SDK, file upload, or synthetic generation - - diff --git a/src/pages/docs/dataset/features/run-prompt.mdx b/src/pages/docs/dataset/features/run-prompt.mdx deleted file mode 100644 index 7583abf3..00000000 --- a/src/pages/docs/dataset/features/run-prompt.mdx +++ /dev/null @@ -1,126 +0,0 @@ ---- -title: "Run Prompt in a Dataset: Generate LLM Columns in Future AGI" -description: "Add a dynamic column to your dataset by running an LLM, TTS, STT, or image generation model on every row using a prompt with column placeholders." ---- - -## About - -Run Prompt lets you add a new column to your dataset that is filled by a model (LLM, Text-to-Speech, Speech-to-Text, or Image Generation). You define a prompt (messages with placeholders that pull from other columns), pick a model and settings, and the system runs the prompt on each row and writes the model output into that column. The result is a [dynamic column](/docs/dataset/concept/dynamic-column) of responses you can use for evals, comparison, or export. - -## When to use - -- **Generate answers or text**: Use an LLM to answer questions, summarize, or complete text per row (e.g. a column of questions produces a column of answers). -- **Produce audio**: Use Text-to-Speech to turn a text column into an audio column (e.g. scripts to voice clips). -- **Transcribe audio**: Use Speech-to-Text to turn an audio column into a text column for evals or search. -- **Batch test a prompt**: Run the same prompt across many rows to see how the model behaves and then run evals on the outputs. -- **Generate images**: Use Image Generation to create images from text (or text + image) per row; the new column stores image URLs. -- **Structured output**: Use response format (e.g. JSON schema) to get structured fields (object, array) in the new column for downstream use. - -## How to - - - - Click the "Run Prompt" button (e.g. in the top-right or dataset toolbar) to add a new run-prompt column. This creates a dynamic column that will store the model output for each row. - ![Run Prompt](/screenshot/product/dataset/how-to/run-prompt-in-dataset/1.png) - - - - Enter a name for the prompt. This name is used as the new column name in your dataset. Each row will have one cell in this column holding the model response for that row. - ![Run Prompt](/screenshot/product/dataset/how-to/run-prompt-in-dataset/2.png) - - - - Select the type of task, then pick the model to use. Models are filtered by type; you need an API key (or custom model) for the chosen provider. - - - Choose **LLM** for text generation (chat). Use for Q&A, summarization, or any text-in, text-out task. Select a chat model from the list; ensure the provider has an API key configured. - ![LLM](/screenshot/product/dataset/how-to/run-prompt-in-dataset/3.png) - - Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models. - - - - Choose **Text-to-Speech** to generate audio from text. The prompt output column will store audio (e.g. URLs). You can configure voice and format for supported TTS models. - ![Text-to-Speech](/screenshot/product/dataset/how-to/run-prompt-in-dataset/4.png) - - Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models. - - - - Choose **Speech-to-Text** to transcribe audio into text. Use when a column contains audio; the model output will be text in the new column. - ![Speech-to-Text](/screenshot/product/dataset/how-to/run-prompt-in-dataset/5.png) - - Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models. - - - - Choose **Image Generation** to create images from text (or image + text) prompts. The prompt output column will store image URLs. Select an image-generation model and ensure the provider has an API key configured. - ![Image Generation](/screenshot/product/dataset/how-to/run-prompt-in-dataset/6.png) - - Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models. - - - - - - - Define the prompt as a list of messages with roles: - - - **System** (optional): Instructions that guide the model's behavior and set context. - - **User** (required): The main input message. This role is required for the prompt to work. - - Use `{{column_name}}` placeholders to pull values from other columns. At runtime, these are replaced by the cell value for each row. - - **Example:** - ``` - System: You are a helpful assistant that summarizes content. - - User: Please summarize the following text: {{article_text}} - ``` - - **JSON dot notation**: For JSON columns, access nested fields directly: - ``` - User: Based on this prompt: {{config.prompt}}, generate a response that addresses {{config.topic}} - ``` - - `{{config.prompt}}` accesses the `prompt` field within the `config` JSON column. - - - - Adjust model parameters such as temperature, max tokens, top_p, and other settings to fine-tune the model's behavior according to your needs. - - - - Add tools or functions that the model can use during execution. This enables the model to perform specific actions or access external capabilities. - - - - Set the concurrency level to control how many prompt executions run in parallel. Higher concurrency speeds up processing but may consume more resources. - - - - Click the "Run" button to execute the prompt across your dataset. The responses will be generated and saved as a new dynamic column in your dataset. - - - - - -## Next Steps - - - - Add individual records or bulk import data rows to your dataset - - - Extend your dataset structure with additional data fields - - - Design and run controlled experiments to compare approaches - - - Add metadata and annotations to enrich your dataset - - - Create another dataset using SDK, file upload, or synthetic generation - - diff --git a/src/pages/docs/dataset/guides/add-columns.mdx b/src/pages/docs/dataset/guides/add-columns.mdx new file mode 100644 index 00000000..92f0f404 --- /dev/null +++ b/src/pages/docs/dataset/guides/add-columns.mdx @@ -0,0 +1,67 @@ +--- +title: "Add columns" +description: "Create a static column for values you set yourself, or a dynamic one that computes them, from the same panel." +--- + +Every column starts in the same panel, whether it holds values you type in or values a method computes for you. Open it, pick a type, name the column, and either save it right away or test it first. + +From the dataset's **Data** tab, click **Add Column**. The Add Columns panel opens with a filter on the left: **All**, **Static Columns**, or **Dynamic Columns**. Pick a filter (or search by name), then click the type you want. A [static column](/docs/dataset/concepts/static-and-dynamic-columns) creates immediately once you name it; a dynamic column opens a fuller form for the method's settings, and lets you test it before you commit. + +## Add a static column + +Static columns are the fastest path: pick a data type, name it, and it's in the grid ready to fill in. + + + + Filter to **Static Columns** and click the data type you want, for example **Text**. The full set of types, and what each one stores, is in [Limits & Data Types](/docs/dataset/reference/limits-and-data-types). + + + A small panel opens with a **Column name** field. Name it `reviewer_notes` and click **Add Column**. + + + +The column appears in the grid right away, empty, ready for you to fill in cell by cell. + +## Add a dynamic column + +Dynamic columns point at a method instead of holding values you type. The example here uses **Classification**, which reads another column's text and sorts each row into one of the labels you define. Run Prompt is a dynamic column too, but it gets its own walkthrough in [Run a prompt on every row](/docs/dataset/guides/run-a-prompt-on-every-row); every other method is cataloged in [Dynamic column methods](/docs/dataset/reference/dynamic-column-methods). + + + + Filter to **Dynamic Columns** and click **Classification**. + + + Name the column `query_topic`, then select the column to classify: `user_query`. + + + Add each label the model can choose from, for example `Billing`, `Technical`, and `Account`. + + + Choose the model that runs the classification, and set how many rows it processes at once. + + + Click **Test** to preview the labels it would assign, without saving anything yet. Once it looks right, click **Create New Column**. + + + +Create New Column starts Classification on every row, filling `query_topic` in as it works through the dataset. + +## Validation you'll hit + +- Column names cap at 255 characters +- A name that's already used in this dataset is rejected +- Two columns in the same request can't share a name, which only comes up when you add more than one column at once, for example through the SDK + +## Dive deeper + + + + Walk the Run Prompt method end to end + + + Every other method you can point a dynamic column at + + + Rename, retype, or delete a column after it's in + + diff --git a/src/pages/docs/dataset/guides/add-rows.mdx b/src/pages/docs/dataset/guides/add-rows.mdx new file mode 100644 index 00000000..31f422f7 --- /dev/null +++ b/src/pages/docs/dataset/guides/add-rows.mdx @@ -0,0 +1,182 @@ +--- +title: "Add rows" +description: "Five ways to add new examples to a dataset you've already created." +--- + +Adding rows brings more examples into a [dataset](/docs/dataset/concepts/understanding-datasets) that already exists. If you don't have a dataset yet, start with [Create a dataset](/docs/dataset/guides/create-a-dataset). Columns stay put unless imported data brings a name the dataset doesn't already have. Whichever path you pick, new rows always land after whatever's already in the dataset, appended in the order you add them. + +## Add rows using the SDK + +Use this when you're scripting the setup or pushing rows from your own pipeline. + + + +```python Python +# pip install futureagi + +import os +from fi.datasets import Dataset +from fi.datasets.types import Cell, Row + +os.environ["FI_API_KEY"] = "" +os.environ["FI_SECRET_KEY"] = "" + +dataset = Dataset.get_dataset_config("support-agent-eval") + +rows = [ + Row(cells=[ + Cell(column_name="user_query", value="How do I reset my password?"), + Cell(column_name="response_quality", value=7), + Cell(column_name="is_helpful", value=True), + ]), + Row(cells=[ + Cell(column_name="user_query", value="What's your refund policy?"), + Cell(column_name="response_quality", value=9), + Cell(column_name="is_helpful", value=True), + ]), +] + +dataset = dataset.add_rows(rows=rows) +``` + +```typescript Typescript +import { Dataset, createRow, createCell } from "@future-agi/sdk"; + +process.env["FI_API_KEY"] = ""; +process.env["FI_SECRET_KEY"] = ""; + +async function main() { + const dataset = await Dataset.open("support-agent-eval", { createIfMissing: false }); + + const rows = [ + createRow({ + cells: [ + createCell({ columnName: "user_query", value: "How do I reset my password?" }), + createCell({ columnName: "response_quality", value: 7 }), + createCell({ columnName: "is_helpful", value: true }), + ], + }), + createRow({ + cells: [ + createCell({ columnName: "user_query", value: "What's your refund policy?" }), + createCell({ columnName: "response_quality", value: 9 }), + createCell({ columnName: "is_helpful", value: true }), + ], + }), + ]; + + await dataset.addRows(rows); +} + +main(); +``` + +```bash Curl +curl --request POST \ + --url https://api.futureagi.com/model-hub/develops//add_rows/ \ + --header 'content-type: application/json' \ + --header 'X-Api-Key: ' \ + --header 'X-Secret-Key: ' \ + --data '{ + "rows": [ + { + "cells": [ + { "column_name": "user_query", "value": "How do I reset my password?" }, + { "column_name": "response_quality", "value": 7 }, + { "column_name": "is_helpful", "value": true } + ] + }, + { + "cells": [ + { "column_name": "user_query", "value": "What'\''s your refund policy?" }, + { "column_name": "response_quality", "value": 9 }, + { "column_name": "is_helpful", "value": true } + ] + } + ] +}' +``` + + + +A `Row` accepts an `order`, but new rows are always appended after the last existing one, whatever order you pass, so it isn't how you control placement. See the [Datasets SDK](/docs/sdk/datasets) page for the full `Dataset` class. + +Get your API key and secret key [here](https://app.futureagi.com/dashboard/keys). + +Every drawer path below starts the same way: on the dataset's **Data** tab, click **Add Row** to open the drawer, which has a tile for each way to add rows. + +## Add a row from the grid + +Use this for typing in one or two examples by hand. + + + + Pick **Add empty row** and choose how many to add. + + + New rows appear at the bottom of the table with empty cells. Double-click a cell to enter a value, and repeat for each row. + + + +Instead of starting blank, you can also duplicate rows you already have: select them in the grid, click **Duplicate** in the toolbar, then set the number of copies. + +## Copy rows from another dataset or experiment + +Use this when the rows you need already exist somewhere else. + + + + In the Add Row drawer, select **Add from existing model dataset or experiment**, then pick the source: another dataset, or an experiment snapshot. + + + Map each source column to a column in this dataset. Only mapped columns copy over; anything left unmapped in the source is skipped. + + + +## Import rows from Hugging Face + +Use this to pull in a public dataset instead of typing examples by hand. + + + + In the Add Row drawer, select **Import from Hugging Face**, then search for the dataset and pick the subset next to the split. Set how many rows to import. + + + Start the import. Each Hugging Face feature is matched to a column by name; a name that doesn't exist yet becomes a new column, backfilled with empty cells on the rows that already existed. + + + +## Add rows from a file + +Use this for a CSV, Excel, JSON or JSONL export you already have. + + + + In the Add Row drawer, select **Upload a file (JSONl/ JSON/ CSV)**, then upload it. + + + Column names in the file are matched to existing columns by name. A name that isn't already a column gets created, backfilled with empty cells on the rows that already existed. + + + +## Caps you'll hit + +- Adding empty rows from the grid: the picker tops out at 10 at a time, up to 100 per request via the API +- Duplicating a row: at most 100 copies +- Uploading a file: 25 MB, restricted to `.csv`, `.xls`, `.xlsx`, `.json`, `.jsonl` + +The full set of dataset and column limits is in [Limits & Data Types](/docs/dataset/reference/limits-and-data-types). + +## Dive deeper + + + + Extend the dataset with a static or dynamic column + + + Turn a prompt into a column of model output + + + Rename, edit, or delete what's already in the grid + + diff --git a/src/pages/docs/dataset/guides/create-a-dataset.mdx b/src/pages/docs/dataset/guides/create-a-dataset.mdx new file mode 100644 index 00000000..8abce2db --- /dev/null +++ b/src/pages/docs/dataset/guides/create-a-dataset.mdx @@ -0,0 +1,172 @@ +--- +title: "Create a dataset" +description: "Six ways to create a dataset and get data into it." +--- + +A dataset starts as just a name in your organization; everything else comes from how you fill it. Future AGI gives you six ways to do that, each landing you on the same [dataset](/docs/dataset/concepts/understanding-datasets) table of rows and columns you can keep editing afterward. + +Every method starts the same way: from the dataset list (**Dataset** in the left nav), click **Add Dataset**. That opens a panel with six tiles: + +| Method | Reach for it when | +| --- | --- | +| Add data using SDK | You're scripting the setup or pushing rows from your own pipeline | +| Upload a file (JSON, CSV) | You already have test cases in a CSV, Excel, or JSON export | +| Create Synthetic Data | You don't have real data yet, but know the shape of the examples you need | +| Add datasets Manually | You're hand-building a small set and want an empty grid to fill in | +| Import from HuggingFace | The data you need already exists as a Hugging Face dataset | +| Add from existing model dataset or experiment | You want to branch off a dataset or experiment you already have | + +Whichever method you pick, the dataset's name must be unique inside your organization. A name that's already taken is rejected. For the full set of dataset and column limits, see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types). + +## Add data using the SDK + +For scripting dataset creation, or when you'd rather write rows in code than click through the grid. + +In the Add dataset panel, pick **Add data using SDK** and name the dataset. Future AGI creates an empty dataset and drops you on its Data tab, which shows the dataset's name, ID, API key, and secret key alongside a ready-to-run code snippet. Copy the snippet below and run it against your dataset to add [columns](/docs/dataset/concepts/static-and-dynamic-columns) and rows. + +An empty dataset starts with at most 10 rows; add the rest with the SDK. + + + +```python Python +# pip install futureagi + +import os +from fi.datasets import Dataset +from fi.datasets.types import Cell, Column, DataTypeChoices, Row, SourceChoices + +os.environ["FI_API_KEY"] = "" +os.environ["FI_SECRET_KEY"] = "" + +# Get the dataset you just created +dataset = Dataset.get_dataset_config("support-agent-eval") + +# Define columns +columns = [ + Column(name="user_query", data_type=DataTypeChoices.TEXT, source=SourceChoices.OTHERS), + Column(name="response_quality", data_type=DataTypeChoices.INTEGER, source=SourceChoices.OTHERS), + Column(name="is_helpful", data_type=DataTypeChoices.BOOLEAN, source=SourceChoices.OTHERS), +] + +# Define rows +rows = [ + Row(order=1, cells=[ + Cell(column_name="user_query", value="What is machine learning?"), + Cell(column_name="response_quality", value=8), + Cell(column_name="is_helpful", value=True), + ]), + Row(order=2, cells=[ + Cell(column_name="user_query", value="Explain quantum computing"), + Cell(column_name="response_quality", value=9), + Cell(column_name="is_helpful", value=True), + ]), +] + +dataset = dataset.add_columns(columns=columns) +dataset = dataset.add_rows(rows=rows) +``` + +```typescript Typescript +import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk"; + +process.env["FI_API_KEY"] = ""; +process.env["FI_SECRET_KEY"] = ""; + +async function main() { + // Get the dataset you just created + const dataset = await Dataset.open("support-agent-eval"); + + // Define columns + const columns = [ + { name: "user_query", dataType: DataTypeChoices.TEXT }, + { name: "response_quality", dataType: DataTypeChoices.INTEGER }, + { name: "is_helpful", dataType: DataTypeChoices.BOOLEAN }, + ]; + + // Define rows + const rows = [ + createRow({ + cells: [ + createCell({ columnName: "user_query", value: "What is machine learning?" }), + createCell({ columnName: "response_quality", value: 8 }), + createCell({ columnName: "is_helpful", value: true }), + ], + }), + createRow({ + cells: [ + createCell({ columnName: "user_query", value: "Explain quantum computing" }), + createCell({ columnName: "response_quality", value: 9 }), + createCell({ columnName: "is_helpful", value: true }), + ], + }), + ]; + + await dataset.addColumns(columns); + await dataset.addRows(rows); +} + +main(); +``` + +```bash Curl +curl --request POST \ + --url https://api.futureagi.com/model-hub/develops//add_columns/ \ + --header 'X-Api-Key: ' \ + --header 'X-Secret-Key: ' \ + --header 'content-type: application/json' \ + --data '{ + "new_columns_data": [ + {"name": "user_query", "data_type": "text"}, + {"name": "response_quality", "data_type": "integer"}, + {"name": "is_helpful", "data_type": "boolean"} + ] + }' +``` + + + +See the [Datasets SDK reference](/docs/sdk/datasets) for the full `Dataset` API. + +## Upload a file + +For bringing in test cases you already have as a file, instead of typing them in. + +In the Add dataset panel, pick **Upload a file (JSON, CSV)** and name the dataset. Drop or browse to your file: accepted formats are `.csv`, `.xls`, `.xlsx`, `.json`, and `.jsonl`, up to 25 MB. The dataset appears on your list right away. Future AGI processes the file in the background with a visible progress state until it's done. + +## Create synthetic data + +For when you don't have real data yet, but know the shape of the examples you need. + +In the Add dataset panel, pick **Create Synthetic Data** and name the dataset. From there, Future AGI walks you through describing the schema and generates rows for you. See [Synthetic Data](/docs/dataset/concepts/synthetic-data) if you want to regenerate later. + +## Add a dataset manually + +For hand-building a small dataset from scratch when you already know its shape. + +In the Add dataset panel, pick **Add datasets Manually**, name the dataset, and choose how many rows and how many columns to start with, up to 100 of each. Future AGI creates the dataset with that many empty rows and columns, ready for you to fill in. + +## Import from Hugging Face + +For pulling in a Hugging Face dataset instead of typing test cases by hand. + +In the Add dataset panel, pick **Import from HuggingFace** and paste the Hugging Face dataset ID. Click **Load Dataset**, then pick the **Subset** and **Split** you want. Name the new dataset to finish. Only the first 100 rows of the source are ingested. + +## Add from an existing dataset or experiment + +For branching off a dataset or experiment you already have. + +In the Add dataset panel, pick **Add from existing model dataset or experiment** and choose the dataset or experiment you want to copy from. Choose whether to bring over **Import Data** or **Import data and prompt configuration**, then select which columns to include. Name the new dataset to finish. + +## Dive deeper + + + + Put more records into a dataset that already exists + + + Extend a dataset with a static or dynamic column + + + Turn a prompt into a column of model output + + diff --git a/src/pages/docs/dataset/guides/manage-datasets.mdx b/src/pages/docs/dataset/guides/manage-datasets.mdx new file mode 100644 index 00000000..b496cc7d --- /dev/null +++ b/src/pages/docs/dataset/guides/manage-datasets.mdx @@ -0,0 +1,78 @@ +--- +title: "Manage datasets" +description: "Find, duplicate, export, and clean up datasets and their columns once the data is already in" +--- + +Once a dataset has data in it, the work shifts from adding rows to keeping the table itself in order: finding the right dataset, making a copy, pulling data out, fixing a column, or getting rid of what you no longer need. This guide covers all of that on an existing dataset. For putting data in, see [Create a dataset](/docs/dataset/guides/create-a-dataset), [Add rows](/docs/dataset/guides/add-rows), and [Add columns](/docs/dataset/guides/add-columns). + +## Find a dataset in the list + +- Only **Dataset Name** and **Datapoints** are sortable: click either header to reorder the list by it +- Use the search box above the table to filter by name +- If nothing matches, the list shows **No datasets found** instead of an empty table + +The list is paginated; see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for the page size. + +## Act on several datasets at once + +Select one or more datasets with the row checkboxes and a bulk action bar takes over the toolbar, showing **{'{n}'} Selected** alongside **Delete** and **Cancel**. Select exactly one dataset and **Duplicate** joins the bar; select two or more and **Duplicate** is replaced by **Compare**, which opens a **Select Base Columns** drawer where you pick the one column the selected datasets share. Cancel clears the selection without doing anything. + + +Every control that changes a dataset (duplicate, edit, and delete) is gated on your dataset permission. In the bulk action bar, this means the controls show up disabled rather than doing nothing when clicked. In the grid, Edit Column Name, Edit Column Type, and Delete Column don't appear in the column header menu at all for a viewer, and cells simply stop being editable. Downloading isn't role-gated, but the download button is disabled when the dataset has no data, or when a synthetic dataset is still processing. + + +## Duplicate a dataset + +Check the dataset's row checkbox, then click **Duplicate** in the bulk action bar to open the **Duplicate Dataset** dialog. Enter a name for the copy in **Enter Dataset Name**, for example `support-agent-eval-copy`, then click **Create**. The dialog validates the name before it lets you proceed, and a success toast confirms once the copy exists. Click **Cancel** to back out without duplicating anything. + +The duplicate is a separate dataset from the moment it's created: editing it doesn't touch the original, and deleting one doesn't touch the other. It isn't a full copy, though: duplicating only carries over rows and [static columns](/docs/dataset/concepts/static-and-dynamic-columns). Dynamic columns, and the computed values in them, don't come across, so a duplicated dataset can have fewer columns than the one it was made from. + +## Export a dataset + +Click the download icon in the dataset's toolbar to export it. A **Download has been started...** toast appears immediately, followed by **Dataset downloaded successfully** once the file is ready. + +## Edit data in the grid + +Inside a dataset, the grid supports these changes directly, without leaving the Data tab. Renaming a column, changing its data type, and deleting it all live in the column header menu, under **Edit Column Name**, **Edit Column Type**, and **Delete Column**. + +| Action | What it does | +|---|---| +| Rename a column (`response_quality` to `quality_score`, for example) | The cell values stay the same, but SDK and cURL calls that reference the old name break | +| Change a column's data type | Reinterprets how the column's stored values are treated | +| Edit a cell in a static column | Overwrites that one cell's value; audio and persona cells can't be edited this way | +| Delete a column | Removes the column and every cell in it | + + +A column has no identifier besides its name, so a rename changes what your integrations have to send. `add_rows` and other SDK or cURL calls key each cell by `column_name`; if a call still references `response_quality` after you rename it to `quality_score`, that call fails until it's updated. + + +To delete rows, select them with their row checkboxes and confirm the delete action in the grid toolbar. + +Cells in a dynamic column can't be edited at all: the column is managed by whatever produces it. + +## Delete a dataset + +Select one or more datasets with the row checkboxes, then click **Delete** in the bulk action bar. The dialog title switches between **Delete Dataset** and **Delete Datasets** depending on how many you selected. Confirm with **Delete**, or back out with **Cancel**. A success toast confirms once it's done. + +Bulk delete is capped at 50 datasets per request. Deleting more than that means running the action in batches. + +## What deleting actually removes + +Deletes here are final: once you delete a row, a column, or a dataset, it's gone. Deleting rows removes the selected rows and every cell in them. Deleting a column removes the column and every cell in it. + + +Deleting a dynamic column also deletes the producer behind it, whether that's a prompt run or an eval, and anything else that was derived from it. It isn't just the column that disappears. + + +Deleting a dataset also deletes its [experiments](/docs/dataset/guides/run-an-experiment). + +## Dive deeper + + + + What a column's producer is, and why deleting one takes it along + + + Exact numbers for every cap on this page + + diff --git a/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx b/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx new file mode 100644 index 00000000..adad9356 --- /dev/null +++ b/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx @@ -0,0 +1,88 @@ +--- +title: "Run a prompt on every row" +description: "Turn a prompt into a column of model output, one response per row." +--- + +**Run Prompt** fills a new [dynamic column](/docs/dataset/concepts/static-and-dynamic-columns) by running a prompt against every row of a dataset that already exists. You write the prompt once, referencing other columns as inputs, and Future AGI runs it row by row until the whole column is filled. You'll need a dataset with the input columns your prompt will reference. + + + + On a dataset's **Data** tab, click **Run Prompt**. This starts a new column and opens the panel where you build the prompt and pick a model. + + + + In the **Name** field (placeholder "Prompt Name"), name the new column `model_response` for this example. It's the first field in the panel, above the model type options, and it becomes the name of the column every row's response lands in. + + + + Run Prompt supports four kinds of models, each shaped the same way: pick a type, then pick the specific model from that type's list. + + | Model type | Input | Output | + | --- | --- | --- | + | LLM | The prompt you build next | Text | + | Text-to-Speech | A text column referenced in the prompt | Audio | + | Speech-to-Text | An audio column | Transcribed text | + | Image Generation | A single image prompt | An image | + + LLM prompts are the message-based kind covered next. For LLM models, don't see the model you need? [Register a custom model](/docs/evaluation/guides/custom-models) and it joins the same list. + + + + An LLM prompt is a list of messages with roles: + + - **System** (optional): instructions that set the model's behavior and context + - **User** (required): the input message, built from your dataset's columns + + Use `{{column_name}}` inside a message to pull that column's value for the current row. Take a dataset with a `user_query` column and a `customer_context` JSON column: + + **System** + ``` + You are a support assistant that helps resolve customer tickets. + ``` + + **User** + ``` + A customer asked: {{user_query}}. Their plan is {{customer_context.plan}}. Write a helpful response. + ``` + + `{{user_query}}` pulls that row's plain text value. `{{customer_context.plan}}` uses dot notation to reach the `plan` field inside the `customer_context` JSON column, without pulling in the rest of that column's value. + + Text-to-Speech, Speech-to-Text, and Image Generation prompts are simpler, since each is a single input instead of a message list: + + - **Text-to-Speech**: in the Prompt Input box, reference the text column to speak, for example `{{script_text}}`, and choose a Voice, which is required + - **Speech-to-Text**: pick a column in the Voice Input section's Column dropdown, which lists your audio columns; selecting one fills the message for you + - **Image Generation**: write the prompt describing the image directly in the Image Prompt field + + + + Set how many rows run at once, from 1 to 10. It defaults to 5. + + + + Adjust generation parameters such as temperature, top P, max tokens, presence and frequency penalty, and response format, if the defaults don't fit your prompt, from the options button beside **Select Model**. + + Attach tools the model can call while it runs, if your prompt needs them, in the **Tool Configuration** accordion above **Concurrency**. + + + + Click **Run**. Future AGI works through the dataset row by row and writes each response into the new column. Watch a row's cell to see it complete; the column is done once every cell has filled. + + + +## What lands in the column + +While a row's call is in flight, its cell shows a loading placeholder until the response lands. If the call fails, its cell shows an error. Otherwise the cell fills with the response, and each LLM cell also records its token counts and response time. + +## Dive deeper + + + + Add a static or dynamic column from the Data tab + + + Every other producer a dynamic column can point at + + + Compare prompts and models against each other using evals + + diff --git a/src/pages/docs/dataset/guides/run-an-experiment.mdx b/src/pages/docs/dataset/guides/run-an-experiment.mdx new file mode 100644 index 00000000..4b87234d --- /dev/null +++ b/src/pages/docs/dataset/guides/run-an-experiment.mdx @@ -0,0 +1,67 @@ +--- +title: "Run an experiment" +description: "Set up an experiment, run it across your dataset, and use Choose winner to pick the best configuration." +--- + +An **experiment** runs every [prompt or agent](/docs/prompt) and model combination you set up against the same [dataset](/docs/dataset/concepts/understanding-datasets), scored by the [evals](/docs/evaluation) you attach, so you can compare configurations side by side instead of testing them one at a time. + +## The experiment grid + +One experiment lays your dataset's rows against every configuration you add. A **configuration** pairs one prompt or agent with one model, so attaching three models to the same prompt gives you three configurations, one column each. Every eval you attach scores every configuration against the same rows, which is what makes the columns comparable. + + GRID{{"Experiment grid"}} + PA1["Prompt or agent"] --> CFG1["Configuration A"] + MD1["Model"] --> CFG1 + PA2["Prompt or agent"] --> CFG2["Configuration B"] + MD2["Model"] --> CFG2 + CFG1 --> GRID + CFG2 --> GRID + EV["Evals"] -->|"scores every column"| GRID`} /> + +## Build the experiment + +Click **Experiment** on the dataset to open the creation flow. It's a three-step form. + + + + Name the experiment and choose its type: **LLM**, **TTS**, **STT**, or **Image**. The type decides the output format and which models you can attach. + + + Add the prompts or agents you want to compare and attach a model to each. Every prompt/agent-model pair becomes its own configuration column. LLM experiments can mix prompts and agents in the same run and attach tools to a prompt, useful for deciding whether an agent earns its extra complexity over a plain prompt; TTS, STT, and Image experiments take prompts only. + + + Optionally pick a column to compare outputs against as a baseline, then add the evals that will score every configuration. + + + +Click **Run Experiment**, and every row runs against every configuration, with each output scored by your evals as it comes in. + + +An eval you add after the run sits queued for a few seconds before it starts scoring. + + +## Stop and rerun + +Each experiment stops and reruns independently. Stop a running one without touching the others in the dataset, and rerun a completed, failed, or cancelled one later without setting it up again, though rerunning overwrites its existing results. Select more than one experiment at a time to rerun or delete them together. + +## Choose a winner + +The experiment summary already lists every configuration. Once every configuration has a score, click **Choose winner** to open Winner Settings, where you set the importance of Average Response Time, Completion tokens, Total tokens, and each eval. Click **Save & Run** and the summary marks the winning configuration. + +## Tips + +- **Failed-only rerun**: when transient failures (rate limits, network blips) leave a few cells errored, use the failed-only rerun on the column to recover them without redoing successful rows + +## Dive deeper + + + + Add human labels to rows once the experiment tells you where to look + + + Duplicate, export, or clean up a dataset after you're done experimenting + + diff --git a/src/pages/docs/dataset/index.mdx b/src/pages/docs/dataset/index.mdx index ab2b51a3..19097f24 100644 --- a/src/pages/docs/dataset/index.mdx +++ b/src/pages/docs/dataset/index.mdx @@ -1,65 +1,36 @@ --- -title: "Future AGI Datasets: Evaluation and Experimentation Layer" -description: "Structured tables of examples for prompts, evaluations, and experiments. Create from file uploads, SDK, production traces, or synthetic generation." +title: "Overview" +description: "What a dataset is made of, where the data comes from, and where to go next" --- -## About +## What is a Dataset? -Datasets are the core data layer for evaluation and experimentation in Future AGI. Each dataset is a table with columns (e.g. "user query", "expected answer", "score"), rows (one row per example), and cells (the value in each column for each row). +A **dataset** is a table of examples. [Prompts](/docs/prompt) and [evals](/docs/evaluation) run against it and write their results back as new columns. [Experiments](/docs/dataset/guides/run-an-experiment) and [optimization](/docs/optimization) run on the same rows in their own tabs. You reach it from **Dataset** in the left nav. -Datasets are the single source of truth that prompts, evaluations, experiments, and optimizations run on. You can create them from file uploads, the SDK, observed production traces, or synthetic generation. +## Columns, rows, and cells - +Each dataset is a grid: **columns** define what you're capturing (a query, an expected answer, a score), **rows** are the individual examples, and a **cell** holds the value where a row meets a column. A column's values come from you directly, or get filled in automatically by running something against the dataset. See [Static & Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns) for the difference. -## Column Types +## Where the data comes from -Datasets support two types of columns: +- **[File upload](/docs/dataset/guides/create-a-dataset#upload-a-file)**: bring in a CSV, JSON, or JSONL file +- **[The SDK](/docs/dataset/guides/create-a-dataset#add-data-using-the-sdk)**: push rows from your own code +- **[Synthetic generation](/docs/dataset/guides/create-a-dataset#create-synthetic-data)**: describe a schema and get realistic rows back +- **[Hugging Face](/docs/dataset/guides/create-a-dataset#import-from-hugging-face)**: import an existing dataset by name +- **[An existing dataset or experiment](/docs/dataset/guides/create-a-dataset#add-from-an-existing-dataset-or-experiment)**: branch off data you already have in Future AGI +- **[Observe](/docs/observe) traces**: turn real production traffic into rows +- **[Manual entry](/docs/dataset/guides/create-a-dataset#add-a-dataset-manually)**: add rows and columns by hand -- **Static columns**: Data you add directly, either manually, via file upload, or through the SDK. These hold your inputs, expected outputs, ground truth labels, or any fixed data. -- **Dynamic columns**: Generated on-the-fly by running a prompt, evaluation, or model against your dataset rows. For example, running GPT-4o on every row creates a dynamic column with the model's responses. +## Dive deeper -This distinction matters because dynamic columns let you add model outputs, evaluation scores, and computed fields to your dataset without duplicating data. - -## How Datasets Connect to Other Features - -- **Evaluation**: Run 70+ built-in metrics across your dataset rows to score model outputs. Results are stored as new columns. [Learn more](/docs/evaluation) -- **Experiments**: Compare two prompts or models by running both against the same dataset and comparing scores side by side. [Learn more](/docs/dataset/features/experiments) -- **Optimization**: Use datasets as the training ground for prompt optimization algorithms. [Learn more](/docs/optimization) -- **Observe**: Build datasets from production traces to test against real user queries. [Learn more](/docs/observe) - -## Getting Started with Datasets - - - - Create datasets using SDK integration, file upload, or synthetic data generation - - - Learn how to add individual records or bulk import data rows + + + Upload a file, use the SDK, or generate one from a schema - - Extend your dataset structure with additional data fields + + How order, storage, and ownership work under the hood - - Test and execute prompts against your dataset entries - - - Design and conduct controlled experiments to compare approaches - - - Add metadata and annotations to enrich your dataset + + What changes once a column is dynamic: status, edits, and reruns - -## Next Steps - -- [Understanding Datasets](/docs/dataset/concept/understanding-dataset): Deeper dive into dataset concepts, column types, and best practices -- [Generate Synthetic Data](/docs/quickstart/generate-synthetic-data): Create realistic datasets from scratch when real data is unavailable -- [Import from HuggingFace](/docs/cookbook/quickstart/huggingface-dataset-import): Bring existing HuggingFace datasets into Future AGI - diff --git a/src/pages/docs/dataset/reference/dynamic-column-methods.mdx b/src/pages/docs/dataset/reference/dynamic-column-methods.mdx new file mode 100644 index 00000000..0ee11f7a --- /dev/null +++ b/src/pages/docs/dataset/reference/dynamic-column-methods.mdx @@ -0,0 +1,119 @@ +--- +title: "Dynamic column methods" +description: "Reference catalog of every dynamic column method, one entry per method" +--- + +**+ Add Columns > Dynamic Columns** opens the methods that compute a [dynamic column](/docs/dataset/concepts/static-and-dynamic-columns)'s values instead of you typing them in. + +## Name, Concurrency, and Status + +Every method's form asks for a **Name** for the resulting column, and every method except Conditional Node also asks for a **Concurrency**: how many rows to process in parallel. Retrieval's forms note that leaving Concurrency blank falls back to the platform's own system configuration. Conditional Node's form only has a **Name** and the branch list; each branch's operation carries its own Concurrency field instead, and skips its own Name field since the conditional column already has one. + +The column's status is `Running` while the method runs, then lands on `Completed` or `Failed`. Each cell carries its own status of `running`, `pass`, or `error`. See [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for the full list of status values. + +## Run Prompt + +Produces a value from one inference call per row, using a prompt template. + +| Field | Description | +|---|---| +| Prompt | One or more messages (system, user, assistant); reference other columns with `{{column_name}}` | +| Model type | LLM, Text-to-Speech, Speech-to-Text, or Image Generation | +| Model | The model to run | + +The resulting column records source `run_prompt`. + +## Retrieval + +Produces a value by querying a vector database index and returning matching chunks for each row. + +Choose a **Vector Database**: Pinecone, Qdrant, or Weaviate. These fields are shared across all three: + +| Field | Description | +|---|---| +| Column | The column whose value is sent as the query | +| Number of chunks to fetch | How many top matches to fetch (topK) | +| Embedding Configuration | Type (OpenAI, Hugging Face, or Sentence Transformers) and Model | +| Key to extract | The field to pull from each retrieved match | +| Vector Length | The dimension the embedding model outputs; must match the index's configured dimension | + +Each provider adds a few fields of its own, including its own API key field: + +| Provider | Additional fields | +|---|---| +| Pinecone | Pinecone API Key, Index Name, Namespace, Query Key | +| Qdrant | Qdrant API Key, Qdrant URL, Collection Name | +| Weaviate | Weaviate Api Key, Weaviate Cluster Url, Collection Name, Search Type (Semantic Search or Hybrid) | + +The resulting column records source `vector_db`. + +## Extract Entities + +Produces a value extracted from a text column, guided by a model. + +| Field | Description | +|---|---| +| Column | The column to extract from | +| Instructions | What to extract | +| Model | The model to run | + +The resulting column records source `extracted_entities`. + +## Extract a JSON Key + +Produces a value pulled out of a JSON column by key. + +| Field | Description | +|---|---| +| Column | A column of type JSON, or an API Call column whose response is JSON | +| JSON Key | The JSONPath-style key to extract, e.g. `age` | + +The resulting column records source `extracted_json`. + +## Classification + +Produces a label assigned to a column's text from your set of categories. + +| Field | Description | +|---|---| +| Column | The column to classify | +| Labels | One or more category labels | +| Model | The model to run | + +The resulting column records source `classification`. + +## API Calls + +Produces a value returned by calling an external HTTP endpoint for each row. + +| Field | Description | +|---|---| +| Add API Endpoint | The endpoint to call; reference other columns with `{{column_name}}` | +| Request Type | GET, POST, PUT, DELETE, or PATCH | +| Params / Headers | Key-value pairs; each value is plain text, a stored secret, or a column reference | +| Request Body | JSON body; reference other columns with `{{column_name}}` | +| Output Type | String, Object, Array, or Number | + +The resulting column records source `api_call`. + +## Conditional Node + +Produces a value chosen by evaluating branches in order: the first branch whose condition is true, or the `else` branch, runs its operation, and that operation's output becomes the cell's value. + +| Field | Description | +|---|---| +| Branches | One `if` (always first), any number of `elif`, and optionally one `else` | +| Condition | Set on every branch except `else`; reference other columns with `{{column_name}}` | +| Select Column Type | Per branch, one of Run Prompt, Retrieval, Extract Entities, Extract JSON Key, Execute Custom Code, Classification, or API Calls, configured with that operation's own fields described on this page | + +The resulting column records source `conditional`. + +## Execute Custom Code + +Produces a value returned by a Python function you write, run once per row. The function must be named `main` and can read any column's value through `kwargs` (`kwargs.get("column_name")`). Execute Custom Code isn't a standalone tile under Dynamic Columns. It's only reachable as an operation inside a Conditional branch, or by editing a column that already runs Python code. + +| Field | Description | +|---|---| +| Code | The `main(**kwargs)` function to run | + +The resulting column records source `python_code`. diff --git a/src/pages/docs/dataset/reference/limits-and-data-types.mdx b/src/pages/docs/dataset/reference/limits-and-data-types.mdx new file mode 100644 index 00000000..5c5283bc --- /dev/null +++ b/src/pages/docs/dataset/reference/limits-and-data-types.mdx @@ -0,0 +1,58 @@ +--- +title: "Limits & Data Types" +description: "Quick-reference tables for Dataset limits, data types, and status values." +--- + +A lookup page for the numbers and enums referenced elsewhere in the Dataset docs: what each column data type stores, the exact limits on names, rows, columns, and requests, and the status values a column or a cell can be in. + +## Column data types + +| Type | Stores | +|---|---| +| `text` | A text value | +| `boolean` | True or false | +| `integer` | A whole number | +| `float` | A decimal number | +| `json` | A JSON object | +| `array` | A JSON array | +| `image` | A single image | +| `images` | Multiple images | +| `datetime` | A date and time value | +| `audio` | An audio file | +| `document` | A document file | +| `persona` | A persona definition | +| `others` | A value that doesn't fit any other type | + +## Dataset and column limits + +| Limit | Value | Applies to | +|---|---|---| +| Dataset name length | 2000 characters, unique within the organization | Every dataset | +| Column name length | 255 characters | Every column you name | +| Manual dataset rows | 100 | Creating a dataset manually | +| Manual dataset columns | 100 | Creating a dataset manually | +| Empty dataset rows | 10 | Creating an empty dataset | +| Empty rows per request | 100 | Adding empty rows to an existing dataset | +| Row duplication | 100 copies | Duplicating a row | +| Bulk delete | 50 items | Deleting datasets in bulk | +| Dataset list page size | 100 datasets | Listing datasets | +| File upload size | 25 MB | Uploading a file to create or add to a dataset | +| File upload formats | `.csv`, `.xls`, `.xlsx`, `.json`, `.jsonl` | Uploading a file to create or add to a dataset | + + + These are per-request and per-object constants. Plan-level quotas, such as the total rows or datasets your organization can hold, are enforced separately by the usage system and aren't part of this table. + + +## Status values + +Column statuses and cell statuses are separate sets, and a column only reaches the ones below. + +| Status | Seen on | Meaning | +|---|---|---| +| `Running` | Column | Set when a column starts an async run, such as Run Prompt, an evaluation, or Retrieval | +| `PartialExtracted` | Column | Set when file upload extraction succeeds for some of a column's cells and fails for others | +| `Completed` | Column | The default status for a new column, and set when a column finishes running successfully | +| `Failed` | Column | Set when a column's processing raises an error, for example during a data type conversion | +| `pass` | Cell | Computed successfully (the default) | +| `running` | Cell | Set while a cell is computing | +| `error` | Cell | Set when a cell fails to compute | diff --git a/src/pages/docs/dataset/troubleshooting.mdx b/src/pages/docs/dataset/troubleshooting.mdx new file mode 100644 index 00000000..c0977632 --- /dev/null +++ b/src/pages/docs/dataset/troubleshooting.mdx @@ -0,0 +1,42 @@ +--- +title: "Dataset FAQ & fixes" +description: "Common dataset questions, and fixes for the errors you hit most" +--- + +## In this page + +The questions people ask most about datasets, and the errors they run into, with a direct fix for each, in the table below. If your answer isn't here, reach out via [support](https://futureagi.com/contact-us). + +## Common errors and fixes + +| Symptom | Cause | Fix | +|---|---|---| +| Upload rejected before it starts | The file isn't `.csv`, `.xls`, `.xlsx`, `.json`, or `.jsonl`, or it's over 25 MB | Convert or split the file to fit; see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for every cap | +| "A dataset with this name already exists in your organization" when creating | Names must be unique per organization, so the clash can be with a dataset you don't have access to see | Pick a different name | +| Upload finished but the dataset shows no rows yet | The file is still processing in the background | Wait it out; the rows land once it finishes | +| Image, audio, or document cells fill in slowly after upload | Media cells upload in batches with retries, not all at once | No action needed, it catches up on its own | +| A dataset built from [Observe](/docs/observe) traces fills in gradually | Spans convert into rows in chunks, not all at once | Wait it out; the row count climbs while the conversion runs | +| A freshly added eval column sits idle for a few seconds | Evals are picked up by a poller on a short cycle rather than dispatched the instant you add the column | Give it a few seconds; it starts on its own | +| A cell is stuck showing running | The [producer](/docs/dataset/concepts/static-and-dynamic-columns) behind it, a run prompt, an eval, or another [dynamic column method](/docs/dataset/reference/dynamic-column-methods), hasn't been picked up yet or is still executing | Wait a few seconds; if the column status shows `Failed` or `Error`, rerun the column | +| A cell shows error | That row failed whatever produces the column | Rerun the column | +| Synthetic generation fails | It runs as a background job and can fail partway through | Reopen it with **Configure Synthetic Data**; the drawer returns with your saved configuration, so fix whatever caused the failure and generate again | +| A column or row you expected is gone | Most likely it was deleted; the grid can also hide columns and filter rows. Deleting a column also removes the producer behind it (a run prompt, an eval) and anything derived from it | There's no copy to restore; recreate the column or add the rows back | +| Add Row or Add Column is greyed out | Both are disabled while the dataset is processing or synthetic generation hasn't finished; Add Column alone also stays disabled until the dataset has at least one row | Wait for processing or synthetic generation to finish, or add a row first if you're adding a column to an empty dataset | +| Duplicate or Delete is greyed out | Permission gates both: duplicating a row needs update access, deleting one needs delete access | Check with an org admin about your dataset permission | + +## Keep exploring + + + + Every way to get a dataset that exists and has data in it + + + Rename, duplicate, export, and delete once the data is in + + + Where a column's values come from, and what deleting one takes with it + + + Every row, column, and file cap in one table + + diff --git a/src/pages/docs/error-feed/concepts/how-it-works.mdx b/src/pages/docs/error-feed/concepts/how-it-works.mdx deleted file mode 100644 index dc003468..00000000 --- a/src/pages/docs/error-feed/concepts/how-it-works.mdx +++ /dev/null @@ -1,94 +0,0 @@ ---- -title: "How Error Feed Works: Trace Analysis and Issue Grouping" -description: "The mental model behind Error Feed: how traces become analyzed issues, how similar errors are grouped, and how findings surface in the UI." ---- - -## About - -This page walks through what happens between a raw trace arriving and an issue showing up in the Feed. It's the mental model, not the internals. - -## The pipeline in four steps - - - - Every trace sent to a Future AGI Observe project is a candidate. Error Feed works on a sample, configurable per project. See [Sampling](/docs/error-feed/features/sampling). - - No extra instrumentation needed. If your agent is already instrumented with any of the [supported integrations](/docs/error-feed/#supported-integrations), it's already sending what Error Feed needs. - - - For every sampled trace, Error Feed: - - - Reads the full span tree: inputs, outputs, tool calls, LLM responses, errors, metadata - - Checks for failures across the [error taxonomy](/docs/error-feed/concepts/taxonomy) (five categories covering reasoning, safety, tool failures, workflow gaps, and reflection) - - Scores the trace on four quality dimensions, 0–5: Factual Grounding, Privacy & Safety, Instruction Adherence, Optimal Plan Execution - - Traces that pass without issues still get scored. A score isn't a severity, it's a quality signal. - - - When multiple traces fail in semantically similar ways (same error type, same part of the workflow), Error Feed groups them into a single **issue**. The cluster name describes what's going wrong, e.g. "Hallucinated entity in product lookup". - - The number of traces in a cluster is its **trace count**. One cluster might represent a single noisy span seen once; another might represent a systematic failure across thousands of traces. - - The point is to triage *problems*, not individual trace failures. - - - Each issue in the list shows: - - - Error name and its [taxonomy category](/docs/error-feed/concepts/taxonomy) - - [Severity](/docs/error-feed/concepts/severity-and-status) (Critical / High / Medium / Low) - - [Status](/docs/error-feed/concepts/severity-and-status) (Unresolved, Acknowledged, Resolved, Escalating) - - Trace count (cluster size) - - A sparkline showing whether the issue is getting worse, improving, or stable - - Click an issue to open the [detail view](/docs/error-feed/features/issue-overview): description, root causes, evidence, agent flow, recommendations, and every trace in the cluster. - - - -## Two levels of analysis - -There are two kinds of analysis, at different cost points: - -**Continuous scan** runs automatically on every sampled trace. It produces the description, root cause, immediate fix, long-term recommendation, evidence snippets, and quality scores on the Overview tab. Always on. - -**[Deep Analysis](/docs/error-feed/features/deep-analysis)** is on-demand. It runs a more thorough investigation on the cluster's representative trace, producing more detailed pattern analysis and recommendations. Trigger it manually from the metadata panel when the continuous scan finds something worth digging into. - -## What "representative trace" means - -When a cluster has many traces, Error Feed picks one to stand in for the cluster throughout the detail view. The Overview tab's analysis, Agent Flow diagram, and Deep Analysis all run against this representative trace. The Traces tab shows every trace in the cluster, so you can jump to any specific one. - -## Continuous vs. sampled coverage - -Error Feed doesn't analyze 100% of traces by default. The sampling rate controls what fraction get analyzed. Lower rates are faster and cheaper but miss infrequent errors. 100% gives full coverage at higher cost. - -Important: issues are formed only from traces that were actually analyzed. At 20% sampling, five identical errors in a batch of 25 traces might show up in the cluster as one occurrence — the one that got sampled. - -See [Sampling](/docs/error-feed/features/sampling) for per-project configuration. - - -Set sampling to 100% during development or testing. In production with high volume, 10–20% is a reasonable starting point. - - -## Scores vs. errors - -A trace can have a low quality score with no detected error, or a detected error with otherwise decent scores. The score measures overall quality on four dimensions; error detection flags specific failure patterns from the taxonomy. Both show up in the issue detail: scores in the metadata panel's Evaluations section, errors on the Overview tab. - -If an issue has a low Factual Grounding score but no Hallucinated Content error was flagged, that's still worth a look — the classifier missed something the score caught. - -*** - -## Next steps - - - - What error types Error Feed detects and how they're categorized. - - - How the four quality dimensions are defined and how to interpret scores. - - - The issue list page — filters, columns, and how to navigate it. - - - Configure what percentage of traces Error Feed analyzes. - - diff --git a/src/pages/docs/error-feed/concepts/scoring.mdx b/src/pages/docs/error-feed/concepts/scoring.mdx deleted file mode 100644 index 7f841c3d..00000000 --- a/src/pages/docs/error-feed/concepts/scoring.mdx +++ /dev/null @@ -1,90 +0,0 @@ ---- -title: "Error Feed Quality Scoring: Four Trace Metrics Explained" -description: "The four quality metrics Error Feed uses to score every analyzed trace, what each one measures, how scores are assigned, and how to interpret them." ---- - -## About - -Every analyzed trace gets scored on four quality dimensions, 0 to 5, where 5 is best. Scores show up in the metadata panel's Evaluations section and in the Trends tab's Score Trends chart. - -Scoring is separate from error detection. A trace can score badly on one dimension without triggering a classified error, and a trace with a detected error can still score fine on unrelated dimensions. Both are useful: scores give you a continuous quality gradient, error detection gives you discrete failure labels. - -![Evaluation scores shown in the metadata panel for an issue](/images/docs/error-feed/concepts/scoring-metadata-panel.png) - -## The four dimensions - -### Factual Grounding - -How well the agent's output is anchored in verifiable evidence: the retrieved context, provided documents, or facts the agent had access to when it responded. - -A low score means the output makes claims the input data doesn't support. This is the main signal for hallucination risk. An agent confidently answering with information it couldn't have derived from its context will score low here. - -Common causes: -- Retrieving the wrong chunks (or failing to retrieve at all) and answering anyway -- Summarizing beyond what the source actually says -- Inventing specific details like names, numbers, or dates - -### Privacy & Safety - -How well the agent follows safety and security practices: PII protection, credential hygiene, safe advice, output fairness. - -A low score means the output may expose personal data, leak credentials, give advice that could cause harm, or contain biased content. This matters most for agents that handle user data, hit external services, or operate in sensitive domains. - -Common causes: -- Including user names, emails, phone numbers, or IDs in outputs that shouldn't have them -- Echoing API keys or tokens from tool responses back into text -- Generating advice with material risk attached (medical, legal, financial) -- Producing content that stereotypes groups - -### Instruction Adherence - -How faithfully the agent follows the instructions it's been given: system prompt, user instructions, formatting constraints, tone guidelines, task-specific rules. - -A low score means the agent did something it was told not to do, skipped something it was told to do, or produced output in the wrong format. This catches prompt compliance failures that don't look like "errors" in the traditional sense. - -Common causes: -- Responding in prose when structured JSON was required -- Ignoring a "respond only in English" constraint -- Answering a question the system prompt explicitly says to deflect -- Skipping required fields in a structured output schema - -### Optimal Plan Execution - -The quality of the agent's decision-making: whether it picked the right tools, in the right order, with the right parameters, and structured its multi-step workflow logically. - -A low score means the plan was inefficient, wrong, or incomplete. The agent may have used the wrong tool, called the same one repeatedly without a clear reason, executed steps out of order, or abandoned the task before finishing it. - -Common causes of low scores: -- Selecting a less-capable tool when a more appropriate one was available -- Calling a tool with incorrect or missing parameters -- Executing steps in an illogical order -- Abandoning a multi-step workflow before reaching a conclusion - -## How scores appear in the UI - -The metadata panel on the right of any issue detail page has an Evaluations section. Each entry is an eval run against the representative trace. LLM-judge evals show a score bar with a percentage; pass/fail evals show a pass or fail verdict. - -The [Trends tab](/docs/error-feed/features/trends) shows score trends over time so you can see whether quality is improving or degrading across the cluster. - - -Scores in the metadata panel reflect whichever trace is selected in the Traces tab. Switch traces and the scores update. - - -## Scores vs. detected errors - -The scoring dimensions intentionally overlap with the [error taxonomy](/docs/error-feed/concepts/taxonomy). A Factual Grounding score of 1/5 lines up with a Hallucinated Content error. A Privacy & Safety score of 2/5 might pair with a PII Leak. - -They're not the same thing though. The score is continuous, 0 to 5. The error classification is a discrete label saying "this specific failure pattern was detected." Both can point to the same problem; using them together gives you a clearer picture. - -A cluster with consistently low scores on all four dimensions is a fundamentally broken workflow, not a narrow edge case. When only one dimension is low, the problem is more targeted. - -## Next Steps - - - - How issues are classified and how to move them through triage. - - - Score trends over time to see if quality is improving. - - diff --git a/src/pages/docs/error-feed/concepts/severity-and-status.mdx b/src/pages/docs/error-feed/concepts/severity-and-status.mdx index 6baeea0e..c10565d1 100644 --- a/src/pages/docs/error-feed/concepts/severity-and-status.mdx +++ b/src/pages/docs/error-feed/concepts/severity-and-status.mdx @@ -1,80 +1,59 @@ --- -title: "Error Feed Issue Severity and Triage Status" -description: "Error Feed severity levels classify how critical each issue is, and status labels track issues through triage from Open to Resolved." +title: "Severity & Status" +description: "How an issue's triage state and its impact level move independently" --- -## About +## Two axes, one issue -Every issue has two independent labels: **severity** and **status**. Severity is how bad the problem is. Status is where it sits in your triage workflow. Different purposes, updated independently. +Every [issue](/docs/error-feed/concepts/understanding-error-feed) in the [feed](/docs/error-feed/concepts/understanding-error-feed) carries two labels that change independently of each other. **Status** is the [triage](/docs/error-feed/guides/triage-issues) axis: where the issue sits in your team's workflow. **Severity** is the impact axis: how bad the problem is. A newly created issue starts at status Escalating and severity Medium. -## Severity +Nothing about either axis moves on its own. Every change is a person picking a new value, and any value can jump straight to any other, in either direction, at any time. -Severity reflects impact and urgency. It's assigned automatically based on the error type and quality scores, but you can override it manually on any issue. +## Status: the triage axis -| Severity | What it means | -|----------|---------------| -| **Critical** | High-confidence, high-impact failure. Likely affecting users now. Examples: safety violations, authentication failures, data exposure, complete task abandonment. | -| **High** | Significant problem with clear user impact. Examples: consistent hallucination, systematic tool misuse, repeated workflow failures. | -| **Medium** | Notable quality degradation, but not catastrophic. Examples: instruction adherence drift, suboptimal tool choices, inconsistent formatting. | -| **Low** | Minor issue or edge case. Low frequency or low impact. Examples: slight verbosity, mild instruction drift on an uncommon input. | +Status has exactly four values. -Severity shows up as a colored badge in the issue list and the issue header. Change it any time from the severity dropdown in the [metadata panel](/docs/error-feed/features/metadata-panel). +| Status | What it means | +|---|---| +| Escalating | The default for a newly created issue. Nothing has been decided about it yet | +| For review | Flagged for a closer look before someone decides what to do | +| Acknowledged | Confirmed as real | +| Resolved | The underlying problem has been fixed | - -Use severity as a prioritization signal. Start every triage session with Critical and High. Low-severity issues are worth tracking but rarely need immediate action. - +These are the same four values you'll see spelled `escalating`, `for_review`, `acknowledged`, and `resolved` in the data and filters, just written out for reading here. -## Status +## Severity: the impact axis -Status tracks where the issue is in your workflow. Severity describes the problem; status describes what your team has decided to do about it. +Severity also has exactly four values, describing how bad the problem is: critical, high, medium, and low, with medium as the default for a newly created issue. That's a different question from the quality [scores](/docs/error-feed/concepts/trace-error-analysis) a trace receives. -| Status | Meaning | -|--------|---------| -| **Unresolved** | The default for any new issue. No one has looked at it yet, or it's been reviewed but not acted on. | -| **Acknowledged** | Someone has seen this issue and confirmed it's real. Work may or may not be underway. Use this to signal "we know about it" to the rest of the team. | -| **Resolved** | The underlying problem has been fixed. Error Feed will continue monitoring, and if the same pattern resurfaces it will create a new issue. | -| **Escalating** | The issue is actively getting worse — increasing frequency, expanding impact, or a fix hasn't held. Use this to flag urgency. | + +Severity is stored as `priority` on the issue: critical maps to `urgent`, high to `high`, medium to `medium`, and low to `low`. + -A **For Review** status is also available from the dropdown for issues that need a closer look before someone decides what to do. +Severity is a label your team sets, then filters and sorts by. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the full value table and the feed's other filterable fields. -## Changing status +## How the two axes relate -Three ways: +|"triage axis"| STATUS["Status · escalating, for review, acknowledged, resolved"] + ISSUE -->|"impact axis"| SEVERITY["Severity · critical, high, medium, low"] +`} /> -1. **Header action buttons**: the issue detail header has Resolve, Acknowledge, and Ignore issue buttons for one-click updates. -2. **Status dropdown in the metadata panel**: click the current status chip in the Triage section of the right sidebar to switch to any state. -3. **Triage workflow**: see [Triage Workflow](/docs/error-feed/features/triage-workflow) for the full picture, including assigning issues. +Because the two axes are independent, an issue can sit in any combination of the two. A critical issue can still be sitting at escalating: its impact is as bad as it gets, but no one has picked it up to move it forward yet. - -Resolving an issue doesn't suppress future detection. If the same error pattern shows up again after a fix, it creates a new issue. That's deliberate: regressions should be visible. - +## Why it matters -## The "first seen" marker +Keeping status and severity separate means status can say nothing about how bad an issue is, and severity can say nothing about where it stands in triage. Collapsing them into one field would lose that distinction, and with it the ability to tell "critical but untouched" apart from "already being worked." -Recently detected issues show a **first seen** indicator in the feed. This makes it easy to spot new issues without opening every one. - -The metadata panel's Timeline section has the exact first-seen, last-seen, and age (in days) for every issue. - -## How severity and status interact - -They're independent. An issue can be Critical and Acknowledged (you know it's bad, you're working on it), or Low and Escalating (started small but keeps coming up). Set each accurately rather than using one as a proxy for the other. - -The feed list is sortable and filterable by both. Typical workflow: - -1. Filter to **Unresolved + Critical** to find the most urgent uninvestigated issues -2. Acknowledge issues you've reviewed, assign them to the right person -3. Mark Resolved once the fix is deployed and verified -4. Watch for the same cluster reappearing — if it does, the fix didn't hold - -See [Triage Workflow](/docs/error-feed/features/triage-workflow) for step-by-step guidance on working through a batch of issues. - -## Next Steps +## Keep exploring - - Step-by-step guidance for working through a backlog. + + The per-trace Scores accordion, a separate view that doesn't drive an issue's status or severity - - Filter issues by severity and status. + + Change an issue's status or severity, then filter and act on the feed diff --git a/src/pages/docs/error-feed/concepts/taxonomy.mdx b/src/pages/docs/error-feed/concepts/taxonomy.mdx deleted file mode 100644 index 1d29aefb..00000000 --- a/src/pages/docs/error-feed/concepts/taxonomy.mdx +++ /dev/null @@ -1,125 +0,0 @@ ---- -title: "Error Feed Taxonomy: Five AI Agent Error Categories" -description: "Reference for the five categories of errors Error Feed detects in AI agent traces, with every subcategory and error type defined." ---- - -## About - -Error Feed classifies every detected failure into one of five top-level categories. Each one covers a distinct class of agent failure: bad reasoning, broken tools, unsafe output, and so on. Knowing the taxonomy helps you figure out where to look when an issue lands in the feed. - -![Error taxonomy overview showing five category cards](/images/docs/error-feed/concepts/taxonomy-overview.png) - -The five categories: - -- **Thinking & Response Issues**: failures in reasoning, factual grounding, and output quality -- **Safety & Security Risks**: outputs or behaviors that could cause harm, expose data, or break security practices -- **Tool & System Failures**: errors from broken tools, APIs, or execution environments -- **Workflow & Task Gaps**: breakdowns in multi-step orchestration, memory, and retrieval -- **Reflection Gaps**: failures to reason through problems or self-correct - -*** - -## Thinking & Response Issues - -Mistakes in understanding, reasoning, factual grounding, or output formatting. - -| Subcategory | Error Type | Description | -|-------------|------------|-------------| -| **Hallucination Errors** | Hallucinated Content | Output includes information that is invented or not supported by input data. | -| | Ungrounded Summary | Summary includes claims not found in the retrieved chunks or original context. | -| **Information Processing** | Poor Chunk Match | Retrieved irrelevant or unrelated context. | -| | Wrong Chunk Used | Response based on wrong part of retrieved content. | -| | Tool Output Misinterpretation | Misread or misunderstood the output returned by a tool or API. | -| **Decision Errors** | Wrong Intent | Misunderstood the core user goal or instruction. | -| | Tool Misuse | Used a tool incorrectly or in the wrong context. | -| | Wrong Tool Chosen | Selected an inappropriate tool for the task. | -| | Invalid Tool Params | Passed malformed, missing, or incorrect parameters to a tool. | -| | Missed Detail | Skipped a key part of the user prompt or prior context. | -| **Format & Instruction** | Bad Format | Output is not valid JSON, CSV, or code. | -| | Instruction Adherence | Didn't follow instruction or style. | - -*** - -## Safety & Security Risks - -Any output or behavior that may cause harm, leak personal data, or violate security best practices. - -| Subcategory | Error Type | Description | -|-------------|------------|-------------| -| **Ethical Violations** | Unsafe Advice | Could lead to harm if followed. | -| | PII Leak | Sensitive personal info exposed in output. | -| | Biased Output | Stereotyped, unfair, or discriminatory content. | -| **Security Failures** | Token Exposure | Secrets, API keys, or auth tokens were exposed in output or logs. | -| | Insecure API Usage | Used HTTP instead of HTTPS, skipped auth headers, or lacked rate limits. | - -*** - -## Tool & System Failures - -Errors due to tool, API, environment, or runtime failures. - -| Subcategory | Error Type | Description | -|-------------|------------|-------------| -| **Setup Errors** | Tool Missing | Tool not registered or available. | -| | Tool Misconfigured | Tool or API setup is incorrect (e.g., bad schema, invalid registration). | -| | Env Incomplete | Missing tokens, secrets, or setup environment variables. | -| **Tool/API Failures** | Rate Limit | Too many requests hit the limit. | -| | Auth Fail | Authentication to tool or service failed. | -| | Server Crash | Tool/API returned internal error. | -| | Resource Not Found | Requested endpoint or resource does not exist or is not reachable. | -| **Runtime Limits** | Out of Memory | RAM or resource limit breached. | -| | Timeout | Execution took too long and was halted. | - -*** - -## Workflow & Task Gaps - -Breakdowns in multi-step task execution, orchestration, or memory. - -| Subcategory | Error Type | Description | -|-------------|------------|-------------| -| **Context Loss** | Dropped Context | Missed relevant past messages or data. | -| | Overuse | Unnecessary context/tools invoked. | -| **Retrieval Errors** | Poor Chunk Match | Retrieved irrelevant or unrelated context. | -| | Wrong Chunk Used | Response based on wrong part of retrieved content. | -| | No Retrieval | Failed to run retrieval when needed. | -| **Task Flow Issues** | Goal Drift | Strayed from intended objective. | -| | Step Disorder | Steps executed out of logical order. | -| | Redundant Steps | Repeated same tool or action unnecessarily. | -| | Task Orchestration Failure | Agent failed to plan or interleave actions properly across tools or steps. | -| **Trace Completion** | Incomplete Task | No final result or closure. | - -*** - -## Reflection Gaps - -Agent failed to engage in introspective reasoning or revise steps appropriately. - -| Error Type | Description | -|------------|-------------| -| Missing CoT | No intermediate thinking steps (Chain of Thought) were used to justify actions. | -| Missing ReAct Planning | Agent failed to interleave reasoning with action; took action without planning. | -| Lack of Self-Correction | Agent didn't revise response or plan after detecting error or contradiction. | - -*** - -## How taxonomy categories appear in the UI - -On the [Overview tab](/docs/error-feed/features/issue-overview), each detected error shows its taxonomy type as a chip. The Description section says what went wrong in this trace; Root Cause says why; Evidence quotes the relevant spans directly. - -On the [Feed list](/docs/error-feed/features/the-feed), issues are tagged with their primary error type so you can filter by category when you're hunting a specific class of failure. - - -A single trace can trigger errors in multiple categories. A tool failure that causes the agent to hallucinate a fallback answer will register as both a Tool & System Failure and a Thinking & Response Issue. - - -## Next Steps - - - - Where taxonomy categories appear in the issue detail UI. - - - How quality scores complement error detection. - - diff --git a/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx b/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx new file mode 100644 index 00000000..c621af39 --- /dev/null +++ b/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx @@ -0,0 +1,50 @@ +--- +title: "Trace error analysis" +description: "The four per-trace quality scores and why they stay out of the feed" +--- + +## What Error Analysis is + +Open a [trace](/docs/observe/concepts/traces) that has one and you'll find **Error Analysis** in the **Scores** accordion: four quality dimensions, each scored for that one trace. It's a separate, per-trace view, not a property of an [issue](/docs/error-feed/concepts/understanding-error-feed) or a cluster. + +## The four dimensions + +- **Factual Grounding**: whether the response holds up against the evidence and context the agent actually had +- **Privacy And Safety**: whether the response handles sensitive data and follows safe practices +- **Instruction Adherence**: whether the response follows the instructions the agent was given +- **Optimal Plan Execution**: whether the agent's sequence of decisions and tool calls was the right one for the task + + +The UI title-cases the raw dimension name, so what you'd write as "Privacy & Safety" renders as Privacy And Safety in the product. + + +## Where you see it + +Open the [trace detail drawer](/docs/observe/guides/explore-dashboard#open-a-trace) and its **Scores** accordion shows one chip per dimension: the label and the score out of 5, for example "Factual Grounding 4/5". + +## When to check the scores + +Read the scores when you're already looking at a specific trace, for example while working through an issue in the [Investigate an issue](/docs/error-feed/guides/investigate-an-issue) guide, and want a read on that trace beyond the issue's category. Treat a dimension scoring lower than the others as a pointer to look closer at the plan, the tool calls, or the response, not a verdict on its own. + +## A different pipeline from the Error Feed scanner + +|Error Feed scanner| FD["Finding → issue in the feed"] + TR -->|Error Analysis| SC["Four dimension scores → Scores accordion"]`} /> + +Error Analysis and the [Error Feed](/docs/error-feed) scanner are two independent reads on the same trace. The feed reads traces, groups the problems it finds into issues, and tells you where to fix them; the Scores accordion tells you how one specific trace performed on these four dimensions. + +These four scores don't feed the [error taxonomy](/docs/error-feed/reference/error-taxonomy): a low score doesn't create an issue, and it isn't a value you can filter the feed by. Checking both means opening the trace and reading the accordion yourself. + +## Keep exploring + + + + Open an issue and work through the evidence, trace by trace + + + The fixed set of groups, categories, and fix layers a scan can assign + + diff --git a/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx b/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx new file mode 100644 index 00000000..3119e1ad --- /dev/null +++ b/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx @@ -0,0 +1,53 @@ +--- +title: "Understanding Error Feed" +description: "How a scan or eval failure becomes one issue, and what that issue actually owns" +--- + +## What an issue is + +[Error Feed](/docs/error-feed) reads your [traces](/docs/observe/concepts/traces) and turns the problems it finds in them into issues you triage. + +Nothing marks a trace as failing beforehand. Error Feed samples traces at whatever rate the project is set to, reads each sampled trace in full, and decides for itself whether something went wrong in it. A trace is "failing" only in the sense that a scan found at least one problem in it, which is why raising the sampling rate surfaces more issues: it isn't finding more failures, it's reading more traces. + +An **issue** isn't one such trace. It's the one problem behind many of them. When ten traces go wrong the same way, the feed doesn't hand you ten rows to read one by one, it hands you the single issue they all point at, and that issue is what you work. + +## What an issue carries + +Every issue in the feed carries the same things, and each one is there to help you decide what to do about it: + +- **A title**, naming the problem in a line +- **A category and a group**, the two labels that place the problem in the [error taxonomy](/docs/error-feed/reference/error-taxonomy) +- **A fix layer**, the part of your system the fix belongs in +- **A severity and a status**, the two independent axes covered in [Severity & Status](/docs/error-feed/concepts/severity-and-status) +- **An assignee**, once someone picks it up +- **How often and how widely it happened**: the number of times it fired, the number of [traces](/docs/observe/concepts/traces) affected, and the number of [users](/docs/observe/concepts/users) behind those traces +- **When it started and when it last happened** +- **The evidence behind it**: the traces, [spans](/docs/observe/concepts/spans), and [sessions](/docs/observe/concepts/sessions) that contributed, so you can open the exact span rather than hunting through a trace + +Issues also come from a failing [eval](/docs/evaluation), not only from a scan. Those group by the eval that failed, and the feed shows the eval's name where a scan issue shows its group. + +For example, ten traces that all call the wrong tool for a refund lookup surface as one cluster: title "Wrong tool selected for refund lookup", group Tool Failures, fix layer Tools, 34 total events across 12 unique traces and 9 users affected, first seen 09:14 and last seen 14:02. + +## Fix layers: where the fix belongs + +A fix layer is one of Prompt, Tools, Orchestration, or Guardrails. It's the product's actual answer to "so what do I do about this": it names where in your system the fix belongs, not just what went wrong. Scanner clusters always carry one, taken straight from the finding. Eval clusters carry one where it can be determined: a best-effort step tries to infer it, and it's left unset when that step can't. The error taxonomy is the reference for which specific error types map to which fix layer. + +Fix layer rides on the cluster itself, so it's also a live filter in the feed, letting you work through everything that needs a prompt change before you touch anything that needs an orchestration change. + +## Why it matters + +Working one issue instead of a thousand traces is what makes the feed usable at scale. A single misbehaving tool can fail on every call for an hour and put a problem in thousands of traces; read them one at a time and you're reading the same failure a thousand times, work the issue and you fix it once. And because the fix layer sits on the issue itself, the feed tells you what to change before you've opened a single trace inside it. + +## Keep exploring + + + + The two independent axes every issue carries, and how they change + + + A separate per-trace scoring pipeline, and why it doesn't drive the feed + + + Open an issue and work through the evidence behind it + + diff --git a/src/pages/docs/error-feed/features/deep-analysis.mdx b/src/pages/docs/error-feed/features/deep-analysis.mdx deleted file mode 100644 index c36add3f..00000000 --- a/src/pages/docs/error-feed/features/deep-analysis.mdx +++ /dev/null @@ -1,96 +0,0 @@ ---- -title: "Error Feed Deep Analysis: On-Demand Trace Investigation" -description: "On-demand investigation that runs deeper root cause analysis on an Error Feed issue's trace and produces richer findings than the continuous scan." ---- - -## About - -Every issue in Error Feed comes with analysis generated automatically by the continuous scan: description, root cause, evidence, recommendations. For most issues that's enough to understand and act on the problem. - -Deep Analysis is an additional, on-demand investigation you trigger manually. It runs a more thorough analysis of the issue's representative trace and produces richer findings, especially around root cause precision and the Recommendations & Fixes section. - -## When to use - -Use Deep Analysis when: - -- The continuous scan's findings feel incomplete or you want more specificity about root cause -- The issue is high-severity and you want confidence in the diagnosis before investing in a fix -- The Overview tab's Probable Root Cause or Recommendations sections are sparse -- You're doing a post-incident review and need detailed evidence for a write-up - -You don't need to run it on every issue. Save it for the ones where the standard analysis leaves open questions. - -## How to run it - -Deep Analysis is triggered from the **Deep Analysis section** in the [metadata panel](/docs/error-feed/features/metadata-panel) on the right side of the issue detail page. - - - - Navigate to any issue from the [Feed list](/docs/error-feed/features/the-feed). - - - In the right sidebar, scroll to the **Deep Analysis** section. If no analysis has been run yet, you'll see a **Run Deep Analysis** button. - - - Click the button. A toast confirms the analysis has started, and you'll see a progress indicator: "Running analysis…" - - Takes about a minute. You can navigate away; analysis continues in the background. - - - When it finishes, the metadata panel shows "Analysis complete." Go back to the **Overview tab** to see the updated findings. Probable Root Cause and Recommendations & Fixes will be populated with more detailed content. - - - -![Deep Analysis button in the metadata panel, and the Running state with progress indicator](/images/docs/error-feed/features/deep-analysis-states.png) - -## What it produces - -Results appear in the [Overview tab](/docs/error-feed/features/issue-overview) under **Probable Root Cause** and **Recommendations & Fixes**, with more detail than the continuous scan produces. - -Specifically, Deep Analysis tends to produce: - -- More specific identification of the failing component or step -- More targeted recommendations that reference the actual tool, prompt structure, or workflow pattern involved -- Richer pattern analysis when the cluster has multiple traces with consistent failure characteristics - -## Re-running analysis - -Once Deep Analysis completes, the metadata panel shows a **Re-run** button alongside "Analysis complete." Use Re-run when: - -- You've made changes to your agent and want fresh analysis to confirm whether the root cause has shifted -- The cluster has grown significantly since the last run and you want updated findings - -Re-running discards the previous result and generates new findings from scratch against the current representative trace. - -## Analysis states - -| State | What you see | What to do | -|-------|--------------|------------| -| Idle | "Run Deep Analysis" button | Click to start | -| Running | Progress spinner + "Running analysis…" | Wait or navigate away | -| Complete | "Analysis complete" + Re-run option | Check Overview tab for results | -| Failed | "Retry Deep Analysis" button | Click to retry | - - -If no trace is selected or the cluster has zero analyzed traces, the Deep Analysis button is disabled. It re-enables as soon as at least one trace enters the cluster. - - -## Relationship to continuous scan - -Continuous scan runs automatically on every sampled trace. Deep Analysis runs once, on demand, against the representative trace. They're complementary: - -- Continuous scan gives you breadth: every trace gets analyzed, issues surface automatically -- Deep Analysis gives you depth: one trace gets a thorough investigation when you need it - -Neither replaces the other. Typical pattern: continuous scan surfaces the issue, Deep Analysis helps you understand it well enough to fix it confidently. - -## Next Steps - - - - How Deep Analysis output appears in the Overview tab. - - - When to escalate from continuous scan to deep analysis. - - diff --git a/src/pages/docs/error-feed/features/issue-overview.mdx b/src/pages/docs/error-feed/features/issue-overview.mdx deleted file mode 100644 index f47ee9c0..00000000 --- a/src/pages/docs/error-feed/features/issue-overview.mdx +++ /dev/null @@ -1,107 +0,0 @@ ---- -title: "Error Feed Issue Overview: Header to Recommendations" -description: "A walkthrough of the Overview tab on an issue detail page: every section explained, from the header to the AI-generated recommendations." ---- - -## About - -The issue detail page is where you understand a problem well enough to fix it. It opens when you click any issue in the [Feed list](/docs/error-feed/features/the-feed). Three zones: a header, a tab area, and a metadata panel on the right. - -This page covers the **Overview tab**, which is the default view. For the others see [Traces](/docs/error-feed/features/traces), [State Graph](/docs/error-feed/features/state-graph), and [Trends](/docs/error-feed/features/trends). For the right sidebar see [Metadata Panel](/docs/error-feed/features/metadata-panel). - -![Issue detail page with header, Overview tab active, and metadata panel visible](/images/docs/error-feed/features/issue-detail-full.png) - -## The header - -The header stays visible regardless of which tab you're on. - -![Issue detail header showing error title, type chip, severity badge, status, trace count, and action buttons](/images/docs/error-feed/features/issue-detail-header.png) - -**Breadcrumb**: "Error Feed" links back to the list. The chip next to it shows the error type (e.g. "Hallucination", "Tool Failure"). - -**Error title**: the cluster name, describing what's going wrong. - -**Status chips row** (left to right): error type dot, [status chip](/docs/error-feed/concepts/severity-and-status), [severity badge](/docs/error-feed/concepts/severity-and-status), trace count chip. - -**Action buttons** (top right): -- **Copy cluster ID**: copies the cluster identifier, handy for tickets or Slack -- **Share**: copies a direct link to this issue -- **Resolve**: marks the issue resolved in one click -- **Acknowledge**: marks the issue acknowledged -- **Ignore issue**: suppresses the issue from the default view - - -The Resolve and Acknowledge buttons in the header are shortcuts. The full triage workflow (assigning to a team member, changing severity) lives in the metadata panel. See [Triage Workflow](/docs/error-feed/features/triage-workflow). - - -## Overview tab layout - -Two columns. The left lists every trace in the cluster (the Traces affected panel) along with a small events-and-users chart. The right shows analysis for the selected trace. - -Click any trace on the left to focus the right column on it. - -## Always-visible sections - -These sections come from the continuous scan that runs on every sampled trace. They show up the moment you open the issue, no extra action required. - -### Selected trace header - -A compact bar at the top of the right column with the selected trace's ID, latency, cost, and total tokens. Treat it as the breadcrumb for which trace's analysis you're currently looking at. - -### Pattern Summary - -A summary of patterns across the whole cluster, not just the selected trace. The card has a one-line takeaway plus four headline metrics (e.g. "68% of errors involve retrieval step", "3.4× vs. baseline error rate", "12s median time-to-fail", "GPT-4o top-affected model"). - -Most useful when the cluster has a large trace count, where individual trace analysis won't surface systemic patterns that are obvious at scale. - -### Agent Flow - -A visual diagram of the steps the agent took (LLM calls, tool invocations, sub-agent interactions) and where the failure happened. - -The diagram makes it obvious whether the error is in step one or step five, whether it's in the LLM response or a downstream tool call, and whether there's a clear bifurcation point between traces that succeed and traces that fail. - -![Agent Flow diagram showing a multi-step workflow with a failure highlighted at the tool call step](/images/docs/error-feed/features/overview-tab-agent-flow.png) - -### Trace Evidence - -A side-by-side view of a failing trace and a working trace with the differences highlighted inline. Two tabs: **Failing Trace** (default) and **Working Trace**. - -Each reel walks through user input, retrieved context, model output, and eval verdict, quoting the actual content with deltas color-coded so you can spot where the failing run diverged. This is the section that shows you concretely *what* went wrong. - -## After running Deep Analysis - -The continuous scan is fast and cheap, but it stops at "what happened." For *why* it happened (and what to do about it), trigger **Deep Analysis** from the metadata panel on the right. Takes about a minute. - -When it finishes, two new sections appear at the bottom of the Overview tab: - -### Probable Root Cause - -A ranked list of causes (usually two to four) explaining *why* the cluster is failing. Each cause has a short title and a longer explanation. Ordered by how strongly the analysis supports them, so the top one is the best candidate to investigate first. - -### Recommendations & Fixes - -A ranked list of suggested fixes, each with a priority chip (High / Medium / Low). Click any recommendation to expand it: - -- **Description**: what the recommendation is, in plain language -- **Immediate Fix**: the minimal change to apply right now (often a one-liner you can paste into a prompt or config) -- **Insights**: why this fix works and how it relates to the root cause -- **Evidence**: the trace data or pattern that supports the recommendation - -Recommendations link back to the Probable Root Causes that motivated them, so you can see which fix addresses which cause. - -![Probable Root Cause and Recommendations & Fixes appear after Deep Analysis runs](/images/docs/error-feed/features/overview-tab-recommendations.png) - - -If Probable Root Cause and Recommendations & Fixes aren't on the Overview tab, Deep Analysis hasn't run for this issue yet. Click **Run Deep Analysis** in the metadata panel to trigger it. See [Deep Analysis](/docs/error-feed/features/deep-analysis). - - -## Next Steps - - - - View every trace in the cluster. - - - Trigger a deeper investigation on this issue. - - diff --git a/src/pages/docs/error-feed/features/linear-integration.mdx b/src/pages/docs/error-feed/features/linear-integration.mdx deleted file mode 100644 index e7c5c32c..00000000 --- a/src/pages/docs/error-feed/features/linear-integration.mdx +++ /dev/null @@ -1,75 +0,0 @@ ---- -title: "Creating Linear Tickets from Error Feed Issues" -description: "Create Linear tickets directly from Error Feed issues to link your AI error monitoring to your engineering workflow and track fixes." ---- - -## About - -Error Feed integrates with Linear so you can turn a detected issue into a tracked engineering task without leaving the platform. The integration is action-only: it creates tickets from Error Feed issues. Linear issues don't sync back to Error Feed. - -## Prerequisites - -You need a Linear account and a Future AGI account with the Linear integration connected. If you haven't connected Linear yet: - - - - Navigate to the Future AGI dashboard settings and open the Integrations page. - - - Find the Linear integration and follow the OAuth flow to authorize the connection. You'll need Linear admin or member access. - - - -Once connected, the Linear row in the metadata panel of every Error Feed issue will show "Connected." - -## Creating a ticket - - - - Navigate to any issue in the [Error Feed list](/docs/error-feed/features/the-feed) and open it. - - - Scroll to the bottom of the [metadata panel](/docs/error-feed/features/metadata-panel) on the right side. The Integrations section shows the Linear row with a **Create issue** button. - - - A dialog opens with your Linear teams. Pick the team you want the ticket in. - - - The ticket is created in Linear and immediately linked to the Error Feed issue. The metadata panel updates to show the Linear issue ID (e.g. "ENG-1234") with a **View ENG-1234** button that opens the ticket. - - - -## What's in the ticket - -The Linear ticket is pre-populated with context from the Error Feed issue: - -- Cluster name as the ticket title -- Error description and root cause from the Overview tab as the ticket body -- A link back to the Error Feed issue detail page - -Your engineering team has everything they need to understand the problem without cross-referencing Future AGI separately. - -## Viewing a linked ticket - -Once a ticket exists, the Linear row in the metadata panel shows the issue ID and a "View [issue ID]" link. Click to open the ticket in Linear. - -If the issue has already been linked, clicking "Create issue" again opens the existing ticket rather than creating a duplicate. - -## Disconnected state - -If Linear isn't connected, the row shows "Not connected" with a **Connect** button. Click to go to Settings → Integrations and set up the connection. - - -The Linear integration pairs well with the [triage workflow](/docs/error-feed/features/triage-workflow). A typical pattern: review the issue in Error Feed, acknowledge it, create a Linear ticket, assign the ticket to the engineer who owns that component. - - -## Next Steps - - - - How Linear tickets fit into the triage process. - - - The right sidebar where the Linear integration lives. - - diff --git a/src/pages/docs/error-feed/features/metadata-panel.mdx b/src/pages/docs/error-feed/features/metadata-panel.mdx deleted file mode 100644 index 599a847a..00000000 --- a/src/pages/docs/error-feed/features/metadata-panel.mdx +++ /dev/null @@ -1,127 +0,0 @@ ---- -title: "Error Feed Metadata Panel: Triage, Stats, and Integrations" -description: "The right-side metadata panel on an Error Feed issue page covers triage controls, cluster stats, timeline, evaluations, and co-occurring issues." ---- - -## About - -The metadata panel runs along the right side of every issue detail page and stays visible regardless of tab. It's where the operational info about an issue lives: status, assignee, cluster stats, timeline, AI metadata, evaluations, linked issues, and integrations. - -![Metadata panel showing Triage, Cluster, Deep Analysis, Timeline, and AI Metadata sections](/images/docs/error-feed/features/metadata-panel-overview.png) - -## Triage - -The Triage section handles status, severity, and assignee. - -**Status**: click the status chip for a dropdown with all states: Escalating, Acknowledged, For Review, Resolved. The chip is color-coded by status. See [Severity and Status](/docs/error-feed/concepts/severity-and-status). - -**Severity**: click the severity chip to change it. Options: Critical, High, Medium, Low. Override the auto-assigned severity whenever you have better context about actual user impact. - -**Assignee**: click Assign to assign the issue to a team member. The dropdown lists everyone in your org. Click the assigned name to reassign or unassign. - -Changes take effect immediately and show up in the Feed list view. - -## Cluster - -At-a-glance stats about the issue's scope: - -| Field | What it shows | -|-------|---------------| -| **Traces** | Total number of traces grouped in this cluster | -| **Users affected** | Distinct users whose traces appear in the cluster | -| **Sessions** | Number of distinct sessions represented | -| **Cluster ID** | The unique identifier for this cluster, used for API access and references | - -These give you a fast read on scope without counting rows in the Traces tab. - -## Deep Analysis - -A single button that triggers on-demand investigation of the issue's representative trace. - -- **Idle**: a "Run Deep Analysis" button. Click to dispatch. -- **Running**: a progress indicator with "Running analysis…". Takes about a minute. You can navigate away; analysis continues in the background. -- **Complete**: "Analysis complete" with a Re-run option. Results populate the Overview tab's Probable Root Cause and Recommendations & Fixes. -- **Failed**: "Retry Deep Analysis" if the last run failed. - -See [Deep Analysis](/docs/error-feed/features/deep-analysis) for when to use it and what to expect. - -## Timeline - -When the issue first appeared and when it was most recently seen. - -| Field | What it shows | -|-------|---------------| -| **First seen** | When Error Feed first detected this error pattern, relative to now | -| **Last seen** | The most recent occurrence in the cluster | -| **Age** | Days since first detection | - -A long age (e.g. 45 days) with a recent "last seen" means the issue has been around for a while without being resolved. Useful context for prioritization. - -## AI Metadata - -Trace-level context about the trace currently being viewed (the representative trace, unless you've picked a different one in the Traces tab). - -| Field | What it shows | -|-------|---------------| -| **Model** | The LLM model used in the trace | -| **Version** | The model version | -| **Agent** | The agent name, if set in trace attributes | -| **Pipeline** | The pipeline name, if set | -| **Connector** | The integration connector used | -| **Project** | The Observe project this trace belongs to | -| **Eval score** | The composite evaluation score for this trace | -| **Trace ID** | The trace's unique identifier | - -Fields not set on the trace are omitted. AI Metadata is useful for correlating issues to specific model versions, e.g. confirming a degradation started when you switched model versions. - - -AI Metadata reflects the selected trace. Click a different trace in the Traces tab to see its metadata here. - - -## Evaluations - -Quality scores for the currently selected trace. Each evaluation is a named row with either: - -- A **score bar and percentage** for LLM-judge evaluations (e.g. Factual Grounding: 62%) -- A **pass/fail verdict** with a green check or red X for pass/fail evaluations - -These correspond to the four dimensions in [Scoring](/docs/error-feed/concepts/scoring), plus any custom evaluations set up on the project. - -If every trace in a cluster shows Factual Grounding in the 20–40% range, something is systematically wrong with grounding even if the classifier didn't flag a specific hallucination. - -## Co-occurring Issues - -When multiple distinct clusters tend to appear together in the same traces, they show up here. Each entry has: - -- The co-occurring issue title -- How many traces are shared between the two issues -- Co-occurrence percentage (what fraction of this issue's traces also appear in that cluster) - -High co-occurrence (70%+) usually means a shared root cause: fixing one often fixes the other, or at minimum they're worth investigating together. - -Click any co-occurring issue to jump to its detail page. - -## Activity - -A timeline of significant events for this issue, starting with first detection. Status changes, assignment events, and comments will also land here as the issue moves through triage. - -## Integrations - -Connected issue-tracking tools. Currently **Linear** is supported. - -- Linear not connected: shows a "Connect" button that takes you to Settings → Integrations. -- Connected with no ticket: shows a "Create issue" button. -- Ticket already exists: shows the ticket ID with an option to open it. - -See [Linear Integration](/docs/error-feed/features/linear-integration) for setup and workflow details. - -## Next Steps - - - - The Overview tab the metadata panel sits next to. - - - How to use the panel to move issues through triage. - - diff --git a/src/pages/docs/error-feed/features/sampling.mdx b/src/pages/docs/error-feed/features/sampling.mdx deleted file mode 100644 index 536e8abe..00000000 --- a/src/pages/docs/error-feed/features/sampling.mdx +++ /dev/null @@ -1,76 +0,0 @@ ---- -title: "Error Feed Sampling: Controlling Trace Analysis Rate" -description: "How sampling rate controls what percentage of traces Error Feed analyzes, and how to configure it per project in Observe settings." ---- - -## About - -Error Feed doesn't analyze every trace by default. The **sampling rate** controls what percentage of incoming traces get analyzed. The tradeoff is coverage vs. cost: analyze more traces and you catch more errors, but you pay more for it. - -## Why sampling exists - -Production agents can produce a lot of traces. Analyzing 100% of them at all times gets expensive at scale. Sampling lets you dial in a rate that makes sense for your situation: full coverage during development, a reduced rate in production, or 100% for critical projects where nothing can be missed. - -The rate applies to new traces. Previously analyzed traces aren't affected when you change it. - -## How to configure sampling - -Sampling is configured per project in Observe settings. - - - - Navigate to your project in the Observe section of the Future AGI dashboard. - - - Click the **Configure** (gear) icon in the project header to open the settings drawer. - - - Find the **Error Feed sampling rate** control in the drawer. Drag the slider right to increase coverage, left to decrease. - - - Click **Update** to apply. The new rate kicks in for traces that arrive after the update. - - - - -The new rate only applies to new traces. Previously analyzed traces aren't re-analyzed or de-analyzed when you change the rate. - - -## Choosing a sampling rate - -There's no universally correct rate. It depends on your trace volume, cost tolerance for the project, and how critical full error coverage is. - -| Situation | Recommended rate | -|-----------|-----------------| -| Development or testing | 100% — catch everything while you're actively iterating | -| Low-volume production | 100% or close to it — the absolute cost is low | -| High-volume production | 10–20% — enough to catch systematic issues, affordable at scale | -| Critical path / safety-sensitive | 100% — can't afford to miss errors | -| Cost-constrained, high volume | 5–10% — catches recurring patterns even at low rates | - -At 10% sampling, a systematic error that hits every trace shows up as a cluster with 10% of its true occurrence count. The error still gets detected and surfaced. Sampling reduces counts and may miss rare one-off failures, but it reliably catches recurring patterns. - - -Start at 100% when you first set up a project. Once you understand the error landscape and have addressed the biggest issues, drop the rate to something that makes sense for your production volume. - - -## Effect on cluster trace counts - -The trace count on an issue reflects how many analyzed traces ended up in the cluster, not the total number of traces where that error might have occurred. At 20% sampling, a cluster with 50 traces likely represents around 250 actual occurrences. - -Keep the sampling rate in mind when comparing cluster sizes across projects or time periods. A 100-trace cluster from a 10%-sampled project represents more actual errors than a 100-trace cluster from a 100%-sampled project. - -## Effect on new issue detection - -Rare errors (the ones that show up in only a small fraction of traces) are more likely to be missed at low rates. If you're hunting an edge case that only triggers occasionally, temporarily bump the sampling rate up for the duration of the investigation. - -## Next Steps - - - - How sampled traces become clusters and issues. - - - View issues created from sampled traces. - - diff --git a/src/pages/docs/error-feed/features/state-graph.mdx b/src/pages/docs/error-feed/features/state-graph.mdx deleted file mode 100644 index 42827f43..00000000 --- a/src/pages/docs/error-feed/features/state-graph.mdx +++ /dev/null @@ -1,69 +0,0 @@ ---- -title: "Error Feed State Graph: Agent Decision Flow Diagram" -description: "How to read the State Graph tab, the agent decision flow diagram showing where traces diverge between success and failure paths." ---- - -## About - -The **State Graph** tab visualizes how an agent moves through its workflow and where it fails. It surfaces structural patterns that would take a lot of reading to extract from raw trace data. - -![State Graph tab showing Agent Decision Flow diagram](/images/docs/error-feed/features/state-graph-overview.png) - -## Agent Decision Flow - -A branching diagram mapping the paths an agent takes from invocation to completion: - -- **Shared steps**: the common entry path every trace follows (invocation, initial LLM call, etc.) -- **Fork point**: where traces diverge into success and failure paths -- **Failure branch**: the steps failing traces take, colored red -- **Success branch**: the steps passing traces take, colored green -- **Edge labels**: percentages on the fork edges showing what fraction of traces take each path - -Read left to right. Steps are nodes; transitions between steps are edges. A step that consistently appears only on the failure path is a strong root-cause candidate. - -![Close-up of the Agent Decision Flow fork showing 92% failure path vs 8% success path](/images/docs/error-feed/features/state-graph-flow-detail.png) - -### Reading the fork percentages - -The numbers on the fork edges are the most useful signal. If 92% of traces go down the failure path after a particular step, that step is where the problem concentrates. No need to reason about edge cases; the data is telling you where to look. - -A near-even split (50/50) means the problem depends on input characteristics, not a systematic code issue. A lopsided split (95/5) means a near-universal failure, probably a configuration or logic error. - -### Node types - -Different step types appear as visually distinct nodes: - -| Node type | What it represents | -|-----------|-------------------| -| Invocation | The agent entry point | -| Agent run | An agent execution step | -| LLM call | A language model inference step | -| Tool execution | A tool or function call | -| Evaluation | An inline quality check step | -| Error | A step that resulted in an error state | -| Success | A step that completed successfully | - -## When the State Graph is most useful - -The State Graph is most useful when: - -- The cluster has a moderate-to-large trace count (enough to make the percentages meaningful) -- The agent has multiple steps (single-step agents don't produce interesting flow diagrams) -- You suspect the failure is structural, tied to a specific workflow path, rather than random - -For small clusters or very simple agents, the [Overview tab](/docs/error-feed/features/issue-overview) is usually enough. The State Graph earns its keep when you're looking at a large cluster and want to understand the shape of failure before diving into individual traces. - -## Relationship to the Overview tab's Agent Flow - -The [Overview tab](/docs/error-feed/features/issue-overview) also has an Agent Flow section, but it's based on the representative trace and shows the narrative flow for that single trace. The State Graph is based on the whole cluster and shows the aggregate picture. Use Agent Flow to understand the specific failure; use State Graph to understand how widespread and how structural it is. - -## Next Steps - - - - The Overview tab where Agent Flow first appears. - - - Jump from a State Graph node to the underlying trace. - - diff --git a/src/pages/docs/error-feed/features/the-feed.mdx b/src/pages/docs/error-feed/features/the-feed.mdx deleted file mode 100644 index d7e7ec04..00000000 --- a/src/pages/docs/error-feed/features/the-feed.mdx +++ /dev/null @@ -1,99 +0,0 @@ ---- -title: "The Error Feed Issue List: Filters, Stats, and Columns" -description: "How to read the Error Feed issue list: filters, the stats bar, table columns, trend sparklines, and time range controls." ---- - -## About - -The Feed is the landing page for Error Feed, in the left sidebar under **Error Feed**. It shows every detected issue across your projects as a filterable, sortable list. This is where you start every triage session. - -![Error Feed list page with filter bar, stats bar, and issue table](/images/docs/error-feed/features/feed-list-overview.png) - -## Filter bar - -The filter bar at the top narrows the list to what matters right now. - -| Filter | Options | -|--------|---------| -| **Project** | Select one or more Observe projects. Defaults to all projects. | -| **Time range** | Last 24 hours / 7 days / 14 days / 30 days / 90 days | -| **Status** | Unresolved, Acknowledged, Resolved, Escalating | -| **Severity** | Critical, High, Medium, Low | - -Filters combine. Selecting **Critical + Unresolved** shows only critical issues that haven't been acted on. - - -Start every session with **Unresolved + Critical** to see the highest-priority uninvestigated issues first. - - -## Stats bar - -The stats bar below the filter gives a snapshot of the current view: - -- **Total issues**: distinct clusters matching the current filters -- **Total occurrences**: individual trace errors across those clusters -- **New issues**: clusters that first appeared within the selected time range - -These update as you change filters. Use the occurrence count to gauge scope: a cluster with 5 traces is very different from one with 500. - -## Issue table - -Each row in the table represents one issue (cluster). The columns are: - -### Error name - -A human-readable name for the cluster, generated from the error pattern. This is what you read to figure out what's going wrong — e.g. "Tool parameter validation failed on search_docs" or "Hallucinated product availability". - -### Severity - -Critical (red), High (orange), Medium (yellow), Low (gray). See [Severity and Status](/docs/error-feed/concepts/severity-and-status) for what each level means and how to change it. - -### Status - -Current triage status: Unresolved, Acknowledged, Resolved, or Escalating. Issues you haven't looked at yet stay Unresolved. - -### Traces - -Number of individual traces grouped into this cluster. A high trace count means the error is recurring frequently. - -### Trend - -A sparkline showing how often this error appeared over the selected time range. Upward trend, getting worse. Flat, stable. Downward, resolving on its own (or you fixed something upstream). - -A directional indicator next to the sparkline shows the same thing at a glance. - -## Clicking an issue - -Click any row to open the issue detail page. It has four tabs (Overview, Traces, State Graph, Trends) and a metadata panel on the right. - -See the feature pages for each: - - - - Description, root cause, evidence, and recommendations. - - - All traces in the cluster. - - - Visual breakdown of where and how the agent failed. - - - Error frequency, score trends, and activity heatmap. - - - -## Navigation tip - -The list remembers your last filter state within a session. Open an issue, hit the breadcrumb to go back, and your filters are still applied. - -## Next Steps - - - - Description, root cause, evidence, and recommendations. - - - Move issues from new to resolved efficiently. - - diff --git a/src/pages/docs/error-feed/features/traces.mdx b/src/pages/docs/error-feed/features/traces.mdx deleted file mode 100644 index 27d37103..00000000 --- a/src/pages/docs/error-feed/features/traces.mdx +++ /dev/null @@ -1,59 +0,0 @@ ---- -title: "Error Feed Traces Tab: Navigating Failure Clusters" -description: "How to use the Traces tab on an issue detail page to navigate every trace in a cluster and understand the distribution of failures." ---- - -## About - -The **Traces** tab lists every individual trace grouped into the issue's cluster. The tab label shows the count, e.g. "Traces 47" means 47 traces have been grouped under this issue. - -![Traces tab showing a list of traces with status indicators and metadata](/images/docs/error-feed/features/traces-tab-overview.png) - -## What you see - -Each row is one trace from the cluster: - -- **Trace ID**: unique identifier you can use to look it up in Observe -- **Status**: whether this trace was a failure or (where applicable) a comparison success trace -- **Timestamp**: when the trace happened -- **Latency, tokens, and cost metadata**: where available on the original trace - -The tab header shows the total count so you can size up the cluster without scrolling. - -## Navigating between traces - -Clicking a trace selects it as the active trace for the detail view. Two effects: - -1. The **metadata panel** on the right updates to show AI Metadata and Evaluations for the selected trace instead of the representative trace. -2. The **Deep Analysis** button in the metadata panel will run against the selected trace if triggered. - -Useful for investigating specific traces within a cluster: comparing a trace from three days ago against a recent one, or pulling up a trace from a particular user or session. - - -The Overview tab always shows analysis for the cluster's representative trace, not whichever one you've selected in the Traces tab. Check the metadata panel's AI Metadata section to confirm which trace is being analyzed. - - -## Using the Traces tab to understand scope - -The trace count on the tab label is the most direct signal of how widespread the problem is. One trace might be a one-off; 500 traces is hitting a large fraction of your traffic (relative to sampling rate). - -For high-count clusters, check whether the traces are clustered in time (a systemic issue that appeared on a specific day) or spread evenly over weeks (a recurring edge case, not a single event). - -The [Trends tab](/docs/error-feed/features/trends) gives you the time-series view, which is the better tool for that kind of temporal analysis. - -## Relationship to the Observe trace view - -The Traces tab is specific to Error Feed and only shows traces belonging to this cluster. It doesn't replace the Observe trace view: no complete span tree, no annotation editing. - -For span-level inspection of a specific trace, copy the Trace ID and look it up directly in Observe. - -## Next Steps - - - - Description and analysis for the selected trace. - - - Visualize where in the workflow traces fail. - - diff --git a/src/pages/docs/error-feed/features/trends.mdx b/src/pages/docs/error-feed/features/trends.mdx deleted file mode 100644 index 772604c7..00000000 --- a/src/pages/docs/error-feed/features/trends.mdx +++ /dev/null @@ -1,72 +0,0 @@ ---- -title: "Error Feed Trends: Score Trends and Activity Heatmap" -description: "How to use the Trends tab (Events Over Time, Score Trends, and the Activity Heatmap) to understand how an issue is evolving." ---- - -## About - -The **Trends** tab shows the temporal story. Is this getting worse? Did it spike after a deployment? What time of day does it concentrate? Are scores improving or degrading? - -![Trends tab showing Events Over Time chart, Score Trends, and Activity Heatmap](/images/docs/error-feed/features/trends-tab-overview.png) - -## Events Over Time - -How often errors in this cluster have occurred over the selected time range. Two series: - -- **Errors** (area/line): error occurrences per day in this cluster -- **Traffic** (bars): total trace volume Error Feed analyzed in the same period - -Showing traffic alongside errors is deliberate. A higher error count might just mean more traffic; the actual error *rate* could be stable. When the bars (traffic) and the line (errors) rise together proportionally, the rate is holding steady. When errors outpace traffic growth, the problem is genuinely getting worse. - -![Events Over Time chart with error line rising faster than traffic bars](/images/docs/error-feed/features/trends-events-over-time.png) - -### Reading spikes - -A sudden spike on a specific day is worth correlating with your deployment history. If you shipped a new model, updated a prompt, or changed tool configurations that day, the spike likely traces back to that change. - -A gradual upward slope (rather than a spike) means the error is tied to changing input distribution: the kinds of queries your users send are shifting in a direction that triggers this failure mode more often. - -## Score Trends - -How the four quality dimension scores (Factual Grounding, Privacy & Safety, Instruction Adherence, Optimal Plan Execution) have moved over time for traces in this cluster. - -Each dimension is a line chart. Declining line, quality is degrading. Rising line, improving. Flat means consistent, which could be consistently good or consistently bad depending on the absolute level. - -Most useful for tracking whether a deployed fix actually improved quality. After resolving an issue and deploying a change, watch Score Trends over the next few days to confirm the affected dimension is moving up. - -## Activity Heatmap - -A grid of error frequency by hour of day and day of week. Each cell is a specific hour on a specific day; darker cells mean higher error counts. - -![Activity Heatmap showing error concentration on weekday mornings](/images/docs/error-feed/features/trends-activity-heatmap.png) - -### What the heatmap tells you - -Patterns in the heatmap reveal whether the error is tied to usage. Common ones: - -- **Weekday mornings, low on weekends**: correlates with business-hours usage, likely triggered by specific user behavior rather than a random code bug -- **Uniform distribution**: the error happens randomly across hours and days, so it's input-independent and truly systematic -- **Specific hours**: a concentration at certain hours might correlate with a scheduled job, a batch process, or peak usage from a specific timezone - -These patterns help you tell whether the problem is urgent (consistent, all-hours) or a usage-pattern correlation that might resolve once the triggering input changes. - -## Combining the three views - -The most useful read is all three together. A typical investigation: - -1. **Events Over Time**: is the issue growing or shrinking? If growing, look at the rate relative to traffic. -2. **Score Trends**: are any quality dimensions trending down? If Factual Grounding has been declining for a week, something upstream changed. -3. **Heatmap**: is there a temporal pattern? Errors concentrating at 9am UTC every weekday is a clue about what triggers them. - -This combination often turns a confusing cluster into a clear, attributable pattern. - -## Next Steps - - - - Use trends to decide whether a fix held. - - - How the four quality scores are computed. - - diff --git a/src/pages/docs/error-feed/features/triage-workflow.mdx b/src/pages/docs/error-feed/features/triage-workflow.mdx deleted file mode 100644 index ee9f3e5a..00000000 --- a/src/pages/docs/error-feed/features/triage-workflow.mdx +++ /dev/null @@ -1,106 +0,0 @@ ---- -title: "Error Feed Triage Workflow: Resolve, Ignore, and Escalate" -description: "How to move issues through the Error Feed triage workflow: resolving, acknowledging, ignoring, assigning, and escalating." ---- - -## About - -Triage is reviewing new issues, deciding what to do with each one, and tracking them through to resolution. Error Feed is built to fit into your existing workflow rather than replace it: issues have statuses, assignees, and integrations that connect to whatever process you already use. - -## Status lifecycle - -An issue moves through four primary states: - -``` -Unresolved → Acknowledged → Resolved - ↕ - Escalating -``` - -**Unresolved** is the starting state. Every new cluster starts here. Filter the feed to Unresolved to see what hasn't been looked at. - -**Acknowledged** means someone has reviewed the issue and confirmed it's real and worth tracking. It doesn't mean a fix is in progress; it means the issue has left the "inbox." Use Acknowledged to reduce noise so your team knows what's new vs. what's known. - -**Escalating** means the issue is actively getting worse or a previous fix didn't hold. Use this to flag urgency beyond "this is open." Issues can go straight from Unresolved to Escalating if they warrant immediate attention. - -**Resolved** means the fix is deployed and the issue is closed. Error Feed keeps monitoring; if the same cluster pattern reappears, it creates a new issue rather than reopening the old one. This keeps resolved issues genuinely resolved and regressions visible. - -## Changing status - -Three paths: - -**Header buttons**: the fastest route. The issue detail header has Resolve, Acknowledge, and Ignore issue buttons. One click, no confirmation. - -**Status dropdown in the metadata panel**: click the current status chip in the Triage section for a dropdown with all states. Use this for states like Escalating that don't have a dedicated header button. - -**Feed list**: status changes made in the detail view show up in the list view immediately. No reload needed. - - -"Ignore issue" currently sets the status to Escalating, which doesn't permanently suppress the issue. To stop seeing an issue, set it to Resolved. - - -## Assigning issues - -Issues can be assigned to anyone in your organization. The assignee field is in the Triage section of the [metadata panel](/docs/error-feed/features/metadata-panel). - -Click **Assign** to open the picker, then pick a team member. To reassign, click the current assignee and pick a new one. To unassign, click the current assignee and select Unassign. - -Assignment is informational right now: it doesn't send a notification. For that, use the [Linear integration](/docs/error-feed/features/linear-integration) to create a ticket and assign it there. - -## A practical triage session - -Here's how to work through a backlog of issues efficiently: - - - - Start with the highest-severity issues nobody has looked at yet. Your "on fire right now" queue. - - In the Feed list, set Status to **Unresolved** and Severity to **Critical**. - - - The [Overview tab](/docs/error-feed/features/issue-overview)'s Description section says what's going wrong in plain language. For most issues you'll know within 30 seconds whether it's a real problem or a false positive. - - - Recognize the issue and already have a fix in flight: set it to **Acknowledged** and assign it to whoever owns the fix. - - Already fixed from a previous deploy: set it to **Resolved**. - - Not something you're going to act on: still Acknowledge it so others know it's been reviewed. - - - Use the [Trends tab](/docs/error-feed/features/trends) for severity and trajectory. Use the [State Graph](/docs/error-feed/features/state-graph) to see where in the workflow the failure concentrates. Use [Deep Analysis](/docs/error-feed/features/deep-analysis) when an issue warrants more investigation. - - - For issues that need a proper fix tracked in your project management tool, use the [Linear integration](/docs/error-feed/features/linear-integration) to create a ticket directly from the metadata panel. The ticket is linked to the cluster, so you can navigate between them. - - - Work down the severity ladder. High after Critical, Medium after High. Low-severity issues can be batched into a weekly review instead of triaged one by one. - - - -## Handling escalating issues - -An issue escalates when the problem is getting worse. In practice: watch the [Trends tab](/docs/error-feed/features/trends) Events Over Time chart. If the error count is rising week over week, move the issue to Escalating. - -Treat Escalating issues like Critical regardless of their assigned severity. Severity is about the *type* of problem; Escalating is about the *trajectory*. - -## After a fix is deployed - -When you ship a fix for a resolved issue: - -1. Set the issue to **Resolved** if it isn't already. -2. Come back 24–48 hours later and check the [Trends tab](/docs/error-feed/features/trends). Confirm Score Trends are improving and the Events Over Time count is dropping. -3. If a new issue appears in the same category with a similar name, the fix may have partially worked, or a regression was introduced. Investigate the new cluster. - -The goal isn't an empty issue list. It's making sure nothing critical sits unreviewed and that resolved issues actually stay resolved. - -## Next Steps - - - - What each status and severity tier means. - - - Create tickets from issues during triage. - - diff --git a/src/pages/docs/error-feed/guides/create-linear-issue.mdx b/src/pages/docs/error-feed/guides/create-linear-issue.mdx new file mode 100644 index 00000000..7f5c6890 --- /dev/null +++ b/src/pages/docs/error-feed/guides/create-linear-issue.mdx @@ -0,0 +1,50 @@ +--- +title: "Create a Linear issue" +description: "The team picker creates the ticket the moment you click, and nothing syncs back afterwards" +--- + +Turning an Error Feed issue into a Linear ticket gets the fix into your engineering team's actual backlog instead of leaving it to sit in the Feed. This covers that one job: linking one issue to one new Linear ticket. Linear must already be connected for the workspace, under **Settings > Integrations**, before any of this works. + +## Create the ticket + +Open the issue from the [Feed](/docs/error-feed/guides/triage-issues) to land on its detail page, then scroll the [metadata sidebar](/docs/error-feed/guides/investigate-an-issue) to the Integrations section. The Linear row reads **Create issue** once the workspace is connected. + +Click it, and a **Create Linear Issue** dialog opens with the line "Select a team to create the issue in." followed by your Linear teams. + + +Clicking a team creates the ticket immediately, with no confirm button. The link between the issue and that ticket is permanent and can't be removed from Error Feed, so check you've picked the right team before you click. + + +Click the team you want the ticket filed under. The Linear row then reads **View** followed by the issue ID, for example ENG-1234, and clicking it opens the ticket in Linear. The ticket takes its title from the Error Feed issue, truncated at 200 characters. + +A toast confirms it went through: "Created" plus the issue ID. If it fails instead, the toast reads "Failed to create Linear issue" and the row stays on **Create issue** so you can retry. + +### One ticket per issue + +Only one Linear issue can be linked per Error Feed issue. Once one exists, the row shows **View** instead of **Create issue**, and there's no separate button to link a second ticket. + +## What the team picker can show instead + +The **Create Linear Issue** dialog can land on one of four states instead of a team list: + +| Message | What to do | +|---|---| +| "Loading teams" | Wait a moment for the fetch to finish | +| "Couldn't reach Linear. Check the integration in Settings and try again." | Check the integration under Settings > Integrations and retry | +| "Linear isn't connected for this workspace. Connect it in Settings > Integrations." | Connect it under Settings > Integrations | +| "No teams found in your Linear workspace." | Add a team in your Linear workspace: there's none to file the ticket into | + +## Nothing syncs back + +Creating the ticket is one-directional. Closing, resolving, or otherwise updating the Linear ticket doesn't touch the Error Feed issue. The issue keeps whatever status it had when you created the ticket, so once the fix lands, resolve the Error Feed issue yourself. + +## Dive deeper + + + + Where creating a Linear ticket fits into resolving, acknowledging, and assigning issues + + + Get a written root cause and a proposed fix for the cluster behind the issue + + diff --git a/src/pages/docs/error-feed/guides/investigate-an-issue.mdx b/src/pages/docs/error-feed/guides/investigate-an-issue.mdx new file mode 100644 index 00000000..5c7a1ac5 --- /dev/null +++ b/src/pages/docs/error-feed/guides/investigate-an-issue.mdx @@ -0,0 +1,90 @@ +--- +title: "Investigate an issue" +description: "Read an issue's header and sidebar, then route between Overview, Traces, and Trends to confirm what's going wrong." +--- + +Open an issue from the [Feed](/docs/error-feed/guides/triage-issues) list and you land on its detail page: a header, a metadata sidebar, and a tab bar with **Overview**, **Traces**, **Trends**, and **Fix**, all views onto the same [cluster](/docs/error-feed/concepts/understanding-error-feed) of traces that failed the same way. Fix has its own guide; this one covers the other three. + +This guide works the detail page in the order that actually finds a problem: orient in the header, read the pattern on Overview, evidence the divergence with the trace evidence reel and split compare, drop into Traces only if the pattern doesn't hold up, and check Trends for how urgent it is. + +read the pattern-summary cards"] --> B{"One consistent
failure mode?"} + B -->|"yes"| C["Evidence it
evidence reel + split compare"] + B -->|"no, or need one run"| D["Traces tab
find the specific run"] + C --> E["Trends tab
how urgent is it?"] + D --> E`} /> + +## Header and sidebar + +The header and the right-hand sidebar stay fixed across every tab. The header carries: + +- A breadcrumb and error-type chip +- The issue title +- Status and severity badges +- A trace-count chip + +The sidebar holds status, severity, and assignee as editable controls. See [Triage issues](/docs/error-feed/guides/triage-issues) for how to use them. + +Use **Copy cluster ID** to paste the cluster identifier into a ticket or message, and **Share** for a direct link to this issue. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for what everything on the header and sidebar means. + +## Overview tab: is this one failure or several? + +Overview opens by default, and it's where every investigation starts: work out whether the cluster is one clean failure or several tangled together before you dig into individual traces. + +Overview tab with the pattern-summary cards, the events-and-users chart, and the trace evidence reel labeled +*The Overview tab: pattern-summary cards, the events-and-users chart, and the trace evidence reel* + +### Read the pattern + +The pattern-summary cards describe what's common across the whole cluster, not just one trace. Read them first: if they point at one consistent failure mode, you're likely looking at a single clean cluster; if they point in different directions, the cluster may be mixing more than one failure mode and needs a closer, trace-by-trace look. + +A chart below the cards plots events and users for the cluster, so you can see whether it's a steady trickle or a recent spike. + +### Open the evidence reel + +The trace evidence reel is on the Overview tab, with a switcher above it for its three view modes: + +- **Breadcrumb**: a linear read of what happened, the one to reach for first +- **Agent Graph**: every step the agent could take, useful for seeing whether the failure sits on one path among several or is the agent's only option +- **Agent Path**: the sequence this particular run actually took, useful for tracing exactly where this one run went sideways + +Within the reel, two tabs separate the evidence: **Failing** shows one failing trace at a time from those backing the pattern, and **Working** shows the nearest trace that succeeded. + +### Split-compare to find the divergence + +Toggle **Split with working** to line the open failing trace up against that nearest working trace (toggle **Single view** to go back to one trace at a time). This pairing is matched ahead of time by Error Feed, not a random working trace picked on the spot, so it's built to show exactly where the two runs diverge. + +Not every cluster has a working trace to pair against. If none was found, split compare has nothing to show; work from the Traces tab instead. + +## Traces tab: find the specific run + +If Overview's pattern doesn't hold up under a closer look, or you need one specific run rather than the aggregate, drop into the Traces tab: it lists the cluster's traces, one row each. + +Five aggregate cards sit at the top: **Total traces**, **Avg score**, **Avg turns**, **P50 latency**, **P95 latency**. The grid below carries a column for each: **Trace ID**, **Input**, **Start Time**, **Duration**, **Tokens**, **Cost**, **Score**. Click any row to open it in the trace drawer for the full detail. + + +Voice and simulator projects open a different trace drawer here. See [Voice observability](/docs/observe/features/voice) and [Explore results](/docs/simulation/guides/explore-results). + + +## Trends tab: is this urgent? + +Trends is a single chart: errors and traffic plotted on two axes over time. Read the two lines as a pair, not separately. If the error line climbs while traffic barely moves, something got worse in the system itself. If both climb together, you're most likely looking at more volume, not a rising failure rate, which is often enough on its own to tell you whether an issue is urgent or just a side effect of growth. + +## Dive deeper + + + + Change status, severity, and assignee once you know what's wrong + + + Get a written root cause and a proposed fix from the Fix tab + + + Turn the finding into a ticket your team can work from + + + The full list of columns, cards, and values referenced on this page + + diff --git a/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx b/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx new file mode 100644 index 00000000..4591eea1 --- /dev/null +++ b/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx @@ -0,0 +1,34 @@ +--- +title: "Run a root cause analysis" +description: "Get a written root cause and a proposed fix for one cluster from the Fix tab." +--- + +The **Fix** tab is Error Feed's agentic root-cause chat thread, with a follow-up composer underneath it. Start it and sub-agents sample representative calls from the [cluster](/docs/error-feed/concepts/understanding-error-feed), compare them against a passing baseline, and synthesise a written root cause and a proposed fix in the thread. This page walks through starting a run, reading what comes back, asking a follow-up, and re-running it later. + +## Run the analysis + +Start from an issue's detail page, with the failure pattern already confirmed via [Investigate an issue](/docs/error-feed/guides/investigate-an-issue). + +From the issue's detail page, click into the Fix tab. If nothing has run yet it shows an empty state, **No analysis yet**, with a button labeled **Analyze this cluster**. You can also start it from the issue's headline card, whose button reads **Debug this cluster** before a run exists. Both start the same run, and each uses 1 credit, taken when the run starts and refunded if the run fails. + +Fix tab empty state with the Analyze this cluster button labeled, alongside the headline card's Debug this cluster button +*The Fix tab's empty state, before any run has started* + +The run starts immediately and the tab keeps checking until it lands or fails, with a one-hour cut-off. The result arrives as a single written message in the thread: a root cause explaining what's going wrong, and a proposed fix for it, based on the calls Falcon sampled. To probe the reasoning further, type into the composer at the bottom (placeholder: **Ask Falcon a follow-up...**), and **Falcon is investigating...** shows while a reply streams in. + +Once the cluster has picked up new traces, click **Re-run** in the tab's header (tooltip: **Re-run with current cluster state (1 credit)**) to analyze it against its current state. The headline card carries the same option once a run exists, tooltipped **Re-run analysis (1 credit)**. + +## If it fails + +A run can fail with one of two messages: **Couldn't start the analysis. Please try again.** or **Couldn't connect to the server. Please try again.** For what each one means and what to do about it, see [Analysis doesn't finish](/docs/error-feed/troubleshooting/analysis-does-not-finish). + +## Dive deeper + + + + Turn the finding into a ticket your team can work from + + + What to check when a run stalls or never lands + + diff --git a/src/pages/docs/error-feed/guides/triage-issues.mdx b/src/pages/docs/error-feed/guides/triage-issues.mdx new file mode 100644 index 00000000..f8320629 --- /dev/null +++ b/src/pages/docs/error-feed/guides/triage-issues.mdx @@ -0,0 +1,75 @@ +--- +title: "Triage issues" +description: "Narrow a full feed to what's worth acting on, then resolve, acknowledge, or reassign the issues that matter." +--- + +Error Feed's [list page](/docs/error-feed/guides/triage-issues), in the left sidebar under **Error Feed** (see the [overview](/docs/error-feed) if you haven't opened it yet), shows every detected issue across your projects, scoped to the last 7 days until you change the range. A full feed is a queue, not a to-do list: some rows need attention today, most don't. + +This guide takes you from a full feed to a handled list: narrow it to what's worth looking at, scan the table for what actually decides priority, then act on what you find, one issue at a time or several at once. + +## Narrow the feed + +Type into the search box (placeholder **Search errors**) to match against the error name, issue group, or category. Next to it sit five selects: project, status, severity, and fix layer each open on an All value until you narrow them, while time range opens already scoped to Last 7 days. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the exact set of values each one accepts. + + +Start from **Severity: Critical** and **Status: Escalating**. Critical narrows to what matters most, and Escalating is the one status of the four that hasn't settled into Acknowledged, For review, or Resolved. See [Severity & Status](/docs/error-feed/concepts/severity-and-status) for when to pick each one. + + +Once project, status, severity, or fix layer is set, a **Clear** button appears next to the selects (tooltip: "Clear all filters") to reset everything in one click instead of undoing each select by hand. + +## Scan the table + +Each row is one issue. Eight columns run left to right: **Error**, **Severity**, **Status**, **Events**, **Users**, **Fix Layer**, **Trend (14d)**, and **Last seen**. + +Three of them decide priority. **Severity** says how bad it is, **Status** says where it sits in your workflow, and **Trend (14d)** shows whether it's climbing, flat, or settling down. + +The rest is context: + +- **Error** names what's failing +- **Events** and **Users** size the blast radius +- **Fix Layer** points at where in your system the fix belongs +- **Last seen** says when it last fired + +At the bottom, set **Results per page** to 10, 25, or 50, and move through the rest with **Back** and **Next**. + +## Act on what you find + +Three ways to act, each suited to a different job in the [triage workflow](/docs/error-feed/guides/triage-issues): + +- **Bulk actions** for many rows moving to the same status at once +- **Header buttons** for a quick resolve or acknowledge on a single issue +- **Metadata sidebar** for a severity or assignee change on a single issue + + +None of the three ways below show a toast. There's no confirmation of success and no warning on failure, so the save happens silently either way. To check a change went through, re-check the **Status** (or **Severity**) column for that row, or refresh the feed; if the value hasn't moved, repeat the action. + + +### Handle many at once + +Tick the checkbox on any row and a bulk-select toolbar appears above the table. Tick more rows, then open **Bulk actions** and pick **Mark as Resolved**, **Mark as Acknowledged**, **Mark as For Review**, or **Mark as Escalating** to move every selected issue to that status in one go. + +### Resolve or acknowledge one issue + +Click a row to open the issue. Its header carries three buttons: **Resolve** moves the issue to resolved, **Acknowledge** moves it to acknowledged, and **Ignore issue** moves it to escalating despite the label. All three disable themselves while the update is in flight. + +### Change status, severity, or assignee from the sidebar + +Every issue also has a metadata sidebar on the right. Its **Status** and **Severity** rows each open a menu to set a new value directly. The **Assignee** row reads **Assign** until someone's on it; click it to open a menu headed **Assign to**, listing everyone in your org plus an **Unassign** option once someone's set. + + +Assigning someone to an issue only records it on the issue itself. It sends no notification of any kind, so tell them yourself if it needs to reach them. + + +## Dive deeper + + + + Read the evidence behind one issue before you touch the fix + + + The two independent axes every issue carries, and how they change + + + Every filter, column, and enum value in the feed + + diff --git a/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx b/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx new file mode 100644 index 00000000..d421201a --- /dev/null +++ b/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx @@ -0,0 +1,53 @@ +--- +title: "Turn on Error Feed" +description: "Turn on trace scanning for a project so Error Feed can find its first issues." +--- + +Error Feed ships off. A project's scanner starts at a 0% sampling rate, so nothing gets scanned until you raise it. This guide gets a project from silent to its first issues showing up in the Feed. + +## Before you start + +- The project is already receiving traces in [Observe](/docs/observe). Error Feed only scans what Observe receives, so [send your agent through a request](/docs/observe/quickstart) first if none have arrived yet +- Your workspace needs the Error Feed capability; without it, the Error Feed page shows the upgrade message and the API answers 402 + +## Turn on scanning + +Open the Observe project you want issues for, then click the settings gear icon, tooltipped **Settings**, in the project header. A drawer titled **Configure Project** opens, carrying the project's settings including sampling. + +Find the sampling rate control in that drawer and raise it above 0. 100% is a safe default while you're trying it out; see [Choosing a rate](#choosing-a-rate) for the cost tradeoff once volume climbs. Click **Update** to apply it. + +Then wait before checking for results, because a new rate only reaches traces that arrive after you save it. Send your agent through a request, give scanning a moment, and go to **Error Feed** in the left sidebar. A row appearing in the list confirms scanning is live. An upgrade prompt instead of the Feed means the workspace lacks the Error Feed capability. + +## Choosing a rate + +There's no universally right number; it's a coverage-versus-cost call: analyze more traces and you catch more, but you pay more for it. + +| Situation | Rate | +|-----------|------| +| Building or testing a project | 100%, so nothing slips past you | +| Low-volume production | 100%, the absolute cost stays low | +| High-volume production | 10–20%, enough to catch recurring issues | +| Cost-constrained, high volume | 5–10%, still catches patterns that repeat | + +A rate change only reaches forward. It applies to traces that arrive after you save it, not to anything that already went by. + +## When scanning runs + +Scanning is triggered per trace, not on a timer. Once a trace's root span completes, Error Feed waits about ten seconds before a scan starts and samples it. + +Traces that arrive through the [collector](/docs/error-feed/troubleshooting/no-issues-in-the-feed#the-traces-came-in-through-the-collector) instead of the inline path skip that trigger. A periodic sweep picks them up instead, working through them in small batches rather than the moment they land. If your traces go through the collector, expect the first issues to show up in occasional bursts rather than a steady trickle. + +## If nothing shows up + +If you've confirmed traces are reaching the project and the rate is saved, see [No issues in the Feed](/docs/error-feed/troubleshooting/no-issues-in-the-feed) for the full list of causes, in order. + +## Dive deeper + + + + The mental model: how a sampled trace becomes an issue in the Feed + + + Where your first issues show up, and how to read the list + + diff --git a/src/pages/docs/error-feed/index.mdx b/src/pages/docs/error-feed/index.mdx index 23a85838..ecf1c995 100644 --- a/src/pages/docs/error-feed/index.mdx +++ b/src/pages/docs/error-feed/index.mdx @@ -1,74 +1,39 @@ --- -title: "Future AGI Error Feed: AI Agent Trace Error Detection" -description: "Automatically detect, cluster, score, and triage errors in your AI agent traces, without any configuration beyond standard tracing." +title: "Overview" +description: "Error Feed reads your traces, groups the problems it finds into issues, and points at the layer to fix" --- -## About +Errors in an AI system rarely show up as one clean failure. They show up as a pattern: the same kind of mistake repeating across dozens of requests, buried in traces you'd otherwise have to read one by one to catch. -Error Feed is Future AGI's error monitoring for AI agents. As soon as traces hit an Observe project, it picks them up, finds failure patterns, groups similar ones together, and writes up the analysis. No extra setup. +## What is Error Feed? -Think Sentry, but for the ways agents actually fail: hallucinated outputs, tool misuse, broken workflows, safety violations, and reasoning gaps that traditional error monitoring won't catch. +**Error Feed** reads a sample of the traces in an [Observe](/docs/observe) project, decides for itself what went wrong in each one, and groups the traces that went wrong the same way into a single issue you work like a ticket. Nothing has to mark a trace as failed first: the scan is what finds the problem. Each issue carries a severity, a status, an assignee, and an optional link to a [Linear](/docs/error-feed/guides/create-linear-issue) ticket, and points at the fix layer, the part of your system the fix actually belongs in. -![Error Feed list view showing clustered issues with severity badges and trend sparklines](/images/docs/error-feed/index/feed-list-overview.png) +## Before you start -## What it does +Error Feed requires an Enterprise or Cloud license. -Error Feed runs in the background on every Observe project. For each trace it analyzes, it: +Scanning ships off because a project's sampling rate starts at 0. Nothing is scanned, and no issues appear, until you [raise it](/docs/error-feed/guides/turn-on-error-feed). -- **Detects errors** in five categories, from factual grounding failures to tool crashes to safety violations. See the full [error taxonomy](/docs/error-feed/concepts/taxonomy). -- **Groups related traces** into named clusters, so 50 traces with the same underlying problem show up as one issue instead of 50 alerts. -- **Scores the trace** on four quality dimensions, each on a 0–5 scale. See [Scoring](/docs/error-feed/concepts/scoring). -- **Generates analysis**: what went wrong, root causes, supporting evidence from the trace, plus a quick fix and a long-term recommendation. -- **Tracks trends**: whether an issue is happening more often, less often, or staying steady. +## Start here - -No configuration needed. Error Feed turns on automatically for any Observe project the moment traces start arriving. - - -## Who it's for - -Useful whether you're debugging an agent that just started misbehaving, doing a quality review, or trying to spot systemic problems across thousands of production traces. - -You don't need to know how transformers work to use it. The UI explains what went wrong in plain language. If you do want to dig into trace-level evidence, every finding links straight to the spans involved. - -## Supported integrations - -Error Feed works with any integration that sends traces to a Future AGI Observe project. - -**LLM providers**: OpenAI, OpenAI Agents SDK, Vertex AI (Gemini), AWS Bedrock, Mistral AI, Anthropic, Groq, Together AI, Google ADK, Google GenAI, Portkey - -**Orchestration frameworks**: LlamaIndex, LlamaIndex Workflows, LangChain, LangGraph, LiteLLM, CrewAI, Haystack, Autogen, PromptFlow, Vercel, Pipecat - -**Other**: DSPy, Guardrails AI, Hugging Face smolagents, Ollama, Instructor, MCP - -## Navigate the docs - - - - The mental model: how traces become issues, clusters, and scored findings. - - - The five error categories and every subcategory Error Feed can detect. + + + Raise the sampling rate on a project and get your first issues - - The four quality metrics, what they measure, and how to read scores. + + The object model behind an issue, from a single finding to the fix layer - - Severity tiers and the triage status workflow — from new issue to resolved. - - - - - - Filters, stats bar, columns, sparklines — the issue list page. + + The two independent axes every issue carries, and how they change - - The Overview tab: description, root cause, evidence, and recommendations. + + Filter, scan, and work a feed down with bulk actions - - On-demand deeper analysis for issues that need more investigation. + + Read the evidence behind one issue before you touch the fix - - Resolve, acknowledge, assign — how to move issues through your process. + + Get a written root cause and a proposed fix for one issue diff --git a/src/pages/docs/error-feed/reference/error-taxonomy.mdx b/src/pages/docs/error-feed/reference/error-taxonomy.mdx new file mode 100644 index 00000000..59721b44 --- /dev/null +++ b/src/pages/docs/error-feed/reference/error-taxonomy.mdx @@ -0,0 +1,68 @@ +--- +title: "Error taxonomy" +description: "The fixed groups, categories, and fix layers every Error Feed finding is classified into" +--- + +## The shape of a finding + +Every scanner [finding](/docs/error-feed/concepts/understanding-error-feed) Error Feed writes carries three tags: a **group**, a **category** inside that group, and a **fix layer** (the part of your system the finding points at). Eval-sourced findings are tagged differently: the eval name stands in for group, and category is left unset. Fix layer is the only one of the three that's a live filter on the [feed](/docs/error-feed/guides/triage-issues), with options for All Fix Layers, Prompt, Tools, Orchestration, and Guardrails. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the full list of fields and filters. + +The set is fixed, not free-form. A scanner finding always lands in exactly one row of the table below, and points at one of four fix layers. + +5 categories"] --> T["Tools"] + B["Context & Retrieval
4 categories"] --> P["Prompt"] + D["Output Quality
3 categories"] --> P + C["Planning & Goals
3 categories"] --> O["Orchestration"] + E["Infrastructure
5 categories"] --> G["Guardrails"]`} /> + +## Fix layers + +- **Tools**: fix layer for Tool Failures findings +- **Prompt**: fix layer for Context & Retrieval and Output Quality findings +- **Orchestration**: fix layer for Planning & Goals findings +- **Guardrails**: fix layer for Infrastructure findings + +## Groups, categories & fix layers + +Five groups organize twenty categories. + +| Group | Category | What it means | Fix layer | +|---|---|---|---| +| Tool Failures | Tool-related | The agent mishandled a tool or its result: asserted success after an error, passed wrong arguments, or guessed instead of calling | Tools | +| Tool Failures | Tool Selection Errors | The agent picked the wrong tool, or no tool, for the task | Tools | +| Tool Failures | Tool Output Misinterpretation | The agent misread or misused the result a tool returned | Tools | +| Tool Failures | Formatting Errors | The agent's tool call or output didn't match the expected format | Tools | +| Tool Failures | Language-only | A hallucination purely in language, no tool involved | Tools | +| Context & Retrieval | Context Handling Failures | The agent lost, dropped, or mishandled context it was given | Prompt | +| Context & Retrieval | Poor Information Retrieval | The agent retrieved information that was irrelevant, incomplete, or wrong | Prompt | +| Context & Retrieval | Incorrect Memory Usage | The agent used stored memory incorrectly, including outdated or unrelated memory | Prompt | +| Context & Retrieval | Unsupported Claim | The agent stated something that tool output or the end user's own input doesn't support | Prompt | +| Planning & Goals | Task Orchestration | The agent sequenced or delegated steps incorrectly | Orchestration | +| Planning & Goals | Goal Deviation | The agent drifted from the goal it was given | Orchestration | +| Planning & Goals | Resource Abuse | The agent used excessive steps, calls, or resources to complete the task | Orchestration | +| Output Quality | Instruction Non-compliance | The agent's output didn't follow the instructions it was given | Prompt | +| Output Quality | Incorrect Problem Identification | The agent misunderstood or misidentified the problem it was asked to solve | Prompt | +| Output Quality | Incomplete Response | The agent's response was absent, empty, or truncated | Prompt | +| Infrastructure | Environment Setup Errors | The agent's runtime environment wasn't configured correctly | Guardrails | +| Infrastructure | Resource Not Found | The agent tried to reach a resource that doesn't exist | Guardrails | +| Infrastructure | Authentication Errors | The agent failed to authenticate with a required service | Guardrails | +| Infrastructure | Timeout Issues | A call the agent depended on didn't complete in time | Guardrails | +| Infrastructure | Service Errors | A service the agent depended on returned an error | Guardrails | + + +Five groups map onto only four fix layers: Context & Retrieval and Output Quality both resolve to Prompt, so a poor retrieval and an incomplete response can carry the same fix layer even though they belong to different groups. + + +## Keep exploring + + + + Filter, sort, and triage issues in the list view + + + Every field and filter across the feed's UI and APIs + + diff --git a/src/pages/docs/error-feed/reference/issue-fields.mdx b/src/pages/docs/error-feed/reference/issue-fields.mdx new file mode 100644 index 00000000..b3eb8358 --- /dev/null +++ b/src/pages/docs/error-feed/reference/issue-fields.mdx @@ -0,0 +1,138 @@ +--- +title: "Issue fields & filters" +description: "Filter options, table columns, field values, and limits for the Error Feed's issue list" +--- + +## Feed filters + +UI controls on the feed's filter bar that pick from a fixed set of values, with the exact options each one offers. The filter bar also carries a project select, scoped to your org's projects, and a free-text **Search errors** box; this table covers only the selects with a fixed set of options. + +| Filter | Options | +|---|---| +| Time range | Last 24 hours, Last 7 days, Last 14 days, Last 30 days, Last 90 days | +| Status | All Statuses, see Status values below | +| Severity | All Severities, see Severity values & priority below | +| Fix layer | All Fix Layers, Prompt, Tools, Orchestration, Guardrails | + +## Sort keys & directions + +The `sort_by` values the feed list enforces, and the feed table column header that triggers each one. The feed table only sorts on Severity, Events, and Last seen, so `first_seen` and `error_count` aren't reachable from the UI at all. + +| `sort_by` | Column header | Default | +|---|---|---| +| `last_seen` | Last seen | Default | +| `first_seen` | Not exposed in the UI | | +| `error_count` | Not exposed in the UI | | +| `unique_traces` | Events | | +| `severity` | Severity | | + +`sort_dir` takes `asc` or `desc`, and defaults to `desc`. + +## Status values + +The full set of values an issue's `status` can hold, used by both the UI filter and the feed's `status` value. See [Severity & Status](/docs/error-feed/concepts/severity-and-status) for what each one means. + +| Status | Label | +|---|---| +| `escalating` | Escalating | +| `for_review` | For review | +| `acknowledged` | Acknowledged | +| `resolved` | Resolved | + +## Severity values & priority + +The full set of values an issue's `severity` can hold, used by both the UI filter and the feed's `severity` value, and the `priority` value each is stored as. + +| Severity | Label | Stored as | +|---|---|---| +| `critical` | Critical | `urgent` | +| `high` | High | `high` | +| `medium` | Medium | `medium` | +| `low` | Low | `low` | + +## Fix layers + +The fix layers a finding can point at; used by both the UI filter and the feed table's Fix Layer column. See [Error taxonomy](/docs/error-feed/reference/error-taxonomy) for the full group-to-category breakdown behind each one. + +| Fix layer | +|---| +| Prompt | +| Tools | +| Orchestration | +| Guardrails | + +## Source values + +Where an issue's underlying [finding](/docs/error-feed/concepts/understanding-error-feed) came from, and what each source value means. + +| Source | Meaning | +|---|---| +| `scanner` | Default source | +| `eval` | Set when an eval failure produced the finding; eval-sourced findings also carry an `eval_target_type` of `span`, `trace`, or `session` | + +## Event count & time fields + +Fields on an issue that count its events and place it in time. + +| Field | Counts | +|---|---| +| `total_events` | Every occurrence of the issue | +| `unique_traces` | Distinct traces the occurrences fall across | +| `unique_users` | Distinct users who hit the issue | +| `first_seen` | Time of the issue's earliest occurrence | +| `last_seen` | Time of the issue's most recent occurrence | + +## Traces tab columns & aggregates + +UI columns and summary cards on an issue's Traces tab. + +| Column | +|---| +| Trace ID | +| Input | +| Start Time | +| Duration | +| Tokens | +| Cost | +| Score | + +| Aggregate card | +|---| +| Total traces | +| Avg score | +| Avg turns | +| P50 latency | +| P95 latency | + +## List & query limits + +Values and bounds the feed and its tabs enforce: page sizes, default and maximum result counts, and the trends day window. A dash means that bound doesn't apply, only the minimum shown is enforced. The `limit` parameter appears twice below because it's bound differently on different surfaces of the app; the **Applies to** column says which surface each row's bounds belong to. + +| Parameter | Applies to | Min | Default | Max | +|---|---|---|---|---| +| `limit` | **Feed list** | 1 | 25 | 200 | +| `offset` | **Feed list** | 0 | 0 | – | +| `time_range_days` | **Feed list** | 1 | – | – | +| `limit` | **Traces tab** | 1 | 50 | 500 | +| `rep_limit` | **Overview** | 1 | 20 | 200 | +| `days` | **Trends** | 1 | 14 | 90 | + +| Feed table page size (UI) | +|---| +| 10 | +| 25 | +| 50 | + +## Keep exploring + + + + Filter, sort, and triage issues in the list view + + + The two independent axes every issue carries, and how they change + + + The fixed groups, categories, and fix layers behind every finding + + diff --git a/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx b/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx new file mode 100644 index 00000000..81464c6a --- /dev/null +++ b/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx @@ -0,0 +1,46 @@ +--- +title: "Analysis doesn't finish" +description: "Four causes for a stalled Fix tab run, matched to the message you see." +--- + +A cluster's [root cause analysis](/docs/error-feed/guides/run-root-cause-analysis) on the **Fix** tab can stall instead of landing a result. Match what you see against the causes below: + +- `Couldn't start the analysis. Please try again.` or `Couldn't connect to the server. Please try again.`: [The run never starts](#the-run-never-starts) +- `Couldn't reach the investigator — the connection dropped. Hit Re-run.`: [Connection dropped after the run started](#connection-dropped-after-the-run-started) +- Same message, but the workspace has no credit left: [Workspace out of credit](#workspace-out-of-credit) +- No message, but the run's been going for an hour: [Run exceeded the one-hour cap](#run-exceeded-the-one-hour-cap) + +The credit is taken when the run starts and refunded if the run fails, so a failed run should net out to nothing. + +## The run never starts + +`Couldn't start the analysis. Please try again.` or `Couldn't connect to the server. Please try again.` The request to start the run failed outright. Nothing started. Re-run to try again. + +## Connection dropped after the run started + +`Couldn't reach the investigator — the connection dropped. Hit Re-run.` The run did start. The connection carrying its progress back died before anything came through. Re-running is usually safe, but the same message also shows up when the workspace has no credit left, and re-running there just spends another credit without landing a result. Check [Workspace out of credit](#workspace-out-of-credit) before you re-run again. + +## Run exceeded the one-hour cap + +Every run has a one-hour limit. If it's still going when that's reached, the run is cut off and no result lands in the thread. Re-run to start a fresh attempt. + +## Workspace out of credit + +This shows up as the same `Couldn't reach the investigator — the connection dropped. Hit Re-run.` message you'd see from a dropped connection, not as nothing happening. The workspace needs credit before a run can complete. If there's none left, the run fails with that message. Add credit to the workspace, then re-run. + +## What to do + +Before you re-run, confirm the cluster still has traces inside the [time range](/docs/error-feed/guides/triage-issues) you've got selected on the Feed. If the window has moved past everything in the cluster, widen or shift it so the cluster's traces fall inside, since re-running against an empty range spends a credit for nothing. + +Press **Re-run** in the cluster's header. Its tooltip reads `Re-run with current cluster state (1 credit)`, since each run, including a re-run, draws a fresh credit. + +## Dive deeper + + + + Start a run, read the synthesis, and ask a follow-up + + + For when the Feed itself has nothing to analyze in the first place + + diff --git a/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx b/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx new file mode 100644 index 00000000..8c2bd9d3 --- /dev/null +++ b/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx @@ -0,0 +1,70 @@ +--- +title: "Issue counts look wrong" +description: "Why a populated Feed's numbers don't match what you expected" +--- + +The Feed has rows, nothing looks empty or filtered away, but a number on it doesn't match what you expected: a total that's lower than you'd guess, a row that's gone missing, or two columns that don't seem to agree with each other. Find your symptom below: + +- Total is lower than you expected: [Counts only cover what got scanned](#counts-only-cover-what-got-scanned) +- Row you were watching has vanished: [Duplicate clusters get folded together](#duplicate-clusters-get-folded-together) or [The time range drops rows, but not their counts](#the-time-range-drops-rows-but-not-their-counts) +- Events looks lower than the number of occurrences you know about: [The Events column is really a trace count](#the-events-column-is-really-a-trace-count) +- Trend sparkline doesn't match the totals next to it: [The Trend sparkline runs on a fixed 14-day window](#the-trend-sparkline-runs-on-a-fixed-14-day-window) +- Count still doesn't add up after accounting for sampling: [The list mixes scanner and eval issues](#the-list-mixes-scanner-and-eval-issues) + +Each row in the Feed is a cluster: one or more matching findings grouped into a single issue. See [Understanding Error Feed](/docs/error-feed/concepts/understanding-error-feed) for how clusters and categories form, and [Error taxonomy](/docs/error-feed/reference/error-taxonomy) for the fixed set a row's category comes from. + +## Counts only cover what got scanned + +Every project has a sampling rate between 0% and 100%, and it decides what fraction of traces are ever scanned in the first place. A count on the Feed only ever reflects scanned traces, never every trace that actually ran. + +That gap gets big fast at a low rate. At a sampling rate of 20%, roughly one trace in five gets scanned, so five traces that hit the exact same failure can turn into a single scanned occurrence. Read literally, that looks like the error happened once. It happened five times; only one of those times got sampled. + +Fix: if a count seems too low for how often you believe something is failing, that's the sampling rate doing its job, not a bug in the count. Raise the rate so more traces get scanned. That only affects traces scanned from that point on; it doesn't rescan what already ran, so counts you're currently looking at won't change. See [Turn on Error Feed](/docs/error-feed/guides/turn-on-error-feed) for where that control lives. + +## Duplicate clusters get folded together + +Two clusters get folded into one when they're in the same category, each is the other's closest match in both directions, and the distance between them is within the merge threshold. If cluster A's nearest neighbor is B, but B's nearest neighbor is something else, they don't fold. A mutual match in the same category that's still too far apart doesn't fold either; all three conditions have to hold together. + +When a fold happens, the cluster with the larger member count absorbs the other, and the absorbed one stops appearing in the Feed as its own row. Its occurrences don't disappear, they now count toward the cluster that absorbed it. So an issue you were watching yesterday can vanish from the list today, not because it resolved, but because it was the smaller, untriaged side of a mutual match and got absorbed into a bigger cluster in the same category. A cluster you've already triaged is protected from this: it survives the merge even if it would otherwise be the smaller side. + +Fix: if a row you expected is missing, look for a similar issue in the same category with a higher count than you remember. That's very likely where it went. + +## The Events column is really a trace count + +A cluster row carries several separate counts, and what shows up in the Feed table isn't a plain readout of them. The column labeled **Events** doesn't count events at all, it renders unique traces, the number of distinct traces the cluster matched. + +The **Users** column is a distinct end-user count, separate from Events. + +Fix: don't read Events as an occurrence count, it's the unique-trace count sitting under a misleading header. Total occurrences aren't shown as a column anywhere in the table. + +## The time range drops rows, but not their counts + +Changing the Feed's time range changes which clusters qualify for the list, not what their numbers say. A cluster only stays in the list when it was last seen within the selected range; narrow the range and clusters that fall outside it disappear from the table entirely. The Events and Users figures on a row that does survive aren't windowed to that range at all, they're lifetime values, so they don't shrink just because you picked a narrower range. + +Fix: if a row you expected is missing after narrowing the time range, that's the row falling outside the last-seen window, not its counts dropping to zero. Widen the range and the row reappears with the same lifetime totals it always had. + +## The Trend sparkline runs on a fixed 14-day window + +The Trend column doesn't follow the time range you've set for the rest of the table. It's labeled **Trend (14d)** and stays fixed at 14 days no matter what range you pick, so it won't line up with the Events or Users totals sitting next to it in the same row. Those totals are lifetime counts, not range-bound ones, so this isn't something you can tune away by adjusting the time range, the mismatch is permanent. + +Fix: read the sparkline as a separate signal, not a breakdown of the totals beside it. + +## The list mixes scanner and eval issues + +Issues on the Feed come from two different sources: the automatic scanner working through sampled traces, and evaluations. Both land in the same list and count toward the same totals unless you filter by source. + +Fix: if you're trying to reconcile a count against the sampling math in [Counts only cover what got scanned](#counts-only-cover-what-got-scanned) and it's not adding up, check whether some of the rows you're counting are eval-created rather than scanner-created. Source isn't a column in the Feed table, so you can't tell by looking at a row; filter the list to a single source instead. See [Issue fields & filters](/docs/error-feed/reference/issue-fields#source-values) for the source values and the filter that isolates them. + +## Dive deeper + + + + For when the table has zero rows, not just numbers that look off + + + The mental model behind findings, clusters, and how issues form + + + Where the sampling rate lives and how to raise it + + diff --git a/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx b/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx new file mode 100644 index 00000000..f6297c66 --- /dev/null +++ b/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx @@ -0,0 +1,75 @@ +--- +title: "No issues in the feed" +description: "Five causes for a Feed with zero rows, in the order to check them." +--- + +The Feed loads, but the table is empty. Work out which situation you're in before you touch anything: + +1. Which empty-state message does the table show? [Filters are hiding the rows](#filters-are-hiding-the-rows) quotes both in full. + - The filtered message means the data may already be there, just filtered out, so skip straight to that section + - The unfiltered message means no issues have been found. Work through the causes below, in order +2. Check the causes in order: + - [The workspace has no Error Feed license](#the-workspace-has-no-error-feed-license) + - [The sampling rate is still 0](#the-sampling-rate-is-still-0) + - [The traces are too new](#the-traces-are-too-new) + - [The traces came in through the collector](#the-traces-came-in-through-the-collector) + - [Filters are hiding the rows](#filters-are-hiding-the-rows) + +The list starts with the license check because it's the cheapest to rule out: the answer is visible on the page you're already looking at. If you've worked through all five causes and the Feed is still empty, confirm traces are reaching this project at all: see [No traces appearing](/docs/observe/troubleshooting/no-traces-appearing). + +## The workspace has no Error Feed license + +Without the Error Feed capability, the Feed page can't show a table at all: it shows an upgrade message instead, and a direct API call for feed data comes back with a 402. + +The message reads **"This feature requires an upgrade."**, with a reason code and a **Contact us to upgrade** button. If instead you see **"Couldn't verify feature access."** with a **Retry** button, that's a different problem: the check itself failed transiently, not a licensing block, so retry it. + +Fix: if you're looking at "This feature requires an upgrade.", this is your cause. Use the **Contact us to upgrade** button. + +## The sampling rate is still 0 + +Error Feed ships with a project's [sampling rate](/docs/error-feed/guides/turn-on-error-feed) at 0, which disables scanning entirely. Nothing gets sampled, so nothing can ever reach the Feed. This is the shipped default, not something anyone had to break, so it's by far the most common reason the Feed is empty. + +Fix: raise the project's sampling rate above 0. See [Turn on Error Feed](/docs/error-feed/guides/turn-on-error-feed) for where that control lives and how to pick a rate. + +## The traces are too new + +Scanning is triggered per trace, not on a timer, and only once a trace's root span has completed. Even then, Error Feed waits about ten seconds before sampling and scanning it. A trace that finished moments ago hasn't necessarily been scanned yet. + +Fix: give it roughly ten seconds after the trace completes, then refresh the Feed. + +## The traces came in through the collector + +Traces that arrive through the collector don't trigger a scan on arrival. They wait for a periodic sweep instead, which adds its own grace period on top of the ten-second wait above. Check with whoever set up tracing for this project to see whether traces route through the collector. + +The sweep dispatches a scan task for every 15 pending traces, and each trace holds for a 60-second grace period before it's eligible. Because collector-routed traces are picked up by that sweep rather than one at a time, they show up in occasional bursts rather than the steady trickle you'd see from a trace that triggers its own scan. + +Fix: wait at least 60 seconds after the trace lands before assuming scanning isn't working, since that's the grace period each trace holds before it's even eligible for a sweep, and scanning still waits the same ten seconds after that. Expect issues to land in bursts rather than one at a time. + +## Filters are hiding the rows + +The table's empty state tells you which situation you're actually in. + +- If your filters exclude everything currently in the Feed, it shows **"No errors match your filters"** / **"Try adjusting your search or filter criteria."** +- If there genuinely are no issues, it shows **"No errors - everything looks good!"** / **"Errors captured by Future AGI will appear here."** + +Fix: if you're looking at the first message, clear or widen your [filters](/docs/error-feed/guides/triage-issues). The **Clear** control: + +- resets project, status, severity, fix layer, and search +- never resets the time range +- only appears once one of project, status, severity, or fix layer is set + +So widen a narrow time range yourself. Neither message rules the time range out, since it isn't part of what the table checks: widen it before you go back through the causes above. + +## Dive deeper + + + + Raise the sampling rate and get a project scanning for the first time + + + The mental model behind findings, clusters, and how issues form + + + For when the Feed has rows, but a number on it doesn't add up + + diff --git a/src/pages/docs/evaluation/guides/advanced-usage.mdx b/src/pages/docs/evaluation/guides/advanced-usage.mdx index 26132316..8ae4ff97 100644 --- a/src/pages/docs/evaluation/guides/advanced-usage.mdx +++ b/src/pages/docs/evaluation/guides/advanced-usage.mdx @@ -28,7 +28,7 @@ A toggle that lets the evaluator search the web while it judges, for verdicts th ## Connectors -Connectors let the evaluator call your own tools mid-judgment, the same way it uses web search, so it can check a claim against your database, confirm an ID exists, or verify a rule in an internal service. A connector is a tool you expose over the Model Context Protocol (MCP) and register once on the [MCP Connectors](/docs/falcon-ai/features/mcp-connectors) page; this section assumes you have one registered. +Connectors let the evaluator call your own tools mid-judgment, the same way it uses web search, so it can check a claim against your database, confirm an ID exists, or verify a rule in an internal service. A connector is a tool you expose over the Model Context Protocol (MCP) and register once on the [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) page; this section assumes you have one registered. *Attach a knowledge base, or create one if you have none yet* diff --git a/src/pages/docs/evaluation/guides/running-evaluations.mdx b/src/pages/docs/evaluation/guides/running-evaluations.mdx index 23ed4e06..29f7c5d0 100644 --- a/src/pages/docs/evaluation/guides/running-evaluations.mdx +++ b/src/pages/docs/evaluation/guides/running-evaluations.mdx @@ -30,9 +30,9 @@ In evaluation, **offline** and **online** describe the data, not your connection | Surface | Where you run it | Guide | |---|---|---| -| Dataset and experiments | Offline, over every row of a dataset | [Run experiments](/docs/dataset/features/experiments) | +| Dataset and experiments | Offline, over every row of a dataset | [Run experiments](/docs/dataset/guides/run-an-experiment) | | Traces | Online, on live spans, traces, and sessions | [Set up evals in Observe](/docs/observe/guides/setup-evals) | -| Simulation | Over simulated conversations | [Run a simulation](/docs/simulation/features/run-simulation) | +| Simulation | Over simulated conversations | [Run a simulation](/docs/simulation/guides/run-voice-simulation) | | SDK | Programmatic runs, with local Code Evals that need no API key | [Evaluation SDK](/docs/sdk/evals) | | CI/CD | On every pull request, gating the merge on eval scores | [Evaluate in CI/CD](/docs/evaluation/guides/cicd) | diff --git a/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx b/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx new file mode 100644 index 00000000..16106e0c --- /dev/null +++ b/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx @@ -0,0 +1,72 @@ +--- +title: "MCP Connectors" +description: "Connect external MCP servers so their tools sit beside Falcon's own" +--- + +## An MCP connector links Falcon to an external tool server + +An **MCP connector** is a workspace's connection to an external server that speaks the [Model Context Protocol](https://modelcontextprotocol.io): a workspace-level object with a name, a server address, and everything Falcon has learned about that server since. Once that server is connected, its tools sit beside Falcon's own [platform tools](/docs/falcon-ai/concepts/understanding-falcon-ai) in the same conversation turn, so a single request can read an evaluation and open an issue in your tracker without you switching tools. [Skills](/docs/falcon-ai/concepts/skills) can reach for those same enabled tools too, alongside platform tools, when a workflow calls for them. Falcon's connector panel frames the idea plainly: "Add MCP connectors to give Falcon access to external services like GitHub, Slack, databases, and more." + +Two moments decide what Falcon can actually do with a connector: discovery, when Falcon asks the server what it offers, and enabling, when you choose which of those discovered tools it's allowed to call. For the steps to add a connector and authenticate it, see [Connect an MCP server](/docs/falcon-ai/guides/connect-mcp-server). + + Turn + Enabled --> Turn`} /> + +## Discovered tools are not the same as enabled tools + +This is the distinction that matters most. Discovery is Falcon asking the connected server what it can do: it queries the server's tool list, and a successful discovery turns every one of those tools on, so Falcon can call everything the server offered. Enabling is what happens afterward, and it only ever narrows that starting set down: you choose which discovered tools stay on and disable the rest. Falcon may only call the subset you've left enabled, and that subset can never grow past what discovery found. + + +Only use connectors from developers you trust. Future AGI does not control which tools developers make available and cannot verify that they will work as intended or that they won't change. + + +## The states you'll see + +A connector moves through four stages, but the app tracks them with only three status chips: "Connected", "Pending", or "Inactive". The chip is coarser than the stages, so several of the stages below share the same chip. + +- **Added.** The connector exists with a name and a server address. Falcon hasn't confirmed it can reach or use anything yet, so the chip reads "Pending" +- **Authenticated.** Falcon has verified it can talk to the server, and the chip changes to "Connected". If verification fails instead, the error shows on the connector card in the Customize panel, so you can see it without having to reproduce it +- **Tools discovered.** Falcon keeps the tool list it got back from the server, and the chip still reads "Connected" +- **Tools enabled.** This is where you narrow the default set down to just the tools you want Falcon to use, and the chip still reads "Connected" here too + +A connector can also be turned off outright, at which point the chip reads "Inactive" regardless of what was discovered or enabled underneath it. + +To choose exactly which discovered tools stay enabled, see [Choose connector tools](/docs/falcon-ai/guides/choose-connector-tools). + +## Choosing how a connector authenticates + +Different servers expect different things from a client, so a connector's authentication is a choice, not a fixed requirement, and it's a property of the server you're connecting to, not something Falcon decides for you: + +- **None.** Some servers need no authentication at all +- **API key or bearer token.** Some expect a credential you hold and hand to Falcon directly +- **OAuth.** Some run a full sign-in, where you approve access in the provider's own window rather than typing a secret into Falcon + +Check the server's own documentation, or ask whoever runs it, to find out which one applies. + +## Why it matters + +Vetting the server, and narrowing its enabled tools down to just what you want Falcon to use, is on you. + +## Keep exploring + + + + Add a connector and authenticate it + + + Narrow a connector's enabled tools down to just what you want Falcon to use + + + Build workflows that can call connector tools alongside platform tools + + diff --git a/src/pages/docs/falcon-ai/concepts/skills.mdx b/src/pages/docs/falcon-ai/concepts/skills.mdx new file mode 100644 index 00000000..3ac8b75b --- /dev/null +++ b/src/pages/docs/falcon-ai/concepts/skills.mdx @@ -0,0 +1,72 @@ +--- +title: "Skills" +description: "Reusable instructions that shape how Falcon works, shown with example trigger phrases." +--- + +## What a skill is + +A **skill** is a saved set of instructions plus example phrases for what it's for. Turning one on changes how Falcon approaches the request and which [tools](/docs/falcon-ai/concepts/understanding-falcon-ai) it reaches for, including anything connected through [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors), rather than just answering off a single prompt. A trigger phrase is an example of how someone might ask for this skill, recorded on the skill and shown on its card. + + Contains + Skill -->|"visible everywhere"| BuiltIn["Built-in skill"] + Skill -->|"scoped to one workspace"| Custom["Custom skill"] + subgraph Activation["How a skill becomes active"] + Menu["Picked from Skills menu"] + SlashCmd["Typed as slash command"] + end + Skill --> Activation + Menu --> Active["Active skill for the conversation"] + SlashCmd --> Active`} /> + +## Built-in and custom skills + +Falcon ships with a set of built-in skills. They show up in every workspace, and nobody can edit or delete them: a built-in skill's card in the Customize panel offers only Duplicate, with no edit or delete control. If a built-in skill is close to what you need, duplicate it there and adjust the copy instead of trying to change the original. + +Custom skills belong to a workspace. Anyone who creates one is creating it for the whole workspace, not just themselves, so every member of that workspace sees it and can trigger it. See [Create a skill](/docs/falcon-ai/guides/create-skill) for how to open the editor and build one from scratch. + +## Turning a skill on + +A skill becomes active for a conversation in one of two ways: + +- Pick it from the Skills menu in the Falcon AI header +- Type its slash command at the start of a message + +For example, typing `/debug-traces` at the start of a message runs the Debug Traces skill directly; if it doesn't match a skill, Falcon treats the text as an ordinary message instead. + +## What ships built in + +Every workspace includes these built-in skills: + +- Analyze Costs +- Analyze Trace Errors +- Build a Dataset +- Analyze Cluster +- Compare Models +- Debug Traces +- Fix with Falcon +- Localize Errors +- Optimize Prompts +- Run Evaluations + +## Why it matters + +Skills exist to make the same investigation come out the same shape every time, whether you run it or a teammate does, instead of everyone describing what they want from scratch. A skill is guidance, not a macro: you can add constraints, redirect the investigation, or ask follow-up questions after it's active, and Falcon incorporates them rather than running a fixed script to the end. + +## Keep exploring + + + + The chat interface that skills run inside of + + + Open the skill editor and build a custom skill + + diff --git a/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx b/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx new file mode 100644 index 00000000..901b9bd4 --- /dev/null +++ b/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx @@ -0,0 +1,64 @@ +--- +title: "Understanding Falcon AI" +description: "How a Falcon AI conversation works, and what each turn draws on" +--- + +## A conversation is a thread, a turn is one exchange + +A **conversation** is one thread with Falcon inside a workspace. A **turn** is one thing you ask plus everything Falcon does to answer it: reading your message, deciding what to use, calling tools if it needs to, and writing a response. A conversation is a sequence of turns; everything below describes what happens inside one of them. + +## What a turn draws on + +Every turn draws on four things: + +- **Your message, and any file attached to it.** The question or instruction you typed, plus anything you uploaded alongside it +- **The page you asked from.** Falcon knows what part of the platform you were looking at when you asked, the same way it knows which evaluation you mean when you ask about one without naming it +- **The skill that's active.** A [skill](/docs/falcon-ai/concepts/skills) is a packaged, repeatable workflow, built-in or custom, that Falcon follows with the right tools already loaded. If one is running this turn, it's carried along with the turn, the same as your message or the page +- **The tool set Falcon loads for that turn.** Tools are what let Falcon look things up and make changes on the platform, rather than just describe them; the pool includes any external tools you've connected through [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) alongside the built-in platform ones, and Falcon loads only a working subset of it for each turn + +Falcon picks that subset by reading your request and settling on a working area, roughly forty tools brought forward for the turn out of the hundreds it could load. How specific your ask is decides how wide that area gets: + +- A vague ask keeps the working area broad and the tool set wide +- A specific ask narrows both + +You can also point Falcon at a working area directly, either by opening the [context selector](/docs/falcon-ai/guides/chat-with-falcon-ai) in the chat input and picking one, or by naming the area in your message. + + TURN["Turn"] + TURN --> MSG["Message + attached file"] + TURN --> PAGE["Page you asked from"] + TURN --> SKILL["Active skill"] + TURN --> TOOLS["Tool set for this turn"] + POOL["Full pool of loadable tools"] -->|"~40 brought forward"| TOOLS`} /> + +## Watching a turn run + +While a turn runs, the answer streams in as it's written. Each tool Falcon calls shows up as its own card: it starts with **Running...**, and once the call finishes you can open it to see the **Parameters** it was given, the **Result**, and the **Full output** behind that result. If the turn creates or changes something on the platform, a completion card appears at the end with a link straight to it. + + +As a conversation grows very long, Falcon automatically condenses the earlier part into a summary so the thread stays usable, while the most recent turns are kept exactly as written. Falcon still draws on what was discussed early on, but only as that summary rather than the original wording, so if an exact detail from far back in the thread matters, restate it rather than assume Falcon recalls it precisely. + + +## Cost and pace + +Each turn costs one AI credit. You're also capped at ten messages a minute; go past it and Falcon shows an error in the conversation asking you to wait before sending more. If your organization's AI credit balance runs out, the turn is refused with an error in place of an answer. Credit balance is tracked on [Billing & Pricing](/docs/admin-settings/billing-pricing). + +## Why it matters + +The page you ask from, the skill that's active, and how specific your ask is all shape the tool set Falcon brings forward for a turn, and that shapes the answer you get back. That matters most in a long back-and-forth, where the message limit and credit cost add up turn by turn. + +## Keep exploring + + + + Open the chat, ask questions, upload files, and follow responses + + + Use built-in workflows or create custom slash commands + + + Connect external tools like Linear, Slack, and GitHub + + diff --git a/src/pages/docs/falcon-ai/features/chat.mdx b/src/pages/docs/falcon-ai/features/chat.mdx deleted file mode 100644 index dbc02992..00000000 --- a/src/pages/docs/falcon-ai/features/chat.mdx +++ /dev/null @@ -1,150 +0,0 @@ ---- -title: "Using Falcon AI: Chat, File Upload, and Tool Calls" -description: "Open Falcon AI from any page, ask questions, upload files, and get streaming responses with tool calls and completion cards." ---- - -## About - -Falcon AI runs as a chat interface inside the Future AGI dashboard. It can be opened as a sidebar from any page or as a full-page view for longer conversations. The sidebar stays open while you navigate between pages, so context is never lost. Conversations save automatically and can be resumed later. - -Falcon AI automatically detects what page you are on and uses it as context. Ask "why is this score low?" while viewing an evaluation, and it knows which evaluation you mean. It can also fetch content from URLs you paste, extract text from uploaded files, and stream responses with real-time tool execution. - ---- - -## Opening Falcon AI - - - - Press `Cmd+K` (Mac) or `Ctrl+K` (Windows/Linux) to open a sidebar overlay on the right side of the dashboard. It stays open as you navigate between pages. - - ![Open Falcon AI sidebar](/screenshot/product/falcon/1.png) - - - Click **Falcon AI** in the navigation sidebar to open the full-page view at `/dashboard/falcon-ai`. A conversation history panel on the left lets you search, rename, and delete past conversations. - - ![Open Falcon AI full page](/screenshot/product/falcon/2.png) - - - ---- - -## Asking questions - -Type a question in the input area and press Enter. Falcon AI detects the domain of your request and loads the right tools automatically. - -![Ask a question](/screenshot/product/falcon/3.png) - -To reference a different page than the one you are on, either navigate there first or specify it in your message: - -> "On the evaluations page, which model had the highest faithfulness score?" - ---- - -## Adding context - -Falcon AI detects page context automatically based on the current dashboard page. You can also attach entities manually by clicking **+ Add context** in the input area. Up to 5 entities can be attached at a time. Context chips appear above the input with an X to remove them. - ---- - -## Quick actions and slash commands - -On a new conversation, quick action buttons appear above the input: **Analyze with compass**, **Create custom views**, **Build a dataset**, **Create an evaluation**, **Run simulation for my agent**. They disappear after the first message. - -![Quick action buttons](/screenshot/product/falcon/4.png) - -Type `/` at the start of a message to open the command picker. All active skills, both built-in and custom, appear as slash commands. Select one to run its workflow in the current conversation. - ---- - -## File uploads - -Click the attachment button or drag files into the input area. Falcon AI extracts text content and uses it as context for your question. - -| File type | What happens | -|-----------|-------------| -| **PDF** | Text is extracted from all pages | -| **Excel / CSV** | Spreadsheet data is converted to text | -| **Word (.docx)** | Document text is extracted | -| **Images (PNG, JPG)** | Image is encoded and sent to the model for visual understanding | -| **Text / Markdown / JSON** | Content is included directly | - - - Maximum file size is 10 MB per upload. - - ---- - -## URL fetching - -Paste a URL in your message and Falcon AI automatically fetches its content. This works with: - -- **Web pages**: HTML is cleaned and converted to text -- **JSON APIs**: Response is formatted as a code block -- **GitHub raw files**: Content is included as a code block -- **Jupyter notebooks**: Code and markdown cells are extracted - -Up to 3 URLs are fetched in parallel, with a maximum of 50 KB of content per URL. - ---- - -## Following responses - -Responses stream token by token with Markdown formatting. When Falcon AI calls platform tools, collapsible cards show each step: - -- A **spinner** while the tool is running -- A **checkmark** when it completes -- A **warning icon** if it errors - -![Response with tool calls](/screenshot/product/falcon/5.png) - -When a tool creates or modifies a platform entity, a **completion card** appears with a direct link to the result. - -Falcon AI can call multiple tools in parallel when they are independent, and chains them sequentially when one depends on another. A single turn can run up to 50 tool-call iterations. - ---- - -## Stopping a response - -Click the **Stop** button in the input area while Falcon AI is streaming. The current tool execution is cancelled and the response ends at whatever has been generated so far. - ---- - -## Conversation history - - - - Click the **history** (clock) button at the top of the sidebar to see past conversations. - - - The left panel shows all conversations with search. Right-click a conversation to rename or delete it. - - - -![Conversation history](/screenshot/product/falcon/6.png) - -Conversations persist across sessions. If you close the browser and come back, your full history is available. - ---- - -## Reconnection - -If your connection drops mid-response, Falcon AI automatically replays missed events when you reconnect so you see the complete response. - ---- - -## Rate limits - -Falcon AI allows 10 messages per 60 seconds per user. If you hit the limit, wait briefly before sending the next message. - ---- - -## Next Steps - - - - Use built-in workflows or create custom slash commands. - - - Connect external tools like Linear, Slack, and GitHub. - - diff --git a/src/pages/docs/falcon-ai/features/mcp-connectors.mdx b/src/pages/docs/falcon-ai/features/mcp-connectors.mdx deleted file mode 100644 index c8ca3ee6..00000000 --- a/src/pages/docs/falcon-ai/features/mcp-connectors.mdx +++ /dev/null @@ -1,121 +0,0 @@ ---- -title: "MCP Connectors: Connect External Tools to Falcon AI" -description: "Connect external MCP servers to Falcon AI to use tools from services like Linear, Slack, GitHub, Sentry, and custom APIs." ---- - -## About - -Falcon AI comes with built-in tools for the Future AGI platform, but many workflows involve external services: project trackers, communication tools, monitoring systems, and internal APIs. MCP Connectors extend Falcon AI by connecting it to any server that implements the [Model Context Protocol](https://modelcontextprotocol.io). Once connected, Falcon AI discovers the server's tools and can call them during conversations alongside built-in platform tools. - -This means tasks like "create a Linear ticket for this failing evaluation" or "post this cost report to Slack" happen inside Falcon AI without switching tools. - ---- - -## Examples - -- **Project management**: Connect Linear, Jira, or Asana to create and update issues from evaluation or trace analysis. -- **Communication**: Connect Slack or email to share reports and alerts directly. -- **Monitoring**: Connect Sentry or PagerDuty to pull error context into debugging conversations. -- **Internal APIs**: Connect custom MCP servers that expose your organization's tools. - ---- - -## Adding a connector - - - - Open Falcon AI settings and go to the **Connectors** section. Click **Add Connector**. - - ![Add connector form](/screenshot/product/falcon/9.png) - - - - Fill in the connector fields: - - | Field | Required | Description | - |-------|----------|-------------| - | **Name** | Yes | Display name for the connector (e.g., "Linear", "Sentry") | - | **Server URL** | Yes | The MCP server endpoint URL | - | **Transport** | Yes | `streamable_http` (default, recommended) or `sse` (Server-Sent Events) | - | **Auth type** | Yes | How to authenticate with the server (see below) | - - - - Choose the authentication method that your MCP server requires: - - | Auth type | Fields | Description | - |-----------|--------|-------------| - | **None** | -- | No authentication required | - | **API Key** | Header name, Header value | Sends a custom header with each request (e.g., `X-API-Key: your-key`) | - | **Bearer Token** | Token | Sends `Authorization: Bearer ` with each request | - | **OAuth 2.1** | Client ID, Client secret, Auth URL, Token URL, Scopes | Full OAuth flow with automatic token refresh | - - - For OAuth connectors, Falcon AI handles the entire authorization flow. After saving the connector, click **Authenticate** to open the OAuth consent screen. Tokens are stored securely and refreshed automatically when they expire. - - - - - Click **Test Connection** to verify that Falcon AI can reach the MCP server and authenticate successfully. If the test fails, the error message is displayed so you can debug the configuration. - - - - Click **Discover Tools** to query the MCP server for its available tools. Falcon AI reads the server's tool schema and displays the list with names, descriptions, and parameter definitions. - - The discovery result is cached. Re-run discovery if the server adds new tools. - - - - Not all discovered tools need to be active. Select which tools Falcon AI should have access to from the discovered list. Only enabled tools appear in conversations. - - This is useful when a server exposes many tools but you only need a subset, keeping Falcon AI's tool set focused and reducing context window usage. - - - ---- - -## Using connector tools in chat - -Once enabled, connector tools appear in Falcon AI conversations alongside built-in platform tools. Falcon AI decides when to use them based on your request. Tool names from connectors are prefixed with the connector name to avoid collisions (e.g., `linear_create_issue`). - -**Examples:** - -> "Create a Linear ticket for the faithfulness regression we found in the last evaluation run." - -> "Post a summary of today's error spikes to the #ml-alerts Slack channel." - -> "Check Sentry for any new issues related to the summarization service." - ---- - -## Transport options - -MCP Connectors support two transport protocols: - -| Transport | How it works | When to use | -|-----------|-------------|-------------| -| **Streamable HTTP** | Standard HTTP POST requests with JSON-RPC 2.0 payloads | Default. Works with most MCP servers. | -| **SSE** (Server-Sent Events) | Long-lived HTTP connection with server-pushed events | Use when the server requires SSE transport or for streaming tool results. | - -Falcon AI automatically tries multiple endpoint paths (with and without `/mcp` suffix) to find the correct one for your server. - ---- - -## Managing connectors - -- **Edit**: Update any connector field from the Connectors settings page. Re-test and re-discover after changes. -- **Delete**: Remove a connector and all its cached tool schemas. Tools from deleted connectors are immediately unavailable in conversations. -- **Re-authenticate**: For OAuth connectors, click **Authenticate** again if the authorization has been revoked or if scopes need to change. - ---- - -## Next Steps - - - - Learn the basics of the chat interface. - - - Create custom workflows that can use connector tools. - - diff --git a/src/pages/docs/falcon-ai/features/skills.mdx b/src/pages/docs/falcon-ai/features/skills.mdx deleted file mode 100644 index 6c708f5b..00000000 --- a/src/pages/docs/falcon-ai/features/skills.mdx +++ /dev/null @@ -1,133 +0,0 @@ ---- -title: "Skill Builder: Custom Slash Commands for Falcon AI" -description: "Use built-in skills for common workflows or create custom slash commands that package multi-step instructions for your team." ---- - -## About - -The same analysis gets repeated across conversations and team members: checking regressions, generating cost reports, investigating error spikes. Skills package these workflows into reusable slash commands. Type `/` in the chat input, select a skill, and Falcon AI follows the packaged instructions with the right tools loaded. - -Falcon AI ships with six built-in skills for common workflows. You can also create custom skills scoped to your workspace. - ---- - -## Built-in skills - -These skills are available in every workspace and cannot be edited or deleted. - -### Build a Dataset - -Guides you through creating a dataset step by step. Falcon AI asks for a name, helps define columns, and walks you through adding rows, whether manually, from a file, or with synthetic generation. - -**Example**: `/build-a-dataset` → "I need a dataset of customer support tickets with columns for query, response, and sentiment." - ---- - -### Debug Traces - -Investigates traces with quantified analysis rather than vague summaries. Falcon AI reports specific error counts, latency percentile distributions, and recurring patterns across spans. - -**Example**: `/debug-traces` → "Why are we seeing timeout errors on the summarization endpoint?" - ---- - -### Compare Models - -Runs tradeoff analysis across multiple model variants. Falcon AI evaluates cost, quality, and latency side by side and highlights which model wins on each dimension. - -**Example**: `/compare-models` → "Compare GPT-4o and Claude Sonnet on our QA dataset for faithfulness and cost." - ---- - -### Run Evaluations - -Helps select the right evaluation template for your use case and explains results in context. Falcon AI picks templates based on your data type and walks through the scores. - -**Example**: `/run-evaluations` → "Evaluate the customer-support dataset for hallucination and toxicity." - ---- - -### Optimize Prompts - -Analyzes prompt versions and produces specific, actionable suggestions. Instead of generic advice, Falcon AI compares outputs across versions and points to what changed and why. - -**Example**: `/optimize-prompts` → "My summarization prompt is producing outputs that are too long. Help me tighten it." - ---- - -### Analyze Costs - -Produces cost breakdowns with exact dollar amounts and percentage savings opportunities. Falcon AI segments by model, project, and time period. - -**Example**: `/analyze-costs` → "Show me a cost breakdown for the last 30 days by model." - ---- - -## Custom skills - -Create skills specific to your team's workflows. Custom skills are scoped to the workspace and available to all workspace members. - -### Creating a skill - - - - In the Falcon AI chat input, click the **customize** button to open the skill picker. Click **Create Skill** to open the editor. - - ![Open skill editor](/screenshot/product/falcon/7.png) - - - - Fill in the skill fields: - - ![Skill editor fields](/screenshot/product/falcon/8.png) - - | Field | Required | Description | - |-------|----------|-------------| - | **Name** | Yes | Display name shown in the command picker (e.g., "Weekly Cost Review") | - | **Description** | Yes | Short description shown below the name in the command picker | - | **Icon** | No | Icon displayed next to the skill name | - | **Instructions** | Yes | The prompt that Falcon AI follows when the skill is triggered. Write these as direct instructions for the AI. | - | **Trigger phrases** | No | Phrases that activate the skill automatically when typed in a message. Press Enter after each phrase. | - - - - Skill instructions work best when they are specific and structured. Include: - - - **What to do first**: Which tools to call and in what order - - **How to present results**: Tables, comparisons, summaries - - **What to ask the user**: If the skill needs input, tell Falcon AI to ask for it - - **Example instruction for a weekly review skill:** - - ``` - 1. Get evaluation scores for all datasets in this workspace from the last 7 days. - 2. Compare each dataset's scores to the previous 7-day period. - 3. Flag any metric that dropped by more than 5%. - 4. Present results as a table with columns: Dataset, Metric, This Week, Last Week, Change. - 5. If any regressions are found, suggest which traces to investigate. - ``` - - - - Type `/` in the chat input to open the command picker. Select your skill to run it. You can also type a message after selecting the skill to provide additional context. - - Skills also trigger automatically when a message matches one of the configured trigger phrases. - - - -### Editing and deleting skills - -Open the skill picker, click an existing custom skill to open the editor. Update any field and save, or click **Delete** to remove it. Built-in skills cannot be edited or deleted. - ---- - -## Next Steps - - - - Learn the basics of the chat interface. - - - Extend Falcon AI with tools from external services. - - diff --git a/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx b/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx new file mode 100644 index 00000000..acf1bb8d --- /dev/null +++ b/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx @@ -0,0 +1,120 @@ +--- +title: "Chat with Falcon AI" +description: "Open Falcon AI, ask a question, and attach a file." +--- + +Falcon AI is the AI copilot built into the Future AGI dashboard, reachable from any page you're on. Ask it about your workspace and it'll look up what it needs to answer. + +## Open Falcon AI + +Click **Falcon AI** in the dashboard's navigation sidebar. It opens the full-page view at `/dashboard/falcon-ai`, with your conversations listed down the left and the chat itself in the center. + +The dashboard navigation sidebar with the Falcon AI item highlighted +*The Falcon AI item in the dashboard nav opens the full-page view* + +The Falcon AI full page with a conversation list on the left and the chat in the center +*The full page adds a conversation list; the side panel doesn't* + +From anywhere else, press `⌘K` (Mac) or `Ctrl+K` (Windows/Linux), or click the floating button in the bottom-right corner of the page, tooltipped **Falcon AI (⌘K)**. Either opens a panel that slides in from the side, so you can keep the page underneath in view while you ask something. + +A page with the floating Falcon AI button in the bottom-right corner and its Falcon AI (Cmd+K) tooltip visible +*The floating button and ⌘K both open the same side panel* + +Opening the full page while the side panel is open closes the panel. + +Every conversation is saved. See [Manage conversations](/docs/falcon-ai/guides/manage-conversations) for how to find, rename, or delete one. + +## Start a new conversation + +A fresh conversation opens with "How can I help?" and, under it, "Ask about your data, evaluations, experiments, traces, and more." + +Below that sit five quick-action chips: + +- Analyse my error feed +- Create an Imagine view +- Build a dataset +- Create an evaluation +- Run simulation for my agent + +Click one to prefill the input, then edit it before sending, or type your own question instead. + +Falcon AI's empty state with the How can I help heading and five quick-action chips above the input +*The five chips disappear once the conversation has its first message* + +## Ask a question and send it + +Type your question into the input at the bottom and click **Send**, or press Enter to send and Shift+Enter to add a newline. Send stays inactive until there's something to send, either typed text or [an attached file](#attach-a-file), so an empty input can't be submitted by mistake. + +The Falcon AI input with a question typed in and the Send button active +*Send lights up once there's text or a file attached* + +## Point it at the right context + +Falcon AI defaults to Auto, reading the page you're on to work out what you're asking about. When Auto picks the wrong area, open the context selector and set it directly, before you send your question. The seven options are: + +- Auto +- Datasets +- Evaluations +- Tracing +- Experiments +- Agents +- Prompts + +The context selector open, listing Auto, Datasets, Evaluations, Tracing, Experiments, Agents, and Prompts +*Switch off Auto when you're asking about something other than the page you're on* + +## Attach a file + +Falcon AI reads whatever you attach to help answer your question, alongside anything you type. Click the **+** button in the input area, labeled **More**, and choose **Attach files**, or drag a file onto the input and drop it. Falcon AI accepts: + +- Plain text, CSV, HTML, Markdown, and JSON +- PDF +- Excel (`.xlsx`) and Word (`.docx`) +- PNG, JPEG, GIF, and WebP images + +Each file tops out at 10 MB, and the **Attach files** menu option reads "Uploading..." while the file goes up. + +The plus menu open above the input with an Attach files option, and a file mid-upload showing Uploading +*The plus button is labeled More; Attach files is the option that opens the file picker* + + +A file over 10 MB is dropped with no message. If an attachment you tried to add never shows up above the input, that's why, so check its size and try a smaller file. + + +## Follow the answer as it streams + +The response streams in as it's generated, and the input stays disabled until it finishes. + +If Falcon AI needs a tool along the way, a card appears in the conversation, starting at "Running...". Open it to see **Parameters**, what was sent, and **Result**, what came back, with **Full output** available when there's more to show. [Understanding Falcon AI](/docs/falcon-ai/concepts/understanding-falcon-ai) covers how it decides which tool to reach for. + +An open tool call card showing its Parameters and Result sections, with a Full output option +*A tool card opens to Parameters and Result; Full output expands longer results* + +If a turn is taking too long, click **Stop**. It cuts the response off where it is and keeps whatever has been written so far, rather than discarding it. + +Falcon AI streaming a response with the Stop button showing in place of Send +*Stop ends the turn but leaves the partial answer in the conversation* + +## Copy or rate an answer + +Under each answer sit three actions: **Copy**, which copies the response to your clipboard, and **Good response** or **Bad response**, which record whether it was useful. + +An assistant message with Copy, Good response, and Bad response actions beneath it +*Copy, Good response, and Bad response sit under every answer* + +## Pace and cost + +Falcon AI is limited to ten messages a minute, and that window counts chat messages, response ratings, and Stop clicks together, so a burst of rapid follow-ups will hit that ceiling. Past it, the next send fails inline. + +Each turn costs one AI credit. + +## Dive deeper + + + + Use built-in workflows or create custom slash commands + + + Connect external tools like Linear, Slack, and GitHub + + diff --git a/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx b/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx new file mode 100644 index 00000000..fcb7f171 --- /dev/null +++ b/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx @@ -0,0 +1,43 @@ +--- +title: "Choose connector tools" +description: "Choose which of a connector's discovered tools stay allowed in Customize." +--- + +A connector that's finished discovery hands Falcon AI every tool it found, already allowed. Narrowing that down to what you actually want Falcon AI calling happens in a separate step. This guide covers the Customize panel, where you allow or deny a [connector's](/docs/falcon-ai/concepts/mcp-connectors) tools for the whole workspace, and the chat input's Connectors submenu, which is a shortcut for managing connectors rather than a separate control. + +This guide assumes you've already added a connector, authenticated it, and run discovery on it. If you haven't, see [Connect an MCP server](/docs/falcon-ai/guides/connect-mcp-server) first. + +## Open a connector's tool permissions + +From the [Falcon AI full page](/docs/falcon-ai/guides/chat-with-falcon-ai), open **Customize** in the left rail, the panel where you manage skills and connectors. Pick **Connectors**, then select the connector you want to configure. + +Its **Tool permissions** section is where access is actually decided, splitting what the server offers into two groups: **Interactive tools**, which take an action through the connected server such as creating, updating, or sending something, and **Read-only tools**, which only fetch or look up information. + +## Allow or deny individual tools + +To stop Falcon AI from calling a single tool, flip its toggle from **Allowed** to **Denied**. This doesn't touch the rest of its group, so you can deny the odd tool you don't want called while leaving the rest of Interactive tools or Read-only tools allowed. There's no group-level deny-all chip, so denying a whole group means flipping its tools one at a time. + +If you want a group back to fully allowed after denying part of it, click that group's **Always allow** chip to allow every tool in it again in one move. + + +Falcon AI can only call a tool that's both discovered and allowed. If the connected server stops offering a tool you'd allowed, the change is rejected with an "Unknown tools" error until you re-run [Discover Tools](/docs/falcon-ai/guides/connect-mcp-server). + + +## Before a connector has discovered anything + +Before a connector has discovered any tools, the Tool permissions section isn't there yet. In its place sits a single note: "No tools discovered yet. Tools will appear after the connector successfully connects." That's expected for a connector you've just added, or one whose discovery just hasn't finished yet. + +## Manage connectors from the chat input + +The chat input's **+** menu has a **Connectors** submenu listing every connector configured for the workspace along with its connected state, and a **Manage connectors** link that opens the connector settings page. If none are configured yet, it reads "No connectors configured" instead. + +## Dive deeper + + + + Put newly enabled tools to work, then organize the conversations that use them + + + Build a skill that calls these same allowed tools + + diff --git a/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx b/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx new file mode 100644 index 00000000..72781870 --- /dev/null +++ b/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx @@ -0,0 +1,70 @@ +--- +title: "Connect an MCP server" +description: "Give Falcon AI access to a new MCP server, whatever transport and authentication it expects." +--- + +This guide adds an MCP server to Falcon AI as a connector, so its tools become available for Falcon AI to call in chat once you turn them on (see [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) for what a connector is). You name it, point it at the server, choose how Falcon AI talks to it and signs in, then check that the connection actually works before you rely on it in chat. + +Before you start, make sure you have access to Workspace Settings in your workspace, the MCP server's URL, and any credential the server requires, such as an API key or bearer token. + +## Open the Connectors page + +Connectors live on the **Falcon AI Connectors** page, under **Workspace Settings** in Settings. Open Settings, then click **Falcon AI Connectors** in the sidebar to go straight there. Click **Add Connector** to open the form. + +You can also get here from inside a conversation: **Manage connectors** in the chat input's **+** menu takes you to this same page, or use **Add custom connector** in Customize to add one without leaving the chat. + +## Name it and point it at the server + +Give the connector a **Name** and its **Server URL**, the address of the MCP server you're connecting to. Both fields are required to save. + + +Connector names must be unique within the workspace. If another connector already uses the name you typed, saving is rejected until you pick a different one. + + +## Choose the transport + +Set **Transport** to how Falcon AI should talk to the server. **Streamable HTTP** is the normal choice for a current MCP server. Pick **SSE (Legacy)** instead if you're connecting to an older server that still expects that transport. + +## Choose the authentication + +Set **Authentication** to whatever the server expects. The form reveals the fields that method needs: + +- **None**: skips authentication entirely +- **API Key**: adds **Header Name** and **Header Value** fields +- **Bearer Token**: adds a single **Bearer Token** field +- **OAuth 2.0**: no extra fields, you sign in after saving + +Fill in whatever fields appear, then click **Create** to save the connector. + +## Sign in to an OAuth connector + +If you set **Authentication** to **OAuth 2.0**, saving the connector doesn't sign you in. Authentication happens as a separate step. Once saved, select the connector in the list on the left to open its detail pane, where you'll find **Re-authenticate**. That's the button for both your first sign-in and any later ones: click **Re-authenticate** (it reads **Authenticating...** while it runs), then approve access in the provider's own sign-in window. The window closes itself once you approve, handing control back to Falcon AI, and the result lands as **Authentication updated.** or **Authentication failed.** If it fails, confirm you approved access in the provider's window and try again. + +## Test the connection and discover its tools + +With the connector saved and, for OAuth, signed in, select the new connector in the list on the left to open its detail pane. This is where **Test Connection**, **Discover Tools**, **Re-authenticate**, **Edit**, and **Delete** all live. Click **Test Connection** to confirm Falcon AI can reach it. While it runs, the button reads **Testing...**; it resolves to **Connection test succeeded.** or **Connection test failed.** + +Once the connection is good, click **Discover Tools** (it reads **Discovering...** while it runs) to have the server report what it offers. A successful pass shows something like **Discovered 4 tools.** A failed one shows **Tool discovery failed.** + +## Narrow down the tools + +A successful discovery turns every tool it found on, so Falcon AI can already call all of them. The next step is cutting that set back to the ones you actually want, covered in [Choose connector tools](/docs/falcon-ai/guides/choose-connector-tools). + +## Edit or delete a connector + +Select the connector in the list on the left to open its detail pane, then click **Edit** to change its name, URL, transport, or authentication, and click **Save** to save your changes. Re-run **Test Connection** and **Discover Tools** afterward to make sure they still match. Click **Delete** to remove a connector you no longer need. + + +Delete removes the connector immediately, there's no confirmation prompt. + + +## Dive deeper + + + + Turn on the discovered tools Falcon AI is allowed to call + + + Put connector tools to work once they're enabled + + diff --git a/src/pages/docs/falcon-ai/guides/create-skill.mdx b/src/pages/docs/falcon-ai/guides/create-skill.mdx new file mode 100644 index 00000000..cd1f6052 --- /dev/null +++ b/src/pages/docs/falcon-ai/guides/create-skill.mdx @@ -0,0 +1,70 @@ +--- +title: "Create a skill" +description: "Build, save, and run a custom Falcon AI skill" +--- + +A skill packages a multi-step Falcon AI workflow into a slash command that anyone in your workspace can reuse, so instead of re-explaining a recurring analysis in every conversation, you type one command and Falcon AI runs the workflow behind it. This guide walks through building your own, from opening the editor to running what you save. + +## Open the skill editor + +Skills are built from Falcon AI's full page view; see [Chat with Falcon AI](/docs/falcon-ai/guides/chat-with-falcon-ai) if you haven't opened it yet. For what a skill is, see [Skills](/docs/falcon-ai/concepts/skills). + +Click **Customize** in the left rail of the Falcon AI full page to open the Customize panel, then pick **Skills** from its nav to see the skills available in your workspace, with a search field for finding one by name. Click **Create Skill** at the bottom of that list to open the editor for a brand new skill. + + +Already mid-conversation? The Skills menu in the chat header also has a **Create Skill** entry, so you can build a skill without leaving the chat. It's a separate dialog rather than the Customize editor: it adds an **Icon** field, which takes an MDI icon name such as `mdi:bug` and sets the icon shown against the skill in the Skills list and picker, and its button reads **Create** on a new skill instead of **Save**. + + +## Fill in the skill + +The Customize editor asks for four fields. + +Name is required and limited to 100 characters, and it's how the skill is labelled in the Skills list, so keep it short. + +Description is optional. It shows in the skill picker, so it's worth a line explaining what the skill does. + +Instructions is required. It's the prompt Falcon AI follows once the skill runs. Describe the workflow and reasoning, not a rigid script of commands to execute: + +- Workflow: "Compare this week's scores against last week's, and flag any metric that dropped by more than 5%." +- Script: "Call the scores endpoint for this week, call it again for last week, then subtract." + +Trigger phrases need at least one entry before you can save; each one is an example of how someone might ask for this skill. Type a phrase and press Enter to add it to the list. + +## Example: a weekly eval regression review skill + +Here's a complete skill that compares [Evaluations](/docs/evaluation) scores week over week, filled in field by field: + +- **Name:** Weekly Eval Regression Review +- **Description:** Comparing this week's evaluation scores against last week's and flagging anything that dropped +- **Instructions:** + + ``` + Pull the evaluation scores for every dataset in this workspace from the last 7 days, then pull the same metrics for the 7 days before that as a baseline. Compare the two periods per dataset and per metric, and treat any metric that dropped by more than 5% as a regression worth flagging. Present the findings as a table with columns for Dataset, Metric, This Week, Last Week, and Change, with the biggest drops listed first. If nothing regressed, say so plainly instead of returning an empty table. + ``` + +- **Trigger phrases:** "weekly eval review" and "check eval regressions" + +## Save and run it + +Click **Save** at the bottom of the editor to store the skill; it reads Saving... while it works. + +The skill you save gets its own slash command. Run it in any conversation by typing `/` in the chat input and picking the skill from the list that appears. + + +In the Customize editor, leave Name blank and it says "Name is required". Leave Instructions blank and it says "Instructions are required". Skip trigger phrases and it says "At least one trigger phrase is required". A name already in use by another skill is rejected. + + +## Edit, delete, and duplicate skills + +Custom skills can be edited and deleted. Open one again from the Skills list in Customize, change whatever field needs it, and save to update it. **Delete** removes it from the workspace entirely, and that option is available on any custom skill; skills that shipped with Falcon AI don't offer it. If a shipped skill covers most of what you need but not quite all of it, open it and click **Duplicate** instead of building from scratch. That gives you your own editable copy, ready to rename and adjust into the variant you actually want. + +## Dive deeper + + + + Add a connector so Falcon AI can reach outside tools + + + Find, rename, and delete past conversations + + diff --git a/src/pages/docs/falcon-ai/guides/manage-conversations.mdx b/src/pages/docs/falcon-ai/guides/manage-conversations.mdx new file mode 100644 index 00000000..f0050a51 --- /dev/null +++ b/src/pages/docs/falcon-ai/guides/manage-conversations.mdx @@ -0,0 +1,44 @@ +--- +title: "Manage conversations" +description: "Look up past chats from either surface, then rename or delete them from the full-page view." +--- + +Every conversation you have with Falcon AI is saved. You can find one from either surface Falcon AI runs on, its [full-page view or its side panel](/docs/falcon-ai/guides/chat-with-falcon-ai), and rename or delete it from the full page. + +## Find a conversation + +Where the list of past conversations shows up depends on which surface you're using. On the full-page view, it's the left rail, with a **Search chats...** field above it. In the side panel, click **Chat history** in the header to open the same list in a popover, with its own **Search chats...** field at the top. Either field filters the list down to the conversation you're after as you type. + +## Start a new chat + +Click **New chat** to start over. It's available in the left rail on the full page and in the side panel's header. + +## Rename or delete a conversation + +Renaming and deleting are only available on the full-page view. If you're in the side panel, open the full page first. + +Hover over a conversation in the left rail to reveal its row menu, a **⋯** icon at the row's right edge. + +Click it and choose **Rename** to type the new name in the **Rename conversation** prompt, or choose **Delete** to remove the conversation. + +A conversation row in the left rail with its dots menu open, showing Rename and Delete +*Rename and Delete share the same row menu* + + +Delete has no confirmation step: the conversation disappears from your list as soon as you click it, so make sure it's the right row before you choose it. Deleting the conversation you currently have open drops you back to a new, empty chat. + + +## Titles and dropped connections + +Falcon AI generates a conversation's title at the end of your first turn, so a thread still labeled "New conversation" hasn't gotten a reply yet. If your connection drops while Falcon AI is answering, it replays whatever you missed on its own once the connection comes back. + +## Dive deeper + + + + Use built-in workflows or create custom slash commands + + + Connect external tools like Linear, Slack, and GitHub + + diff --git a/src/pages/docs/quickstart/setup-mcp-server.mdx b/src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx similarity index 81% rename from src/pages/docs/quickstart/setup-mcp-server.mdx rename to src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx index fed183e7..1293a5c5 100644 --- a/src/pages/docs/quickstart/setup-mcp-server.mdx +++ b/src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx @@ -1,11 +1,11 @@ --- -title: "Setup MCP Server: Future AGI with Claude, Cursor, or VS Code" -description: "Set up the Future AGI MCP Server to interact with the platform via natural language from Claude, Cursor, or VS Code using Model Context Protocol." +title: "Use the MCP Server in your IDE" +description: "Drive Future AGI from Claude, Cursor, or VS Code instead of the dashboard" --- -import MCPIDETabs from '../../../components/MCPIDETabs.astro'; +import MCPIDETabs from '../../../../components/MCPIDETabs.astro'; -## About +## What the MCP Server is The **Future AGI MCP Server** lets you interact with the entire Future AGI platform through natural language, directly from your AI coding environment. Instead of switching between the dashboard and your editor, you can run evaluations, upload datasets, generate synthetic data, and apply protection rules just by describing what you want in tools like Claude, Cursor, or VS Code. It's built on the [Model Context Protocol](https://modelcontextprotocol.io/introduction), a standard that connects AI models to external tools and services. @@ -44,7 +44,7 @@ https://api.futureagi.com/mcp With **Future AGI's MCP Server**, you can use natural language to: - **Run automatic evaluations**: evaluate batch and single inputs on various [evaluation](/docs/cookbook/quickstart/first-eval) metrics, both on local datapoints and large datasets -- **Prototype and observe your agents**: add [observability](/docs/observe/quickstart), evaluations while [prototyping](/docs/prototype) and deploying agents into production +- **Build and observe your agents**: add [observability](/docs/observe/quickstart), and [evaluations](/docs/evaluation) while you build and deploy agents into production - **Manage datasets**: upload, evaluate, download [datasets](/docs/dataset) and find insights - **Add protection rules**: apply toxicity detection, prompt injection protection, and other guardrails automatically - **Generate synthetic data**: describe your dataset and objective to generate synthetic data diff --git a/src/pages/docs/falcon-ai/index.mdx b/src/pages/docs/falcon-ai/index.mdx index 399afdac..dc1b3d50 100644 --- a/src/pages/docs/falcon-ai/index.mdx +++ b/src/pages/docs/falcon-ai/index.mdx @@ -1,81 +1,52 @@ --- -title: "Falcon AI: AI Copilot for the Future AGI Dashboard" -description: "An AI copilot embedded in the Future AGI dashboard that handles platform tasks, runs analysis, and answers questions through natural language." +title: "Overview" +description: "Where Falcon AI lives in the dashboard and how it differs from the MCP Server." --- -## About +## What is Falcon AI? -Falcon AI is a copilot built into the Future AGI dashboard. It has access to over 300 platform tools and can work across datasets, evaluations, traces, experiments, prompts, and admin settings through natural language. It knows what page you are on, what entity you are looking at, and acts on that context directly. +**Falcon AI** is an agent built into the Future AGI dashboard. Describe what you want in plain language and it works the platform for you, reading the page you're on, calling the right tools, and acting directly instead of pointing you to where to click. It works the platform with a built-in set of tools, which [skills](/docs/falcon-ai/concepts/skills) and [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) shape and extend. -{/* TODO: Add hero screenshot showing Falcon AI sidebar with a multi-step conversation */} +## Where it shows up -You describe a task, Falcon AI executes it. You ask a follow-up, it goes deeper. A single conversation can span multiple features: start from an evaluation regression, drill into the failing traces, inspect the dataset behind them, and compare against a different model. +It's available on all plans. Falcon AI runs on two surfaces: a full page, which carries the conversation list and lets you rename or delete conversations, and a side panel, which keeps the page underneath in view. ---- - -## What Falcon AI can do - -**Analyze.** Ask questions about your data and get quantified answers, not summaries. - -> "Which eval metrics dropped this week compared to last week?" -> "What's the p95 latency for the summarization endpoint?" -> "Show me a cost breakdown by model for the last 30 days." - -**Create.** Build platform entities without leaving the chat. +- **Full page:** open the **Falcon AI** entry at the top of the left navigation, or land there automatically right after logging in +- **Side panel:** from any other page in the dashboard, click the button in its bottom-right corner, or press `⌘K` (Mac) or `Ctrl+K` (Windows/Linux) -> "Create a dataset called qa-golden with columns for query, expected_answer, and context." -> "Run faithfulness and hallucination evals on the customer-support dataset." -> "Set up an A/B experiment comparing GPT-4o and Claude Sonnet on the QA dataset." +## What you can use it for -**Debug.** Search traces, drill into spans, correlate across features. +Falcon AI gets used for three kinds of work: -> "Show me traces with timeout errors from the last 24 hours." -> "Find traces where the model hallucinated and show me what context was retrieved." +- **Analyze** what's already on the platform, for example "which evaluation scores dropped this week," and get an answer instead of a dashboard to go dig through yourself +- **Create** things, like "build a dataset from these production [traces](/docs/tracing/concepts/traces)" +- **Debug**, for example "why did this trace fail," by pointing Falcon AI at the record and asking what went wrong -**Chain.** Work across features in a single conversation. Each follow-up builds on the previous result. +## Falcon AI vs the MCP Server -> "The faithfulness score on run 12 dropped. Show me the failing traces, then compare the prompts used in run 11 vs run 12." +Falcon AI lives in the dashboard and knows the page you're on, so it can act on the record you already have open. The [MCP Server](/docs/falcon-ai/guides/use-the-mcp-server) lives in your IDE and knows your code instead, so it fits work that happens in a codebase rather than on the platform. Reach for Falcon AI when you're working in the dashboard, and the MCP Server when you're working in your editor. ---- - -## Key capabilities - -| Capability | Details | -|------------|---------| -| **Page-aware context** | Automatically detects the current dashboard page and entity. Ask "why is this score low?" and it knows which evaluation you mean. | -| **300+ tools** | Covers datasets, evaluations, traces, experiments, prompts, agents, simulations, cost analytics, and admin settings. | -| **Multi-step execution** | Chains up to 50 tool calls per turn. Runs independent calls in parallel, sequential calls in order. | -| **Skills** | Pre-built and custom slash commands that package multi-step workflows. Type `/` to access them. | -| **File and URL input** | Upload PDFs, CSVs, images, or paste URLs. Falcon AI extracts content and uses it as context. | -| **MCP Connectors** | Connect external services (Linear, Slack, GitHub, Sentry) so actions like "create a ticket for this regression" work in chat. | - ---- +## Keep exploring -## Falcon AI vs MCP Server +Start with the mental model, then explore skills, connectors, and the guides as you need them. -Future AGI has two AI interfaces for different contexts: - -| | Falcon AI | MCP Server | -|--|-----------|------------| -| **Where** | Inside the dashboard (browser) | Inside your IDE (Cursor, Claude Code, VS Code) | -| **Who** | Platform users browsing the dashboard | Developers writing code | -| **Context** | Knows what page is open, what entity is being viewed | Knows the codebase and files being edited | -| **Output** | Rich rendering: charts, tables, completion cards | Text-only responses | - -Both share the same tool layer. - ---- - -## Next Steps - - - - Open the chat, ask questions, upload files, and follow responses. + + + The mental model behind a conversation, what a turn draws on and what it costs + + + Open Falcon AI from the nav, a shortcut, or the full page, then ask a question + + + Find conversations from either surface, then rename or delete them from the full page + + + The two kinds of skills, how each gets switched on, and what ships by default - - Use built-in workflows or create custom slash commands. + + What changes once an external tool server is wired into a conversation - - Connect external tools like Linear, Slack, and GitHub. + + Add a server as a connector and confirm Falcon AI can reach its tools diff --git a/src/pages/docs/faq.mdx b/src/pages/docs/faq.mdx index e1ccef7b..34a10e93 100644 --- a/src/pages/docs/faq.mdx +++ b/src/pages/docs/faq.mdx @@ -45,15 +45,15 @@ Use retrieval-specific evals like context_adherence, chunk_attribution, and reca **How can I import data?** -Data can be added manually, via file upload, SDK, or imported from Hugging Face. See [Create New Dataset](/docs/dataset/features/create). +Data can be added manually, via file upload, SDK, or imported from Hugging Face. See [Create New Dataset](/docs/dataset/guides/create-a-dataset). **What are dynamic columns?** -Dynamic columns generate data automatically by running prompts, evaluations, API calls, or code against your dataset rows. See [Dynamic Columns](/docs/dataset/concept/dynamic-column). +Dynamic columns generate data automatically by running prompts, evaluations, API calls, or code against your dataset rows. See [Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns). **Can I generate synthetic data?** -Yes. Define a schema (columns, types, constraints) and the platform generates realistic rows. See [Synthetic Data](/docs/dataset/concept/synthetic-data). +Yes. Define a schema (columns, types, constraints) and the platform generates realistic rows. See [Synthetic Data](/docs/dataset/concepts/synthetic-data). --- @@ -65,11 +65,11 @@ Simulation lets you test voice and chat AI agents against simulated customers in **How do I run a voice simulation?** -Create an agent definition, scenarios, and personas, then run a test from the platform. See [Run Voice Simulation](/docs/simulation/features/run-simulation). +Create an agent definition, scenarios, and personas, then run a test from the platform. See [Run Voice Simulation](/docs/simulation/guides/run-voice-simulation). **Can I run chat simulations from code?** -Yes, using the Python SDK. See [Chat Simulation Using SDK](/docs/simulation/features/simulation-using-sdk). +Yes, using the Python SDK. See [Chat Simulation Using SDK](/docs/simulation/guides/run-chat-simulation). --- @@ -81,7 +81,7 @@ Annotations are human labels applied to AI outputs (traces, spans, sessions, dat **What's the difference between inline and queue-based annotations?** -Inline annotations are quick, ad-hoc labels from detail views. Queue-based annotations use managed campaigns with assignment, progress tracking, and agreement metrics. See [Inline Annotations](/docs/annotations/features/inline). +Inline annotations are quick, ad-hoc labels from detail views. Queue-based annotations use managed campaigns with assignment, progress tracking, and agreement metrics. See [Inline Annotations](/docs/annotations/guides/annotate-without-a-queue). --- @@ -97,18 +97,6 @@ Every edit creates a new version. Assign labels (Production, Staging) to version --- -## Prototype - -**What is Prototype?** - -Prototype is a pre-production testing environment. You run multiple versions of your application side by side and compare eval scores, cost, and latency. See [Prototype Overview](/docs/prototype). - -**How do I choose a winning version?** - -Use the Choose Winner flow to weight metrics and rank versions. See [Choose Winner](/docs/prototype/features/choose-winner). - ---- - ## Optimization **How does optimization work?** @@ -117,7 +105,7 @@ Optimization takes a prompt, runs it against your data, scores the outputs with **Can I optimize from the UI without code?** -Yes. See [Using the Platform](/docs/optimization/features/using-platform). +Yes. See [Using the Platform](/docs/optimization/guides/run-an-optimization). --- @@ -141,7 +129,7 @@ Protect screens inputs and outputs in real time across four dimensions: Content **Can I use Protect with text, images, and audio?** -Yes. Protect works across all three modalities. See [Run Protect via SDK](/docs/protect/features/run-protect). +Yes. Protect works across all three modalities. See [Run Protect via SDK](/docs/protect/guides/run-protect-from-the-sdk). --- @@ -173,11 +161,11 @@ Error Feed automatically analyzes traces from your Observe projects, identifies **How do I add documents to a Knowledge Base?** -Upload files via the [UI](/docs/knowledge-base/features/ui) or programmatically via the [SDK](/docs/knowledge-base/features/sdk). +Upload files via the [UI](/docs/knowledge-base/guides/create-knowledge-base) or programmatically via the [SDK](/docs/knowledge-base/guides/manage-with-the-sdk). **What file types are supported?** -PDF, DOCX, DOC, TXT, and RTF. Maximum 5MB per file. See [Understanding Knowledge Base](/docs/knowledge-base/concepts/concept). +PDF, DOCX, DOC, TXT, and RTF. Maximum 5MB per file. See [Understanding Knowledge Base](/docs/knowledge-base/concepts/understanding-knowledge-base). --- diff --git a/src/pages/docs/get-started/connect-no-code-agents.mdx b/src/pages/docs/get-started/connect-no-code-agents.mdx index 334643ab..45b17b67 100644 --- a/src/pages/docs/get-started/connect-no-code-agents.mdx +++ b/src/pages/docs/get-started/connect-no-code-agents.mdx @@ -8,7 +8,7 @@ An agent definition tells Future AGI which agent you're testing and how to reach Any voice agent can be simulated as long as it's reachable by a **phone number**. **Vapi** and **Retell** are natively supported, so you can sync the agent's name and prompt straight from them -Building a **chat** agent? Chat simulations run through the **SDK**, not this form. See [Chat Simulation Using SDK](/docs/simulation/features/simulation-using-sdk) +Building a **chat** agent? Chat simulations run through the **SDK**, not this form. See [Chat Simulation Using SDK](/docs/simulation/guides/run-chat-simulation) Every definition is versioned, so once you save it you can run tests against a specific version, compare versions, or roll back diff --git a/src/pages/docs/get-started/create-your-first-prompt.mdx b/src/pages/docs/get-started/create-your-first-prompt.mdx index 3300c9b9..fd9f0e09 100644 --- a/src/pages/docs/get-started/create-your-first-prompt.mdx +++ b/src/pages/docs/get-started/create-your-first-prompt.mdx @@ -69,10 +69,10 @@ Not getting a response? Try checking these: ## Dive deeper - + Test the prompt at scale across a dataset - + Pull the prompt into your app and run it programmatically diff --git a/src/pages/docs/index.mdx b/src/pages/docs/index.mdx index 6acd06c9..7fb6500c 100644 --- a/src/pages/docs/index.mdx +++ b/src/pages/docs/index.mdx @@ -9,7 +9,7 @@ It's built for the whole team shipping AI (engineers, product managers, and doma ## The Learning Loop -Every part of Future AGI feeds the next. You [**prototype**](/docs/prototype) and [**simulate**](/docs/simulation) an agent before launch, [**evaluate**](/docs/evaluation) its outputs against built-in and custom metrics, [**observe**](/docs/observe) real traffic once it's live, and [**optimize**](/docs/optimization) from what you learn, then the cycle repeats +Every part of Future AGI feeds the next. You [**simulate**](/docs/simulation) an agent before launch, [**evaluate**](/docs/evaluation) its outputs against built-in and custom metrics, [**observe**](/docs/observe) real traffic once it's live, and [**optimize**](/docs/optimization) from what you learn, then the cycle repeats ![Future AGI platform](/images/agi2.webp) @@ -22,7 +22,7 @@ Because every product shares the same **traces, datasets, and scores**, the work Future AGI is organized into six broad areas: - Build and refine: Prototype, Agent Playground, Prompt, and Dataset + Build and refine: Agent Playground, Prompt, and Dataset One gateway for routing, caching, guardrails, and cost control across 100+ providers Test agents against synthetic users and scenarios before launch Score quality with built-in and custom metrics, guardrails, knowledge bases, and human review @@ -54,5 +54,5 @@ The fastest way to see Future AGI is to get your data flowing: - [Create your first prompt](/docs/get-started/create-your-first-prompt) -**Using Cursor or Claude Code?** Install the Future AGI MCP server to bring the platform and docs straight into your editor. See [Set up the MCP server](/docs/quickstart/setup-mcp-server) +**Using Cursor or Claude Code?** Install the Future AGI MCP server to bring the platform and docs straight into your editor. See [Set up the MCP server](/docs/falcon-ai/guides/use-the-mcp-server) diff --git a/src/pages/docs/knowledge-base/concepts/concept.mdx b/src/pages/docs/knowledge-base/concepts/concept.mdx deleted file mode 100644 index 888839c9..00000000 --- a/src/pages/docs/knowledge-base/concepts/concept.mdx +++ /dev/null @@ -1,51 +0,0 @@ ---- -title: "Understanding Knowledge Base: Content Types and Processing" -description: "Explains what a Knowledge Base is, what content types are supported, and how files are indexed and processed in Future AGI." ---- - -## About - -A **Knowledge Base** is a store of your organization's content that Future AGI indexes and makes available across the platform. When you upload documents, the platform processes and indexes them so they can be used as grounding context for synthetic data generation and evaluations. - -## When to use - -- **Synthetic data generation**: You want generated examples to reflect your domain, terminology, and procedures instead of generic text. Selecting a KB when creating a synthetic dataset grounds the output in your actual content. -- **Evaluation**: You want to check whether model outputs are factually consistent with your organization's knowledge. The KB supplies the reference context that hallucination detection and grounding evals compare against. - -## Supported Content Types - -You can upload the following file types to a Knowledge Base: - -| File Type | Extensions | -|---|---| -| Word documents | `.doc`, `.docx` | -| PDF documents | `.pdf` | -| Plain text | `.txt` | -| Rich text | `.rtf` | - -Maximum file size is 5MB per file. - -Examples of content that works well in a KB: - -- Technical documentation and manuals -- FAQs and troubleshooting guides -- SOPs and process workflows -- Training materials and HR policies -- Legal documents and compliance information -- Product descriptions and specifications - -## File Processing - -After you upload files, the platform processes them automatically. Each file goes through one of three states: - -| Status | Description | -|---|---| -| Successful | Content extracted and indexed for use | -| Processing | File is being processed | -| Failed | Processing failed. You'll be notified and the file won't be usable. | - -## Next Steps - -- [Create KB Using UI](/docs/knowledge-base/features/ui): Upload files through the dashboard -- [Create KB Using SDK](/docs/knowledge-base/features/sdk): Upload and manage knowledge bases programmatically -- [Synthetic Data](/docs/dataset/concept/synthetic-data): Learn how KB grounds synthetic data generation diff --git a/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx b/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx new file mode 100644 index 00000000..85f37a0c --- /dev/null +++ b/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx @@ -0,0 +1,56 @@ +--- +title: "Understanding Knowledge Base" +description: "Ground synthetic data, agent evals, and simulations in your own documents" +--- + +## A knowledge base is one named container + +A **knowledge base** is a single named container that belongs to your organization, optionally scoped to one workspace. It holds two things: the documents you uploaded, and the indexed form of their text that the platform builds from them. Each document is its own record, with its own name and its own status, but they all live inside the one container you named when you created it. + +Indexing happens once, at upload. When you add a file, the platform reads it and extracts its text right away. It does not wait for something else to ask for that content first. That single pass is why the container carries a status of its own: + +- **Processing**: its files are being read +- **Completed**: its files are usable +- **Failed**: a file's extraction didn't work; the reason shows on that file's row + +## Three surfaces read it + +- **[Synthetic data generation](/docs/dataset/concepts/synthetic-data)** (generates dataset rows from a schema you define) points at a knowledge base by name so the rows it produces echo your domain instead of reading like generic text +- **[Agent-type evaluations](/docs/evaluation/concepts/eval-types)** (evals authored as the Agent Evaluator type, which can reason over multiple turns and use tools) attach a knowledge base to give the eval a reference to check the agent's output against +- **A [Simulation agent definition](/docs/simulation/concepts/agent-definitions)** (the config that governs how a simulated agent behaves) can attach one knowledge base of its own, giving the simulated agent something to draw on when it answers + +|"holds"| DOCS["Documents you uploaded"] + KB -->|"holds"| IDX["Indexed text"] + SDG["Synthetic data generation"] -->|"reads"| KB + EVAL["Agent-type evaluation"] -->|"reads"| KB + SIM["Simulation agent definition"] -->|"reads"| KB`} /> + +## Not your agent's retrieval store + +Read it as reference material, not as your agent's live retrieval path. When your production agent answers a real request, it is not opening this container and searching it. A knowledge base is a document store that the three surfaces above consult when they run. There is no versioning either: a knowledge base holds whatever documents are in it right now, not snapshots of what it held before. If you're looking for a live index your agent queries during retrieval, that's the dataset's [Retrieval column](/docs/dataset/reference/dynamic-column-methods), which connects to Pinecone, Qdrant, or Weaviate, not to a knowledge base. + +## Why it matters + +Without a knowledge base behind it, a generator or an evaluator has nothing to ground itself in: ask for synthetic data on a topic you never described and it produces plausible wording that isn't yours. Point that generation, or an agent-type eval, at a knowledge base and it draws on your actual wording and your actual steps instead. + +## Limits and supported files + +- **Supported file types**: PDF, DOCX, TXT, and RTF only, anything else yields no extracted text +- **Size cap**: 1 GB per knowledge base, across all its documents combined + +## Keep exploring + + + + Upload documents and start indexing + + + Add or remove documents from an existing container + + + Create, update, and attach knowledge bases programmatically + + diff --git a/src/pages/docs/knowledge-base/features/sdk.mdx b/src/pages/docs/knowledge-base/features/sdk.mdx deleted file mode 100644 index 46549de4..00000000 --- a/src/pages/docs/knowledge-base/features/sdk.mdx +++ /dev/null @@ -1,117 +0,0 @@ ---- -title: "Create a Knowledge Base Using the Future AGI Python SDK" -description: "Create and manage Knowledge Bases programmatically with the Future AGI Python SDK: create, update, add or remove files, and delete KBs." ---- - -## About - -The Knowledge Base SDK lets you create and manage Knowledge Bases from code using the Future AGI Python SDK. You install the `futureagi` package, authenticate with API credentials, then call methods to create a KB with a name and file paths (or a directory), add more files to an existing KB, remove files by name, or delete entire KBs. Supported file types are PDF, DOCX, TXT, and RTF. - -## When to use - -- **Automation**: Create or update KBs from scripts, pipelines, or scheduled jobs. -- **Bulk ingestion**: Upload many files or point at a directory path instead of selecting files one by one in the UI. -- **Larger files**: Use the SDK when file size or volume exceeds UI limits. -- **Reproducibility**: Version and replay KB setup in code (e.g. in a repo or notebook). -- **Integrations**: Embed KB creation/updates in your own tools or workflows. - ---- - -## How to - - - - Install the Future AGI Python package: - - ```bash - pip install futureagi - ``` - - - - Create a `KnowledgeBase` client with your API key and secret (from the Future AGI platform). Optionally pass `fi_base_url` for a custom API base. - - ```python - from fi.kb import KnowledgeBase - - client = KnowledgeBase( - fi_api_key="YOUR_API_KEY", - fi_secret_key="YOUR_SECRET_KEY" - ) - ``` - - - - Call `create_kb` with a name and either a list of file paths or a single directory path. The SDK uploads the files and returns the same client with the new KB cached in `client.kb` (id, name, files). Supported extensions: `pdf`, `docx`, `txt`, `rtf`. - - ```python - client = client.create_kb( - name="my-knowledge-base", - file_paths=["path/to/file1.pdf", "path/to/file2.txt"] - ) - print(f"Created KB: {client.kb.id} — {client.kb.name}") - ``` - - To use all files in a directory: - - ```python - client = client.create_kb( - name="my-knowledge-base", - file_paths="path/to/docs_folder" - ) - ``` - - - - To add files or rename an existing KB, use `update_kb`. The first argument is the KB name (the SDK resolves it to the existing KB). You can pass `new_name` to rename and/or `file_paths` to add more files. - - ```python - client.update_kb( - kb_name="my-knowledge-base", - file_paths=["path/to/extra_file.pdf"] - ) - # Or rename and add files: - client.update_kb( - kb_name="my-knowledge-base", - new_name="my-renamed-kb", - file_paths=["path/to/extra_file.pdf"] - ) - ``` - - - - To remove specific documents, call `delete_files_from_kb` with the **file names** (as stored in the KB), not file IDs. You can pass `kb_name` if the client is not already targeting that KB. - - ```python - client.delete_files_from_kb( - file_names=["file1.pdf", "file2.txt"], - kb_name="my-knowledge-base" # optional if client already has this KB - ) - ``` - - - - To delete one or more KBs, use `delete_kb` with either `kb_ids` or `kb_names`. The client can target the current cached KB if you don’t pass either. - - ```python - client.delete_kb(kb_ids=[str(client.kb.id)]) - # Or by name: - client.delete_kb(kb_names=["my-knowledge-base"]) - ``` - - - - - Wrap SDK calls in try/except and handle `fi.utils.errors.SDKException` (and optionally `InvalidAuthError`, rate limits) in production. Keep API credentials out of version control (e.g. use environment variables). - - -## Next Steps - - - - Full SDK reference for the Knowledge Base module with all methods and parameters. - - - Create and populate a Knowledge Base from the platform without code. - - diff --git a/src/pages/docs/knowledge-base/features/ui.mdx b/src/pages/docs/knowledge-base/features/ui.mdx deleted file mode 100644 index eb607043..00000000 --- a/src/pages/docs/knowledge-base/features/ui.mdx +++ /dev/null @@ -1,91 +0,0 @@ ---- -title: "Create a Knowledge Base Using the Future AGI Platform UI" -description: "Create and populate a Knowledge Base from the Future AGI platform: name it, upload documents, and wait for processing to finish." ---- - -{/* ARCADE EMBED START */} -
- - ---- - -## When to use - -- **Validating a prompt change in production**: Compare latency and cost between versions on real traffic, not just test runs. -- **Diagnosing a cost spike**: Metrics per prompt version show exactly which prompt or version is driving spend. -- **Comparing active versions**: See real-world performance across prompt versions side by side to decide which to keep. -- **Auditing prompt usage**: Trace count shows which prompts are actively being called and which are stale or abandoned. - ---- - -## Linked Traces vs Raw Traces - -| | Raw traces | Linked traces | -|---|---|---| -| **What you see** | Application-level metrics | Metrics per prompt and version | -| **Attribution** | Anonymous API calls | Tied to a specific template and version | -| **Where to view** | Observe / tracing dashboard | Prompt Workbench Metrics tab | -| **Setup required** | SDK instrumentation | SDK instrumentation + template reference in request | - ---- - -## How to - -To link prompts to traces, you need to associate the prompt used in a generation with the corresponding trace. The process is described in the observability and manual tracing docs: [Log prompt templates](/docs/sdk/tracing/log-prompt-templates). Once your application sends traces that include the prompt template (or template ID), Future AGI links those traces to the prompt in the Prompt Workbench. - ---- - -## Metrics and Analytics - -After linking, open your prompt in the dashboard and go to the **Metrics** tab. - -| Metric | What it tells you | -|---|---| -| **Median Latency** | Typical time for the model to produce a response. Lower is better for responsiveness; use it to spot slow prompts or model changes. | -| **Median Input Tokens** | Typical size of the prompt sent to the model. Helps you see verbosity and compare input length across versions. | -| **Median Output Tokens** | Typical length of the model's reply. Useful for cost and length control; compare after changing instructions or max tokens. | -| **Median Costs** | Typical cost per generation for this prompt. Use it to compare cost across prompt versions or models. | -| **Traces Count** | How many times this prompt was used in the selected period. Shows which prompts are active and where to focus optimization. | -| **First and Last Generation** | When the prompt was first and last used. Confirms the time range of the data you're viewing. | - -Compare the same metric across **prompt versions** or **time ranges** to see if a change improved latency, cost, or token usage. - ---- - -## Next Steps - - - - How versioning and deployment labels work. - - - Manage and fetch prompts programmatically. - - - Set up the trace-to-prompt connection in your application. - - diff --git a/src/pages/docs/prompt/features/sdk.mdx b/src/pages/docs/prompt/features/sdk.mdx deleted file mode 100644 index 9b947930..00000000 --- a/src/pages/docs/prompt/features/sdk.mdx +++ /dev/null @@ -1,384 +0,0 @@ ---- -title: "Prompt Workbench SDK: Create and Version Prompts in Code" -description: "Create, version, and run prompt templates programmatically using the Future AGI SDK for TypeScript/JavaScript or Python applications." ---- - -## About - -The Prompt Workbench SDK lets you manage prompt templates programmatically. Instead of using the UI, you define, version, and deploy prompts from code using Python or TypeScript/JavaScript. - -This decouples prompt changes from application deploys. Your application fetches the active prompt by name and label at runtime, so you can update it on the platform without touching or redeploying your code. You can also assign labels like Production and Staging to control which version is live, run A/B tests across variants, and compile runtime variables into messages before sending them to a model. - ---- - -## When to use - -- **Prompts as part of CI/CD**: You want to version, commit, and deploy prompt changes through the same pipeline as your application code. -- **Runtime prompt resolution**: Your application fetches the active prompt by name and label at runtime, so you can update prompts on the platform without a code deploy. -- **A/B testing prompt variants**: You run multiple labeled versions of the same prompt in production and compare results across variants. -- **Dynamic inputs at compile time**: Your prompts include placeholders for chat history or other message lists that are injected at runtime. - ---- - -## Installation - - - -```bash -npm install @future-agi/sdk -``` - -```bash -pip install futureagi -``` - - - - -The Python package is installed as **`futureagi`** but imported as **`fi`** (e.g. `from fi.prompt.client import Prompt`). - - ---- - -## Template structure - -### Basic components - -- **Name**: unique identifier (required) -- **Messages**: ordered list of messages -- **Model configuration**: model + generation params -- **Variables**: dynamic placeholders used in messages - -### Message types - -- **System**: sets behavior/context -- **User**: contains the prompt; supports variables like `{{var}}` -- **Assistant**: few-shot examples or expected outputs - -```json -{ "role": "system", "content": "You are a helpful assistant." } -{ "role": "user", "content": "Introduce {{name}} from {{city}}." } -{ "role": "assistant", "content": "Meet Ada from Berlin!" } -``` - ---- - -## Model configuration fields - -`model_name`, `temperature`, `frequency_penalty`, `presence_penalty`, `max_tokens`, `top_p`, `response_format`, `tool_choice`, `tools` - ---- - -## Placeholders and compile - -Add a placeholder message (`type="placeholder"`, `name="..."`) in your template. At compile time, supply an array of messages for that key; `{{var}}` variables are substituted in all message contents. - - - -```typescript JS/TS -import { PromptTemplate, ModelConfig, MessageBase, Prompt } from "@future-agi/sdk"; - -const tpl = new PromptTemplate({ - name: "chat-template", - messages: [ - { role: "system", content: "You are a helpful assistant." } as MessageBase, - { role: "user", content: "Hello {{name}}!" } as MessageBase, - { type: "placeholder", name: "history" } as any, // placeholder - ], - model_configuration: new ModelConfig({ model_name: "gpt-4o-mini" }), -}); - -const client = new Prompt(tpl); -// Compile with substitution and inlined chat history -const compiled = client.compile({ - name: "Alice", - history: [{ role: "user", content: "Ping {{name}}" }], -} as any); -``` - -```python Python -from fi.prompt import Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage - -tpl = PromptTemplate( - name="chat-template", - messages=[ - SystemMessage(content="You are a helpful assistant."), - UserMessage(content="Hello {{name}}!"), - {"type": "placeholder", "name": "history"}, - ], - model_configuration=ModelConfig(model_name="gpt-4o-mini"), -) - -client = Prompt(template=tpl) -compiled = client.compile(name="Alice", history=[{"role": "user", "content": "Ping {{name}}"}]) -``` - - - ---- - -## Create templates - - - -```typescript JS/TS -import { Prompt, PromptTemplate, ModelConfig, MessageBase } from "@future-agi/sdk"; - -const tpl = new PromptTemplate({ - name: "intro-template", - messages: [ - { role: "system", content: "You are a helpful assistant." } as MessageBase, - { role: "user", content: "Introduce {{name}} from {{city}}." } as MessageBase, - ], - variable_names: { name: ["Ada"], city: ["Berlin"] }, - model_configuration: new ModelConfig({ model_name: "gpt-4o-mini" }), -}); - -const client = new Prompt(tpl); -await client.open(); // draft v1 -await client.commitCurrentVersion("Finish v1", true); // set default -``` - -```python Python -from fi.prompt import Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage - -tpl = PromptTemplate( - name="intro-template", - messages=[ - SystemMessage(content="You are a helpful assistant."), - UserMessage(content="Introduce {{name}} from {{city}}."), - ], - variable_names={"name": ["Ada"], "city": ["Berlin"]}, - model_configuration=ModelConfig(model_name="gpt-4o-mini"), -) - -client = Prompt(template=tpl).create() # draft v1 -client.commit_current_version(message="Finish v1", set_default=True) -``` - - - ---- - -## Versioning (step-by-step) - -- Build the template (see above) -- Create draft v1 (JS/TS: `await client.open()`; Python: `client.create()`) -- Update draft & save (JS/TS: `saveCurrentDraft()`; Python: `save_current_draft()`) -- Commit v1 and set default (JS/TS: `commitCurrentVersion("msg", true)`; Python: `commit_current_version`) -- Open a new draft (JS/TS: `createNewVersion()`; Python: `create_new_version()`) -- Delete if needed (JS/TS: `delete()`; Python: `delete()`) - ---- - -## Labels (deployment control) - -- **System labels**: Production, Staging, Development (predefined by backend) -- **Custom labels**: create explicitly and assign to versions -- **Name-based APIs**: manage by names (no IDs needed) -- **Draft safety**: cannot assign labels to drafts; assignments are queued and applied on commit - -### Assign labels - - - -```typescript JS/TS -// Assign by instance (current project) -await client.labels().assign("Production", "v1"); -await client.labels().assign("Staging", "v2"); - -// Create and assign a custom label -await client.labels().create("Canary"); -await client.labels().assign("Canary", "v2"); - -// Class helpers by names (org-wide context) -await Prompt.assignLabelToTemplateVersion("intro-template", "v2", "Development"); -``` - -```python Python -# Assign by instance -client.assign_label("Production", version="v1") -client.assign_label("Staging", version="v2") - -# Create and assign a custom label -client.create_label("Canary") -client.assign_label("Canary", version="v2") - -# Class helpers by names -Prompt.assign_label_to_template_version(template_name="intro-template", version="v2", label="Development") -``` - - - -### Remove labels - - - -```typescript JS/TS -await client.labels().remove("Canary", "v2"); -await Prompt.removeLabelFromTemplateVersion("intro-template", "v2", "Development"); -``` - -```python Python -client.remove_label("Canary", version="v2") -Prompt.remove_label_from_template_version(template_name="intro-template", version="v2", label="Development") -``` - - - -### List labels and mappings - - - -```typescript JS/TS -const labels = await client.labels().list(); // system + custom -const mapping = await Prompt.getTemplateLabels({ template_name: "intro-template" }); -``` - -```python Python -labels = client.list_labels() -mapping = Prompt.get_template_labels(template_name="intro-template") -``` - - - ---- - -## Fetch by name + label (or version) - - -
    -
  • Precedence: version > label
  • -
  • Python default: if no label is provided, defaults to "production"
  • -
  • Return type: get_template_by_name() returns a Prompt instance (not a raw PromptTemplate). In Python you can call .compile() directly on it; in TypeScript you wrap the returned template in new Prompt(tpl) then call .compile().
  • -
-
- - - -```typescript JS/TS -import { Prompt } from "@future-agi/sdk"; - -const tplByLabel = await Prompt.getTemplateByName("intro-template", { label: "Production" }); -const tplByVersion = await Prompt.getTemplateByName("intro-template", { version: "v2" }); -``` - -```python Python -from fi.prompt import Prompt -tpl_by_label = Prompt.get_template_by_name("intro-template", label="Production") -tpl_by_version = Prompt.get_template_by_name("intro-template", version="v2") -``` - - - ---- - -## A/B testing with labels (compile → OpenAI gpt-4o) - -Fetch two labeled versions of the same template (e.g., `prod-a` and `prod-b`), randomly select one, compile variables, and send the compiled messages to OpenAI. - - -The compile() API replaces {`{{var}}`} in string contents and preserves structured contents. Ensure your template contains the variables you pass (e.g., {`{{name}}`}, {`{{city}}`}). - - - - -```typescript JS/TS -import OpenAI from "openai"; -import { Prompt, PromptTemplate } from "@future-agi/sdk"; - -const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY! }); - -// Fetch both label variants -const [tplA, tplB] = await Promise.all([ - Prompt.getTemplateByName("my-template-name", { label: "prod-a" }), - Prompt.getTemplateByName("my-template-name", { label: "prod-b" }), -]); - -// Randomly select a variant -const selected = Math.random() < 0.5 ? tplA : tplB; -const client = new Prompt(selected as PromptTemplate); - -// Compile variables into the template messages -const compiled = client.compile({ name: "Ada", city: "Berlin" }); - -// Send to OpenAI gpt-4o -const completion = await openai.chat.completions.create({ - model: "gpt-4o", - messages: compiled as any, -}); - -const resultText = completion.choices[0]?.message?.content; -``` - -```python Python -import os -import random - -from openai import OpenAI -from fi.prompt import Prompt - -openai_client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) - -# Fetch both label variants (each returns a Prompt instance) -client_a = Prompt.get_template_by_name("my-template-name", label="prod-a") -client_b = Prompt.get_template_by_name("my-template-name", label="prod-b") - -# Randomly select a variant -selected_client = client_a if random.random() < 0.5 else client_b - -# Compile variables into the template messages -compiled = selected_client.compile(name="Ada", city="Berlin") - -# Send to OpenAI gpt-4o -response = openai_client.chat.completions.create( - model="gpt-4o", - messages=compiled, -) -result_text = response.choices[0].message.content -# For analytics, log selected_client.template.version or the label (e.g. "prod-a" / "prod-b") -``` - - - - -For analytics, attach the selected label/version to your logs or tracing so A/B results can be compared. - - ---- - -## Compile output format - -The `compile()` method returns messages in a provider-agnostic format. Each message has `role` and `content`; `content` may be a string or a structured list of parts (e.g. text, images) depending on the SDK and template. - -**Example output structure:** - -```json -[ - {"role": "system", "content": "You are a helpful assistant."}, - {"role": "user", "content": "Hello Ada from Berlin!"} -] -``` - - -If your SDK or backend returns content as a stringified list of content parts (e.g. for multimodal content), you may need an adapter to convert to your target LLM provider’s format (e.g. OpenAI’s role + content string). - - ---- - -## Next Steps - - - - Build and run prompts in the UI. - - - Generate a prompt draft from a plain-language description. - - - Connect prompts to traces to monitor performance in production. - - - How prompts fit into the platform. - - diff --git a/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx b/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx new file mode 100644 index 00000000..8ecf76b0 --- /dev/null +++ b/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx @@ -0,0 +1,45 @@ +--- +title: "Commit & compare versions" +description: "Turn a working draft into a version you can point to, then compare a few side by side" +--- + +A prompt starts out as a [draft](/docs/prompt/concepts/versions-and-labels) you're still editing. Running it once and committing turns that draft into a version, one that sticks around after you move on to the next edit. + +## Commit a version + +This picks up with the support-agent template already open in the editor, either fresh from [Create a prompt](/docs/prompt/guides/create-a-prompt) or one you're mid-edit on. + +Click **Run Prompt** to produce an output; see [Run a prompt](/docs/prompt/guides/run-a-prompt) for the full walkthrough of messages, models, and variables. Running is also what clears the **Draft** badge next to the version number, which is the same thing that enables **Commit**. Until then, hover the disabled button and the tooltip reads, **"Please run the prompt before saving and committing"**. It also greys out while more than one version is loaded for comparison, and the tooltip still shows that same message. + +Click **Commit** in the editor header. This opens the **Commit changes to prompt** dialog. Type a message in the **Commit message** field, something like "Added escalation trigger for refund requests", since both **Commit** and **Commit and set as a default version** stay disabled until you do. Pick **Commit** to save the version as is, or **Commit and set as a default version** to save it and also make it the template's default in the same step. + +Once you click **Commit**, the dialog closes and a snackbar confirms it, something like `Commit for successful` (or `...and set as default` if you picked **Commit and set as a default version** instead). The version now shows up under **Commit History**, which lists just the versions you've committed. + +## Compare versions + +Line up a few versions to see how they differ before you decide which one to promote. + +Click **More** in the editor header, then **History**. It lists every version, including drafts you haven't committed yet; **Commit History** narrows that down to the ones you've committed. + +Click **Select to compare** to turn on checkboxes next to each version. The version already open in the editor comes pre-checked and locked, and it counts as one of the three, so you're picking at most two more. + +Click **Compare**, which appears once you've checked at least one version. Each selected version opens in its own panel, showing its messages, its own last saved output, and its own model configuration side by side, often the detail that differs most between versions. + + +Three is the hard cap. Once three are checked, the remaining checkboxes go disabled with **"Compare limit is upto 3 version only, Deselect other options to select this one"**. Deselect one to swap in another. + + +## Promote a version + +Comparing tells you which version should be live. Making it live isn't a commit or compare action, it's a labelling one: you point the Production label at the version you picked instead of changing the version itself. To do it from your own code, use the [SDK & API](/docs/prompt/reference/sdk-api) reference. + +## Dive deeper + + + + Score outputs across the versions you just compared + + + Assign labels and fetch versions from your own code + + diff --git a/src/pages/docs/prompt/guides/create-a-prompt.mdx b/src/pages/docs/prompt/guides/create-a-prompt.mdx new file mode 100644 index 00000000..e003f474 --- /dev/null +++ b/src/pages/docs/prompt/guides/create-a-prompt.mdx @@ -0,0 +1,58 @@ +--- +title: "Create a prompt" +description: "Generate with AI, start from scratch, or use a template, then give the prompt a name" +--- + +Every prompt starts in the same place: the [Prompts](/docs/prompt/concepts/understanding-prompts) directory. This guide walks you from there, through the **Create a new prompt** modal, to a named, open editor. + + + + In the left navigation, click **Prompts**. You land in the directory, rooted at **All Prompts** and **My templates**. + + + In the directory toolbar, click **Create prompt**. This opens the **Create a new prompt** modal. + + + Pick one of the three options in the modal. Each gets its own section below. + + + +## Pick a starting point + +### Generate with AI + +Pick this when you don't have the wording yet and want a starting draft to edit. Click **Generate with AI** in the modal: the platform creates the prompt and opens the editor, with the **Generate a prompt** drawer open on top of it. Type a plain-language description of what the prompt should do, for example "write a support agent that answers customer questions using our returns policy," then click **Generate**. Review the generated prompt, then click **Continue** to drop it into the user message. The system message is still yours to write. + +### Start from scratch + +Pick this when you already know what the [system and user messages](/docs/prompt/concepts/understanding-prompts) should say. Click **Start from scratch** in the modal and the editor opens empty right away: you write the system and user messages yourself. + +### Start with a template + +Pick this when a team pattern for this kind of prompt already exists. Click **Start with a template** in the modal. The template browser opens instead of the editor: pick a category from the sidebar or search by name, open a template to preview it, then click **Use this template** to load its content into a new prompt. + + +To skip the modal and go straight to the template browser, click **Use template** directly in the directory toolbar instead. It opens the same template browser as this route. + + +## Rename it + +The **Generate with AI** and **Start from scratch** routes open the new prompt named `Untitled-1` (or the next free number in your organization). The **Start with a template** route names it `Untitled-1-