HITL Overview
Allow your agent and users to collaborate on complex tasks.
Open your coding agent in your project's folder, or in an empty folder for a new app.This runs in a coding agent on your computer.
"""AG2 agent with weather and sales tools for CopilotKit showcase.Uses AG2's ConversableAgent with AGUIStream to exposethe agent via the AG-UI protocol."""from __future__ import annotationsimport jsonimport loggingfrom typing import Annotated, Anyimport openaifrom autogen import ConversableAgent, LLMConfigfrom autogen.ag_ui import AGUIStreamfrom dotenv import load_dotenvfrom pydantic import ValidationErrorload_dotenv()# Import shared tool implementationsfrom tools import ( get_weather_impl, query_data_impl, manage_sales_todos_impl, get_sales_todos_impl, schedule_meeting_impl, search_flights_impl, build_a2ui_operations_from_tool_call, RENDER_A2UI_TOOL_SCHEMA,)from tools.types import Flightfrom ._header_forwarding import get_forwarded_headersfrom ._request_context import get_latest_user_messagelogger = logging.getLogger(__name__)# Module-level async client: re-used across requests (httpx connection pool is# thread-safe). Using AsyncOpenAI inside an `async def` avoids blocking the# ASGI event loop on the secondary LLM call._async_openai_client = openai.AsyncOpenAI()# =====# Tools# =====async def get_weather( location: Annotated[str, "City name to get weather for"],) -> str: """Get current weather for a location.""" result = get_weather_impl(location) # Return a JSON string (not a dict): autogen serializes dict returns with # str(), producing a Python repr (single quotes) that the frontend's # parseJsonResult/JSON.parse cannot parse — the weather card then renders # "--" placeholders. Same pattern as search_flights below. return json.dumps( { "city": result["city"], "temperature": result["temperature"], "feels_like": result["feels_like"], "humidity": result["humidity"], "wind_speed": result["wind_speed"], "conditions": result["conditions"], } )async def query_data( query: Annotated[str, "Natural language query for financial data"],) -> str: """Query financial database for chart data.""" # Return a JSON string (not a list): autogen serializes non-str returns # with str(), producing a Python repr (single quotes) that the frontend's # parseJsonResult/JSON.parse cannot parse. Same pattern as get_weather. return json.dumps(query_data_impl(query))async def manage_sales_todos( todos: Annotated[list, "Complete list of sales todos"],) -> str: """Manage the sales pipeline.""" # See contract comment on query_data above — return JSON, not dict. # SalesTodo is a Pydantic model; coerce via model_dump for serialisability. result = [t.model_dump() for t in manage_sales_todos_impl(todos)] return json.dumps({"todos": result})async def get_sales_todos() -> str: """Get the current sales pipeline.""" # See contract comment on query_data above — return JSON, not list. # SalesTodo is a Pydantic model; coerce via model_dump for serialisability. return json.dumps([t.model_dump() for t in get_sales_todos_impl(None)])async def schedule_meeting( reason: Annotated[str, "Reason for the meeting"],) -> str: """Schedule a meeting with user approval.""" # See contract comment on query_data above — return JSON, not dict. return json.dumps(schedule_meeting_impl(reason))async def search_flights( flights: Annotated[ list[dict[str, Any]], "List of flight objects to display as rich A2UI cards" ],) -> str: """Search for flights and display the results as rich cards. Return exactly 2 flights. Each flight must have: airline, airlineLogo, flightNumber, origin, destination, date (short readable format like "Tue, Mar 18" -- use near-future dates), departureTime, arrivalTime, duration (e.g. "4h 25m"), status (e.g. "On Time" or "Delayed"), statusColor (hex color for status dot), price (e.g. "$289"), and currency (e.g. "USD"). For airlineLogo use Google favicon API: https://www.google.com/s2/favicons?domain={airline_domain}&sz=128 """ try: typed_flights: list[Flight] = [Flight(**f) for f in flights] except ValidationError as exc: logger.warning( "search_flights: invalid flight shape type=%s err=%s", type(exc).__name__, exc, exc_info=True, ) return json.dumps({"error": f"invalid flight shape: {exc}"}) result = search_flights_impl(typed_flights) return json.dumps(result)async def generate_a2ui( context: Annotated[str, "Conversation context to generate UI for"],) -> str: """Generate dynamic A2UI components based on the conversation. A secondary LLM designs the UI schema and data. The result is returned as an a2ui_operations container for the middleware to detect. """ # A13: AsyncOpenAI inside async def (was sync openai.OpenAI which blocks # the ASGI event loop). Forward x-* headers via extra_headers in addition # to the global httpx hook so aimock context routing is explicit at the # call site. # # R2-A1 / A4: thread the latest user prompt from the inbound # RunAgentInput.messages payload (captured into a per-request ContextVar # by RequestUserMessageMiddleware — see agents/_request_context.py) into # the inner LLM call so each pill's request body is byte-distinct. # Without this, every pill landing on the omnibus agent (agentic-chat / # tool-rendering / chat-customization-css / hitl) produces an IDENTICAL # inner-LLM body and the aimock fixture cannot disambiguate. Falls back # to the original hardcoded prompt when the middleware captured nothing # (parse failure already logged at WARNING). user_prompt = get_latest_user_message() or ( "Generate a dynamic A2UI dashboard based on the conversation." ) forwarded = get_forwarded_headers() try: response = await _async_openai_client.chat.completions.create( model="gpt-5-mini", messages=[ { "role": "system", "content": context or "Generate a useful dashboard UI.", }, { "role": "user", "content": user_prompt, }, ], tools=[ { "type": "function", "function": RENDER_A2UI_TOOL_SCHEMA, } ], tool_choice={"type": "function", "function": {"name": "render_a2ui"}}, extra_headers=forwarded or None, ) except Exception as exc: logger.error( "generate_a2ui: inner LLM call failed type=%s err=%s", type(exc).__name__, exc, exc_info=True, ) return json.dumps({"error": f"inner LLM call failed: {type(exc).__name__}"}) if not response.choices: logger.warning("generate_a2ui: LLM returned no choices") return json.dumps({"error": "LLM returned no choices"}) choice = response.choices[0] if not choice.message.tool_calls: logger.warning("generate_a2ui: secondary LLM produced no render_a2ui tool call") return json.dumps({"error": "LLM did not call render_a2ui"}) try: args = json.loads(choice.message.tool_calls[0].function.arguments) result = build_a2ui_operations_from_tool_call(args) return json.dumps(result) except (json.JSONDecodeError, KeyError, TypeError, ValueError) as exc: logger.error( "generate_a2ui: failed to parse render_a2ui args type=%s err=%s", type(exc).__name__, exc, exc_info=True, ) return json.dumps( {"error": f"failed to parse render_a2ui args: {type(exc).__name__}"} )# =====# Agent# =====agent = ConversableAgent( name="assistant", system_message=( "You are a helpful sales assistant. You can look up current weather " "for any city using the get_weather tool, query financial data with " "query_data, manage the sales pipeline with manage_sales_todos and " "get_sales_todos, schedule meetings with schedule_meeting, search " "flights and display rich A2UI cards with search_flights, and " "generate dynamic A2UI dashboards with generate_a2ui. " "When asked about the weather, always use the tool rather than guessing. " "Be concise and friendly in your responses." ), llm_config=LLMConfig({"model": "gpt-5-mini", "stream": True}), human_input_mode="NEVER", # Guard against infinite tool-call loops: AG2's ConversableAgent with # human_input_mode="NEVER" will keep executing tool calls indefinitely # if the LLM keeps requesting them. Without this limit the agent floods # Railway's log stream (500 logs/sec rate-limit), becomes unresponsive # to health probes, and gets killed by the watchdog. max_consecutive_auto_reply=15, functions=[ get_weather, query_data, manage_sales_todos, get_sales_todos, schedule_meeting, search_flights, generate_a2ui, ],)# AG-UI stream wrapperstream = AGUIStream(agent)See this in Inspector
Open Inspector on localhost. Go to Agents, then Frontend Tools. Your tool and its schema are listed.
More detail: Inspector.
What is this?#
Human-in-the-loop (HITL) lets an agent pause mid-run to collect input, confirmation, or a choice from the user, then resume with that answer folded back into its reasoning. It's what turns an autonomous workflow into a collaborative one: the agent keeps its context, the user keeps the steering wheel.
When should I use this?#
Use HITL when you need:
- Quality control — a human gate at high-stakes decision points
- Edge cases — graceful fallbacks when the agent's confidence is low
- Expert input — lean on the user for domain knowledge the model lacks
- Reliability — a more robust loop for real-world, production traffic
Two patterns for HITL in CopilotKit#
CopilotKit ships two complementary ways to pause an agent turn and ask the human something. They look similar from the outside (the chat pauses, a custom component appears, the user answers, the run resumes) but they're wired differently on the backend, and each has its own niche.
| Pattern | Who decides to pause? | Backend surface |
|---|---|---|
useHumanInTheLoop | The LLM, by calling a registered client-side tool | A frontend-only tool description (Zod schema + render) |
useInterrupt | The graph, by calling interrupt(...) during a node | A server-side interrupt() call in your LangGraph agent |
Pick useHumanInTheLoop when the pause is an agent-initiated
decision — the model chose to ask the user — and you want the picker UI
inlined into the normal tool-call flow.
Pick useInterrupt when the pause is a graph-enforced checkpoint —
the code path deterministically requires a human answer — and you want
langgraph.interrupt() as the server-side contract.
Pattern 1 — useHumanInTheLoop (tool-based)#
The agent registers a HITL tool on the client with useHumanInTheLoop.
When the LLM calls that tool, CopilotKit routes the call through your
render function, which shows a custom component and calls respond
with the user's answer. The agent sees the answer as the tool result and
continues from there.
import React from "react";import { CopilotKit, CopilotChat, useHumanInTheLoop, useConfigureSuggestions,} from "@copilotkit/react-core/v2";import { z } from "zod";import { TimePickerCard, TimeSlot } from "./time-picker-card";const DEFAULT_SLOTS: TimeSlot[] = [ { label: "Tomorrow 10:00 AM", iso: "2026-04-19T10:00:00-07:00" }, { label: "Tomorrow 2:00 PM", iso: "2026-04-19T14:00:00-07:00" }, { label: "Monday 9:00 AM", iso: "2026-04-21T09:00:00-07:00" }, { label: "Monday 3:30 PM", iso: "2026-04-21T15:30:00-07:00" },];export default function HitlInChatDemo() { return ( <CopilotKit runtimeUrl="/api/copilotkit" agent="hitl-in-chat"> <div className="flex justify-center items-center h-screen w-full"> <div className="h-full w-full max-w-4xl"> <Chat /> </div> </div> </CopilotKit> );}function Chat() { useConfigureSuggestions({ suggestions: [ { title: "Book a call with sales", message: "Please book an intro call with the sales team to discuss pricing.", }, { title: "Schedule a 1:1 with Alice", message: "Schedule a 1:1 with Alice next week to review Q2 goals.", }, ], available: "always", }); useHumanInTheLoop({ agentId: "hitl-in-chat", name: "book_call", description: "Ask the user to pick a time slot for a call. The picker UI presents fixed candidate slots; the user's choice is returned to the agent.", parameters: z.object({ topic: z .string() .describe("What the call is about (e.g. 'Intro with sales')"), attendee: z .string() .describe("Who the call is with (e.g. 'Alice from Sales')"), }), render: ({ args, status, respond }: any) => ( <TimePickerCard topic={args?.topic ?? "a call"} attendee={args?.attendee} slots={DEFAULT_SLOTS} status={status} onSubmit={(result) => respond?.(result)} /> ), });The picker UI is fed a static list of candidate slots — this is just data the demo page owns, so you can swap in real availability, a calendar API, or anything else:
import React from "react";import { CopilotKit, CopilotChat, useHumanInTheLoop, useConfigureSuggestions,} from "@copilotkit/react-core/v2";import { z } from "zod";import { TimePickerCard, TimeSlot } from "./time-picker-card";const DEFAULT_SLOTS: TimeSlot[] = [ { label: "Tomorrow 10:00 AM", iso: "2026-04-19T10:00:00-07:00" }, { label: "Tomorrow 2:00 PM", iso: "2026-04-19T14:00:00-07:00" }, { label: "Monday 9:00 AM", iso: "2026-04-21T09:00:00-07:00" }, { label: "Monday 3:30 PM", iso: "2026-04-21T15:30:00-07:00" },];Pattern 2 — useInterrupt (graph-paused)#
With LangGraph's interrupt() the pause is enforced by the graph
itself: a node calls interrupt({...}), the run suspends, the client
receives the payload, renders a UI, and resumes the run with the user's
answer. CopilotKit's useInterrupt hook is the render contract.
See the useInterrupt deep dive for
the full walkthrough, including the backend tool and render-prop wiring.
Going headless#
Both patterns above ship with a render prop — CopilotKit handles the
"when to show the picker" logic for you. If you want to drive
interrupt resolution from a custom UI that lives anywhere in the tree
(not necessarily inside a chat), see the
headless interrupts guide — it shows
how to compose useAgent, agent.subscribe, and copilotkit.runAgent
to build your own useInterrupt equivalent.