Recipes
Voice agent
A complete voice agent surface: the orb, live captions and call controls, with the agent's tool calls and transcript beside it. Style it in the Studio, pick a layout, connect your voice session.
AriaCalendar and email assistant
Connecting
Use it in your app
- 1Design the screenSet layout, captions, waveform, status, hand-off and backdrop in the Voice Call Studio, and the orb in the Voice Orb Studio. Leave orb out to use your VoiceOrbProvider.
<VoiceAgent session={session} layout="focus" captions="transcript" waveform backdrop="glow" orb={{ variant: "glass", palette: "iris", glow: 0.3, size: 180 }} /> - 2Place itSplit shows the agent's actions and transcript beside the call, for a full page. Focus is the call alone, for a dialog or a side panel's voice mode.
<VoiceAgent session={session} layout="focus" agentName="Aria" subtitle="Your workspace assistant" /> - 3Connect your sessionThe UI only reads a VoiceAgentSession. Start with the simulated one, then return the same shape from your realtime provider.
const session = useSimulatedVoiceAgent() // swap for useMyVoiceSession()
Install
pnpm dlx shadcn@latest add @jds/voice-agent
Installs the recipe and every component it uses into your project, ready to run.
Connect your backend
The UI only reads this session. Build a hook that returns it from your realtime provider, then swap it for the simulated one.
Session interface
/** Something the agent did during the call, shown as a tool call. */
type AgentAction = {
id: string
/** Tool name, e.g. "calendar.availability". */
tool: string
state: ToolState
input?: unknown
output?: unknown
}
/**
* Everything <VoiceAgent> reads from a voice session. Implement it on top of your
* realtime provider (OpenAI Realtime, ElevenLabs, Vapi, LiveKit…) and pass it in.
* `useSimulatedVoiceAgent` is a scripted stand-in so the UI runs before the backend does.
*/
type VoiceAgentSession = {
call: CallState
/** When the call connected, for the timer. */
startedAt?: number
/** Drives the orb: speaking while the agent talks, listening while the user does. */
orb: VoiceState
/** Loudness 0 to 1 of whoever is talking. */
level: number
transcript: TranscriptSegment[]
actions: AgentAction[]
/** How the call ended, shown once it's over. */
outcome?: { title: string; detail: string }
muted: boolean
setMuted: (muted: boolean) => void
/** Cut the agent off mid-sentence. */
interrupt: () => void
/** Hand the conversation to a person. */
transfer: () => void
end: () => void
restart: () => void
}Source
components/voice-agent/voice-agent.tsx
"use client"
import * as React from "react"
import { AnimatePresence, motion } from "motion/react"
import { cn } from "cn"
import { ToolCall, ToolCallContent, ToolCallHeader, ToolCallSection } from "@/components/ai/tool-call"
import { Button } from "@/components/ui/button"
import { CallControls, CallEnd, CallInterrupt, CallMute, CallStatus } from "@/components/voice/call-controls"
import { LiveTranscript } from "@/components/voice/live-transcript"
import { VoiceOrb } from "@/components/voice/voice-orb"
import { Waveform } from "@/components/voice/waveform"
import { SuccessIcon } from "@/lib/icons"
import { duration, ease } from "@/lib/motion"
import type { VoiceAgentSession } from "./session"
/** Orb styling, exactly as the Voice Orb Studio generates it. State and level come from the session. */
type VoiceAgentOrb = Omit<React.ComponentProps<typeof VoiceOrb>, "state" | "level">
/**
* A voice agent surface: the orb, a live caption and call controls, and (in the split
* layout) the agent's actions and transcript beside it. Style the orb with `orb`, pick a
* `layout`, and pass any `VoiceAgentSession`; `useSimulatedVoiceAgent` plays a sample.
*/
function VoiceAgent({
session,
orb: orbStyle,
layout = "split",
agentName = "Aria",
subtitle,
captions = "line",
waveform = false,
status = true,
handoff = true,
backdrop = "none",
className,
}: {
session: VoiceAgentSession
/** Paste the props from the Voice Orb Studio. Without it, the orb follows the nearest VoiceOrbProvider. */
orb?: VoiceAgentOrb
/** "split" shows actions and transcript beside the call; "focus" is the call alone, for dialogs and panels. */
layout?: "split" | "focus"
agentName?: string
/** A line under the agent's name, e.g. what it can help with. */
subtitle?: string
/** Under the orb: the latest line, the last few lines as a rolling transcript, or nothing. */
captions?: "line" | "transcript" | "off"
/** A live waveform under the orb, driven by the session's level. */
waveform?: boolean
/** Connection status and call timer under the name. */
status?: boolean
/** The "Hand to a person" button next to the call controls. */
handoff?: boolean
/** "glow" puts a soft light in the primary color behind the orb. */
backdrop?: "none" | "glow"
className?: string
}) {
const { call, orb, level, transcript, actions, outcome } = session
const split = layout === "split"
const latest = transcript.at(-1)
const ended = call === "ended"
const log = React.useRef<HTMLDivElement>(null)
// A speech-shaped spectrum from the single level: tallest in the middle, rippling with loudness.
const spectrum = React.useMemo(
() =>
Array.from({ length: 28 }, (_, i) => {
const center = 1 - Math.abs(i - 13.5) / 14
return Math.min(1, level * 1.6 * center * (0.55 + 0.45 * Math.abs(Math.sin(i * 1.9 + level * 9))))
}),
[level],
)
// Keep the newest line in view as the transcript grows.
React.useEffect(() => {
const el = log.current
if (el) el.scrollTo({ top: el.scrollHeight, behavior: "smooth" })
}, [latest?.text, transcript.length])
return (
<div
data-slot="voice-agent"
data-layout={layout}
className={cn(
"grid w-full overflow-hidden rounded-2xl border bg-background md:h-160",
split && "md:grid-cols-[minmax(0,1fr)_22rem]",
className,
)}
>
<section
aria-label="Call"
className={cn(
"flex min-h-128 flex-col items-center justify-between gap-6 px-6 py-8",
backdrop === "glow" && "bg-radial from-primary/15 to-transparent to-70%",
)}
>
<div className="flex flex-col items-center gap-1 text-center">
<span className="text-sm font-medium">{agentName}</span>
{subtitle && <span className="text-xs text-muted-foreground">{subtitle}</span>}
{status && <CallStatus state={call} startedAt={session.startedAt} />}
</div>
<div className="flex flex-col items-center gap-4">
<VoiceOrb size={200} {...orbStyle} state={orb} level={level} />
{waveform && <Waveform spectrum={spectrum} active={orb === "speaking" || orb === "listening"} className="w-48" />}
</div>
<div className="flex min-h-20 w-full max-w-md flex-col items-center justify-end text-center" aria-hidden>
{captions === "transcript" && !ended ? (
<LiveTranscript segments={transcript.slice(-3)} agentName={agentName} className="w-full text-left" />
) : (
<AnimatePresence mode="popLayout">
{ended && outcome ? (
<motion.div
key="outcome"
initial={{ opacity: 0, y: 8 }}
animate={{ opacity: 1, y: 0 }}
transition={{ duration: duration.base, ease: ease.out }}
className="flex flex-col items-center gap-1"
>
<span className="flex items-center gap-1.5 text-sm font-medium">
<SuccessIcon className="size-4 text-success" />
{outcome.title}
</span>
<span className="text-xs text-muted-foreground">{outcome.detail}</span>
</motion.div>
) : (
captions !== "off" &&
latest && (
<motion.p
key={latest.id}
initial={{ opacity: 0, y: 8 }}
animate={{ opacity: 1, y: 0 }}
transition={{ duration: duration.base, ease: ease.out }}
className={cn("text-base", latest.speaker === "user" && "text-muted-foreground")}
>
{latest.text}
</motion.p>
)
)}
</AnimatePresence>
)}
</div>
{ended ? (
<Button variant="outline" onClick={session.restart}>
Call again
</Button>
) : (
<div className="flex flex-wrap items-center justify-center gap-3">
<CallControls>
<CallMute pressed={session.muted} onPressedChange={session.setMuted} />
<CallInterrupt disabled={orb !== "speaking"} onClick={session.interrupt} />
<CallEnd onClick={session.end} />
</CallControls>
{handoff && (
<Button variant="outline" size="sm" onClick={session.transfer} disabled={call !== "connected"}>
Hand to a person
</Button>
)}
</div>
)}
</section>
{split && (
<aside aria-label="Agent activity" className="flex min-h-0 flex-col border-t md:border-t-0 md:border-l">
<div className="flex flex-col gap-2 border-b p-4">
<h3 className="text-xs font-medium text-muted-foreground">Actions</h3>
{actions.length === 0 ? (
<p className="text-sm text-muted-foreground">Nothing yet. Tool calls show up here as the agent works.</p>
) : (
<div className="flex flex-col gap-2">
{actions.map((a) => (
<ToolCall key={a.id}>
<ToolCallHeader name={a.tool} state={a.state} />
<ToolCallContent>
<ToolCallSection label="Input" value={a.input} />
{a.output !== undefined && <ToolCallSection label="Output" value={a.output} />}
</ToolCallContent>
</ToolCall>
))}
</div>
)}
</div>
<div ref={log} className="flex min-h-0 flex-1 flex-col gap-3 overflow-y-auto p-4">
<h3 className="text-xs font-medium text-muted-foreground">Transcript</h3>
<LiveTranscript segments={transcript} agentName={agentName} />
</div>
</aside>
)}
</div>
)
}
export { VoiceAgent, type VoiceAgentOrb }components/voice-agent/session.ts
"use client"
import * as React from "react"
import type { ToolState } from "@/components/ai/tool-call"
import type { CallState } from "@/components/voice/call-controls"
import type { TranscriptSegment } from "@/components/voice/live-transcript"
import type { VoiceState } from "@/components/voice/voice-orb"
/** Something the agent did during the call, shown as a tool call. */
type AgentAction = {
id: string
/** Tool name, e.g. "calendar.availability". */
tool: string
state: ToolState
input?: unknown
output?: unknown
}
/**
* Everything <VoiceAgent> reads from a voice session. Implement it on top of your
* realtime provider (OpenAI Realtime, ElevenLabs, Vapi, LiveKit…) and pass it in.
* `useSimulatedVoiceAgent` is a scripted stand-in so the UI runs before the backend does.
*/
type VoiceAgentSession = {
call: CallState
/** When the call connected, for the timer. */
startedAt?: number
/** Drives the orb: speaking while the agent talks, listening while the user does. */
orb: VoiceState
/** Loudness 0 to 1 of whoever is talking. */
level: number
transcript: TranscriptSegment[]
actions: AgentAction[]
/** How the call ended, shown once it's over. */
outcome?: { title: string; detail: string }
muted: boolean
setMuted: (muted: boolean) => void
/** Cut the agent off mid-sentence. */
interrupt: () => void
/** Hand the conversation to a person. */
transfer: () => void
end: () => void
restart: () => void
}
type Step =
| { say: "agent" | "user"; text: string }
| { tool: string; input: unknown; output: unknown; ms: number }
| { outcome: { title: string; detail: string } }
/** A sample conversation to show the UI's states. Replace the whole hook with your own session. */
const script: Step[] = [
{ say: "agent", text: "Hi, I'm Aria. What can I help you with?" },
{ say: "user", text: "Can you move my two o'clock with Luca to tomorrow?" },
{ tool: "calendar.find", input: { with: "Luca", day: "today" }, output: { event: "Design review", at: "2:00 PM" }, ms: 1100 },
{ tool: "calendar.availability", input: { day: "tomorrow", minutes: 30 }, output: { slots: ["10:30 AM", "1:00 PM", "4:00 PM"] }, ms: 1400 },
{ say: "agent", text: "Sure. Tomorrow you're both free at 10:30, 1, or 4. Which works?" },
{ say: "user", text: "10:30, please." },
{ tool: "calendar.move", input: { event: "Design review", to: "Tomorrow 10:30 AM" }, output: { moved: true }, ms: 1200 },
{ tool: "email.send", input: { to: "Luca", template: "meeting-moved" }, output: { sent: true }, ms: 700 },
{ say: "agent", text: "Done. It's tomorrow at 10:30, and I let Luca know. Anything else?" },
{ say: "user", text: "No, that's all. Thanks!" },
{ say: "agent", text: "Anytime." },
{ outcome: { title: "Meeting moved", detail: "Design review with Luca · tomorrow 10:30 AM · Luca notified" } },
]
const WORD_MS = 170
const PAUSE_MS = 700
/** A scripted session that plays `script`, for demos and for building the UI before the backend exists. */
function useSimulatedVoiceAgent(): VoiceAgentSession {
const [call, setCall] = React.useState<CallState>("connecting")
const [startedAt, setStartedAt] = React.useState<number>()
const [step, setStep] = React.useState(0)
const [words, setWords] = React.useState(0)
const [outcome, setOutcome] = React.useState<VoiceAgentSession["outcome"]>()
const [muted, setMuted] = React.useState(false)
const [level, setLevel] = React.useState(0)
const current = call === "connected" ? script[step] : undefined
const saying = current && "say" in current ? current : undefined
const speaking = !!saying && words < saying.text.split(" ").length
const running = current && "tool" in current
// Transcript and actions are derived from how far the script has played, so there's one source of truth.
const reached = call === "connecting" ? -1 : step
const played = script.slice(0, reached + 1)
const transcript: TranscriptSegment[] = played.flatMap((s, i) => {
if (!("say" in s)) return []
const done = i < reached || words >= s.text.split(" ").length
const text = done ? s.text : s.text.split(" ").slice(0, words).join(" ")
return text ? [{ id: String(i), speaker: s.say, text, final: done }] : []
})
const actions: AgentAction[] = played.flatMap((s, i) => {
if (!("tool" in s)) return []
const done = i < reached
// A call ended mid-tool leaves that tool unfinished.
const state: ToolState = done ? "output-available" : call === "ended" ? "output-error" : "input-available"
return [{ id: String(i), tool: s.tool, state, input: s.input, output: done ? s.output : undefined }]
})
const orb: VoiceState =
call === "connecting" || call === "reconnecting"
? "connecting"
: call === "ended"
? "idle"
: running
? "thinking"
: saying?.say === "agent" && speaking
? "speaking"
: "listening"
// Connect.
React.useEffect(() => {
if (call !== "connecting") return
const id = setTimeout(() => {
setCall("connected")
setStartedAt(Date.now())
}, 1200)
return () => clearTimeout(id)
}, [call])
// Advance the script: stream words, run tools, then settle the outcome.
React.useEffect(() => {
if (!current) return
if ("outcome" in current) {
const id = setTimeout(() => {
setOutcome(current.outcome)
setCall("ended")
}, PAUSE_MS)
return () => clearTimeout(id)
}
if ("tool" in current) {
const id = setTimeout(() => setStep((s) => s + 1), current.ms)
return () => clearTimeout(id)
}
const total = current.text.split(" ").length
const id = setTimeout(
() => {
if (words < total) setWords((w) => w + 1)
else {
setStep((s) => s + 1)
setWords(0)
}
},
words < total ? WORD_MS : PAUSE_MS,
)
return () => clearTimeout(id)
}, [current, words])
// A speech-like level while someone talks; the user's side goes quiet when muted.
React.useEffect(() => {
const talking = speaking && !(saying?.say === "user" && muted)
if (!talking) {
const id = requestAnimationFrame(() => setLevel(0))
return () => cancelAnimationFrame(id)
}
let raf = 0
const tick = (t: number) => {
setLevel(Math.max(0, 0.45 + Math.sin(t / 110) * 0.25 + Math.sin(t / 47) * 0.15))
raf = requestAnimationFrame(tick)
}
raf = requestAnimationFrame(tick)
return () => cancelAnimationFrame(raf)
}, [speaking, saying, muted])
const reset = () => {
setStep(0)
setWords(0)
setOutcome(undefined)
setStartedAt(undefined)
}
return {
call,
startedAt,
orb,
level,
transcript,
actions,
outcome,
muted,
setMuted,
interrupt: () => {
if (saying?.say === "agent") setWords(saying.text.split(" ").length)
},
transfer: () => {
setOutcome({ title: "Handed to a person", detail: "They get the transcript and the actions so far." })
setCall("ended")
},
end: () => setCall("ended"),
restart: () => {
reset()
setCall("connecting")
},
}
}
export { useSimulatedVoiceAgent, type AgentAction, type VoiceAgentSession }