JDS.

Search docs

Find a page or component

Recipes

Voice agent

A complete voice agent surface: the orb, live captions and call controls, with the agent's tool calls and transcript beside it. Style it in the Studio, pick a layout, connect your voice session.

AriaCalendar and email assistant
Connecting

Use it in your app

  1. 1
    Design the screenSet layout, captions, waveform, status, hand-off and backdrop in the Voice Call Studio, and the orb in the Voice Orb Studio. Leave orb out to use your VoiceOrbProvider.
    <VoiceAgent
      session={session}
      layout="focus"
      captions="transcript"
      waveform
      backdrop="glow"
      orb={{ variant: "glass", palette: "iris", glow: 0.3, size: 180 }}
    />
  2. 2
    Place itSplit shows the agent's actions and transcript beside the call, for a full page. Focus is the call alone, for a dialog or a side panel's voice mode.
    <VoiceAgent session={session} layout="focus" agentName="Aria" subtitle="Your workspace assistant" />
  3. 3
    Connect your sessionThe UI only reads a VoiceAgentSession. Start with the simulated one, then return the same shape from your realtime provider.
    const session = useSimulatedVoiceAgent() // swap for useMyVoiceSession()
Open the Voice Orb Studio

Install

pnpm dlx shadcn@latest add @jds/voice-agent

Installs the recipe and every component it uses into your project, ready to run.

Connect your backend

The UI only reads this session. Build a hook that returns it from your realtime provider, then swap it for the simulated one.

Session interface
/** Something the agent did during the call, shown as a tool call. */
type AgentAction = {
  id: string
  /** Tool name, e.g. "calendar.availability". */
  tool: string
  state: ToolState
  input?: unknown
  output?: unknown
}

/**
 * Everything <VoiceAgent> reads from a voice session. Implement it on top of your
 * realtime provider (OpenAI Realtime, ElevenLabs, Vapi, LiveKit…) and pass it in.
 * `useSimulatedVoiceAgent` is a scripted stand-in so the UI runs before the backend does.
 */
type VoiceAgentSession = {
  call: CallState
  /** When the call connected, for the timer. */
  startedAt?: number
  /** Drives the orb: speaking while the agent talks, listening while the user does. */
  orb: VoiceState
  /** Loudness 0 to 1 of whoever is talking. */
  level: number
  transcript: TranscriptSegment[]
  actions: AgentAction[]
  /** How the call ended, shown once it's over. */
  outcome?: { title: string; detail: string }
  muted: boolean
  setMuted: (muted: boolean) => void
  /** Cut the agent off mid-sentence. */
  interrupt: () => void
  /** Hand the conversation to a person. */
  transfer: () => void
  end: () => void
  restart: () => void
}

Source

components/voice-agent/voice-agent.tsx
"use client"

import * as React from "react"
import { AnimatePresence, motion } from "motion/react"
import { cn } from "cn"

import { ToolCall, ToolCallContent, ToolCallHeader, ToolCallSection } from "@/components/ai/tool-call"
import { Button } from "@/components/ui/button"
import { CallControls, CallEnd, CallInterrupt, CallMute, CallStatus } from "@/components/voice/call-controls"
import { LiveTranscript } from "@/components/voice/live-transcript"
import { VoiceOrb } from "@/components/voice/voice-orb"
import { Waveform } from "@/components/voice/waveform"
import { SuccessIcon } from "@/lib/icons"
import { duration, ease } from "@/lib/motion"

import type { VoiceAgentSession } from "./session"

/** Orb styling, exactly as the Voice Orb Studio generates it. State and level come from the session. */
type VoiceAgentOrb = Omit<React.ComponentProps<typeof VoiceOrb>, "state" | "level">

/**
 * A voice agent surface: the orb, a live caption and call controls, and (in the split
 * layout) the agent's actions and transcript beside it. Style the orb with `orb`, pick a
 * `layout`, and pass any `VoiceAgentSession`; `useSimulatedVoiceAgent` plays a sample.
 */
function VoiceAgent({
  session,
  orb: orbStyle,
  layout = "split",
  agentName = "Aria",
  subtitle,
  captions = "line",
  waveform = false,
  status = true,
  handoff = true,
  backdrop = "none",
  className,
}: {
  session: VoiceAgentSession
  /** Paste the props from the Voice Orb Studio. Without it, the orb follows the nearest VoiceOrbProvider. */
  orb?: VoiceAgentOrb
  /** "split" shows actions and transcript beside the call; "focus" is the call alone, for dialogs and panels. */
  layout?: "split" | "focus"
  agentName?: string
  /** A line under the agent's name, e.g. what it can help with. */
  subtitle?: string
  /** Under the orb: the latest line, the last few lines as a rolling transcript, or nothing. */
  captions?: "line" | "transcript" | "off"
  /** A live waveform under the orb, driven by the session's level. */
  waveform?: boolean
  /** Connection status and call timer under the name. */
  status?: boolean
  /** The "Hand to a person" button next to the call controls. */
  handoff?: boolean
  /** "glow" puts a soft light in the primary color behind the orb. */
  backdrop?: "none" | "glow"
  className?: string
}) {
  const { call, orb, level, transcript, actions, outcome } = session
  const split = layout === "split"
  const latest = transcript.at(-1)
  const ended = call === "ended"
  const log = React.useRef<HTMLDivElement>(null)
  // A speech-shaped spectrum from the single level: tallest in the middle, rippling with loudness.
  const spectrum = React.useMemo(
    () =>
      Array.from({ length: 28 }, (_, i) => {
        const center = 1 - Math.abs(i - 13.5) / 14
        return Math.min(1, level * 1.6 * center * (0.55 + 0.45 * Math.abs(Math.sin(i * 1.9 + level * 9))))
      }),
    [level],
  )

  // Keep the newest line in view as the transcript grows.
  React.useEffect(() => {
    const el = log.current
    if (el) el.scrollTo({ top: el.scrollHeight, behavior: "smooth" })
  }, [latest?.text, transcript.length])

  return (
    <div
      data-slot="voice-agent"
      data-layout={layout}
      className={cn(
        "grid w-full overflow-hidden rounded-2xl border bg-background md:h-160",
        split && "md:grid-cols-[minmax(0,1fr)_22rem]",
        className,
      )}
    >
      <section
        aria-label="Call"
        className={cn(
          "flex min-h-128 flex-col items-center justify-between gap-6 px-6 py-8",
          backdrop === "glow" && "bg-radial from-primary/15 to-transparent to-70%",
        )}
      >
        <div className="flex flex-col items-center gap-1 text-center">
          <span className="text-sm font-medium">{agentName}</span>
          {subtitle && <span className="text-xs text-muted-foreground">{subtitle}</span>}
          {status && <CallStatus state={call} startedAt={session.startedAt} />}
        </div>

        <div className="flex flex-col items-center gap-4">
          <VoiceOrb size={200} {...orbStyle} state={orb} level={level} />
          {waveform && <Waveform spectrum={spectrum} active={orb === "speaking" || orb === "listening"} className="w-48" />}
        </div>

        <div className="flex min-h-20 w-full max-w-md flex-col items-center justify-end text-center" aria-hidden>
          {captions === "transcript" && !ended ? (
            <LiveTranscript segments={transcript.slice(-3)} agentName={agentName} className="w-full text-left" />
          ) : (
            <AnimatePresence mode="popLayout">
              {ended && outcome ? (
                <motion.div
                  key="outcome"
                  initial={{ opacity: 0, y: 8 }}
                  animate={{ opacity: 1, y: 0 }}
                  transition={{ duration: duration.base, ease: ease.out }}
                  className="flex flex-col items-center gap-1"
                >
                  <span className="flex items-center gap-1.5 text-sm font-medium">
                    <SuccessIcon className="size-4 text-success" />
                    {outcome.title}
                  </span>
                  <span className="text-xs text-muted-foreground">{outcome.detail}</span>
                </motion.div>
              ) : (
                captions !== "off" &&
                latest && (
                  <motion.p
                    key={latest.id}
                    initial={{ opacity: 0, y: 8 }}
                    animate={{ opacity: 1, y: 0 }}
                    transition={{ duration: duration.base, ease: ease.out }}
                    className={cn("text-base", latest.speaker === "user" && "text-muted-foreground")}
                  >
                    {latest.text}
                  </motion.p>
                )
              )}
            </AnimatePresence>
          )}
        </div>

        {ended ? (
          <Button variant="outline" onClick={session.restart}>
            Call again
          </Button>
        ) : (
          <div className="flex flex-wrap items-center justify-center gap-3">
            <CallControls>
              <CallMute pressed={session.muted} onPressedChange={session.setMuted} />
              <CallInterrupt disabled={orb !== "speaking"} onClick={session.interrupt} />
              <CallEnd onClick={session.end} />
            </CallControls>
            {handoff && (
              <Button variant="outline" size="sm" onClick={session.transfer} disabled={call !== "connected"}>
                Hand to a person
              </Button>
            )}
          </div>
        )}
      </section>

      {split && (
      <aside aria-label="Agent activity" className="flex min-h-0 flex-col border-t md:border-t-0 md:border-l">
        <div className="flex flex-col gap-2 border-b p-4">
          <h3 className="text-xs font-medium text-muted-foreground">Actions</h3>
          {actions.length === 0 ? (
            <p className="text-sm text-muted-foreground">Nothing yet. Tool calls show up here as the agent works.</p>
          ) : (
            <div className="flex flex-col gap-2">
              {actions.map((a) => (
                <ToolCall key={a.id}>
                  <ToolCallHeader name={a.tool} state={a.state} />
                  <ToolCallContent>
                    <ToolCallSection label="Input" value={a.input} />
                    {a.output !== undefined && <ToolCallSection label="Output" value={a.output} />}
                  </ToolCallContent>
                </ToolCall>
              ))}
            </div>
          )}
        </div>
        <div ref={log} className="flex min-h-0 flex-1 flex-col gap-3 overflow-y-auto p-4">
          <h3 className="text-xs font-medium text-muted-foreground">Transcript</h3>
          <LiveTranscript segments={transcript} agentName={agentName} />
        </div>
      </aside>
      )}
    </div>
  )
}

export { VoiceAgent, type VoiceAgentOrb }
components/voice-agent/session.ts
"use client"

import * as React from "react"

import type { ToolState } from "@/components/ai/tool-call"
import type { CallState } from "@/components/voice/call-controls"
import type { TranscriptSegment } from "@/components/voice/live-transcript"
import type { VoiceState } from "@/components/voice/voice-orb"

/** Something the agent did during the call, shown as a tool call. */
type AgentAction = {
  id: string
  /** Tool name, e.g. "calendar.availability". */
  tool: string
  state: ToolState
  input?: unknown
  output?: unknown
}

/**
 * Everything <VoiceAgent> reads from a voice session. Implement it on top of your
 * realtime provider (OpenAI Realtime, ElevenLabs, Vapi, LiveKit…) and pass it in.
 * `useSimulatedVoiceAgent` is a scripted stand-in so the UI runs before the backend does.
 */
type VoiceAgentSession = {
  call: CallState
  /** When the call connected, for the timer. */
  startedAt?: number
  /** Drives the orb: speaking while the agent talks, listening while the user does. */
  orb: VoiceState
  /** Loudness 0 to 1 of whoever is talking. */
  level: number
  transcript: TranscriptSegment[]
  actions: AgentAction[]
  /** How the call ended, shown once it's over. */
  outcome?: { title: string; detail: string }
  muted: boolean
  setMuted: (muted: boolean) => void
  /** Cut the agent off mid-sentence. */
  interrupt: () => void
  /** Hand the conversation to a person. */
  transfer: () => void
  end: () => void
  restart: () => void
}

type Step =
  | { say: "agent" | "user"; text: string }
  | { tool: string; input: unknown; output: unknown; ms: number }
  | { outcome: { title: string; detail: string } }

/** A sample conversation to show the UI's states. Replace the whole hook with your own session. */
const script: Step[] = [
  { say: "agent", text: "Hi, I'm Aria. What can I help you with?" },
  { say: "user", text: "Can you move my two o'clock with Luca to tomorrow?" },
  { tool: "calendar.find", input: { with: "Luca", day: "today" }, output: { event: "Design review", at: "2:00 PM" }, ms: 1100 },
  { tool: "calendar.availability", input: { day: "tomorrow", minutes: 30 }, output: { slots: ["10:30 AM", "1:00 PM", "4:00 PM"] }, ms: 1400 },
  { say: "agent", text: "Sure. Tomorrow you're both free at 10:30, 1, or 4. Which works?" },
  { say: "user", text: "10:30, please." },
  { tool: "calendar.move", input: { event: "Design review", to: "Tomorrow 10:30 AM" }, output: { moved: true }, ms: 1200 },
  { tool: "email.send", input: { to: "Luca", template: "meeting-moved" }, output: { sent: true }, ms: 700 },
  { say: "agent", text: "Done. It's tomorrow at 10:30, and I let Luca know. Anything else?" },
  { say: "user", text: "No, that's all. Thanks!" },
  { say: "agent", text: "Anytime." },
  { outcome: { title: "Meeting moved", detail: "Design review with Luca · tomorrow 10:30 AM · Luca notified" } },
]

const WORD_MS = 170
const PAUSE_MS = 700

/** A scripted session that plays `script`, for demos and for building the UI before the backend exists. */
function useSimulatedVoiceAgent(): VoiceAgentSession {
  const [call, setCall] = React.useState<CallState>("connecting")
  const [startedAt, setStartedAt] = React.useState<number>()
  const [step, setStep] = React.useState(0)
  const [words, setWords] = React.useState(0)
  const [outcome, setOutcome] = React.useState<VoiceAgentSession["outcome"]>()
  const [muted, setMuted] = React.useState(false)
  const [level, setLevel] = React.useState(0)

  const current = call === "connected" ? script[step] : undefined
  const saying = current && "say" in current ? current : undefined
  const speaking = !!saying && words < saying.text.split(" ").length
  const running = current && "tool" in current

  // Transcript and actions are derived from how far the script has played, so there's one source of truth.
  const reached = call === "connecting" ? -1 : step
  const played = script.slice(0, reached + 1)
  const transcript: TranscriptSegment[] = played.flatMap((s, i) => {
    if (!("say" in s)) return []
    const done = i < reached || words >= s.text.split(" ").length
    const text = done ? s.text : s.text.split(" ").slice(0, words).join(" ")
    return text ? [{ id: String(i), speaker: s.say, text, final: done }] : []
  })
  const actions: AgentAction[] = played.flatMap((s, i) => {
    if (!("tool" in s)) return []
    const done = i < reached
    // A call ended mid-tool leaves that tool unfinished.
    const state: ToolState = done ? "output-available" : call === "ended" ? "output-error" : "input-available"
    return [{ id: String(i), tool: s.tool, state, input: s.input, output: done ? s.output : undefined }]
  })

  const orb: VoiceState =
    call === "connecting" || call === "reconnecting"
      ? "connecting"
      : call === "ended"
        ? "idle"
        : running
          ? "thinking"
          : saying?.say === "agent" && speaking
            ? "speaking"
            : "listening"

  // Connect.
  React.useEffect(() => {
    if (call !== "connecting") return
    const id = setTimeout(() => {
      setCall("connected")
      setStartedAt(Date.now())
    }, 1200)
    return () => clearTimeout(id)
  }, [call])

  // Advance the script: stream words, run tools, then settle the outcome.
  React.useEffect(() => {
    if (!current) return
    if ("outcome" in current) {
      const id = setTimeout(() => {
        setOutcome(current.outcome)
        setCall("ended")
      }, PAUSE_MS)
      return () => clearTimeout(id)
    }
    if ("tool" in current) {
      const id = setTimeout(() => setStep((s) => s + 1), current.ms)
      return () => clearTimeout(id)
    }
    const total = current.text.split(" ").length
    const id = setTimeout(
      () => {
        if (words < total) setWords((w) => w + 1)
        else {
          setStep((s) => s + 1)
          setWords(0)
        }
      },
      words < total ? WORD_MS : PAUSE_MS,
    )
    return () => clearTimeout(id)
  }, [current, words])

  // A speech-like level while someone talks; the user's side goes quiet when muted.
  React.useEffect(() => {
    const talking = speaking && !(saying?.say === "user" && muted)
    if (!talking) {
      const id = requestAnimationFrame(() => setLevel(0))
      return () => cancelAnimationFrame(id)
    }
    let raf = 0
    const tick = (t: number) => {
      setLevel(Math.max(0, 0.45 + Math.sin(t / 110) * 0.25 + Math.sin(t / 47) * 0.15))
      raf = requestAnimationFrame(tick)
    }
    raf = requestAnimationFrame(tick)
    return () => cancelAnimationFrame(raf)
  }, [speaking, saying, muted])

  const reset = () => {
    setStep(0)
    setWords(0)
    setOutcome(undefined)
    setStartedAt(undefined)
  }

  return {
    call,
    startedAt,
    orb,
    level,
    transcript,
    actions,
    outcome,
    muted,
    setMuted,
    interrupt: () => {
      if (saying?.say === "agent") setWords(saying.text.split(" ").length)
    },
    transfer: () => {
      setOutcome({ title: "Handed to a person", detail: "They get the transcript and the actions so far." })
      setCall("ended")
    },
    end: () => setCall("ended"),
    restart: () => {
      reset()
      setCall("connecting")
    },
  }
}

export { useSimulatedVoiceAgent, type AgentAction, type VoiceAgentSession }