JDS.

Search docs

Find a page or component

Recipes

Voice session

Push to talk with live mic level, orb states, and a rolling transcript.

Hold the button or Space

Transcriptidle

Speak to the scheduling agent. Mic access is optional: the demo simulates input if you decline.

Install

pnpm dlx shadcn@latest add @jds/voice-session

Installs the recipe and every component it uses into your project, ready to run.

Source

components/voice-session/voice-session.tsx
"use client"

import * as React from "react"

import { Kbd } from "@/components/ui/kbd"
import { LiveTranscript, type TranscriptSegment } from "@/components/voice/live-transcript"
import { PushToTalk } from "@/components/voice/push-to-talk"
import { VoiceOrb, type VoiceState } from "@/components/voice/voice-orb"
import { Waveform } from "@/components/voice/waveform"
import { useAudioLevel } from "@/hooks/use-audio-level"
import { useSimulatedSpectrum } from "@/hooks/use-simulated-spectrum"

const USER_LINE = "Move my 3pm with Luca to tomorrow morning."
const AGENT_LINE = "Done. You're meeting Luca tomorrow at 9:30. I've sent him the updated invite."
const BANDS = 28

export default function VoiceSession() {
  const mic = useAudioLevel({ bands: BANDS })
  const [state, setState] = React.useState<VoiceState>("idle")
  const [segments, setSegments] = React.useState<TranscriptSegment[]>([])
  const timers = React.useRef<ReturnType<typeof setTimeout>[]>([])
  const typing = React.useRef<ReturnType<typeof setInterval> | null>(null)

  const simulate = state === "speaking" || (state === "listening" && !mic.active)
  const { spectrum: synthetic } = useSimulatedSpectrum(simulate, BANDS)
  const spectrum = state === "listening" && mic.active ? mic.spectrum : synthetic
  const level = Math.min(1, (spectrum.reduce((a, b) => a + b, 0) / BANDS) * 2.2)

  const clear = () => {
    timers.current.forEach(clearTimeout)
    timers.current = []
    if (typing.current) clearInterval(typing.current)
  }

  const typeInto = (id: string, text: string, speed: number, onDone?: () => void) => {
    let i = 0
    typing.current = setInterval(() => {
      i += 1
      const words = text.split(" ")
      setSegments((prev) =>
        prev.map((s) => (s.id === id ? { ...s, text: words.slice(0, i).join(" "), final: i >= words.length } : s))
      )
      if (i >= words.length) {
        if (typing.current) clearInterval(typing.current)
        onDone?.()
      }
    }, speed)
  }

  const onPressStart = () => {
    clear()
    void mic.start()
    setState("listening")
    const id = `u-${Date.now()}`
    setSegments([{ id, speaker: "user", text: "", final: false }])
    typeInto(id, USER_LINE, 200)
  }

  const onPressEnd = () => {
    mic.stop()
    if (typing.current) clearInterval(typing.current)
    setSegments((prev) => prev.map((s) => (s.speaker === "user" ? { ...s, text: USER_LINE, final: true } : s)))
    setState("thinking")
    timers.current.push(
      setTimeout(() => {
        setState("speaking")
        const id = `a-${Date.now()}`
        setSegments((prev) => [...prev, { id, speaker: "agent", text: "", final: false }])
        typeInto(id, AGENT_LINE, 170, () => timers.current.push(setTimeout(() => setState("idle"), 400)))
      }, 1400)
    )
  }

  React.useEffect(() => clear, [])

  return (
    <div className="grid w-full overflow-hidden rounded-2xl border bg-card md:grid-cols-2">
      <div className="relative flex flex-col items-center justify-center gap-8 border-b p-10 md:border-r md:border-b-0">
        <VoiceOrb state={state} level={level} size={180} />
        <Waveform spectrum={spectrum} active={state === "listening" || state === "speaking"} className="w-56" />
        <div className="flex flex-col items-center gap-3">
          <PushToTalk onPressStart={onPressStart} onPressEnd={onPressEnd} level={level} />
          <p className="flex items-center gap-1.5 text-xs text-muted-foreground">
            Hold the button or <Kbd>Space</Kbd>
          </p>
        </div>
      </div>
      <div className="flex min-h-80 flex-col gap-4 p-6">
        <div className="flex items-center justify-between">
          <span className="text-xs font-medium tracking-wide text-muted-foreground uppercase">Transcript</span>
          <span className="font-mono text-xs text-foreground">{state}</span>
        </div>
        {segments.length === 0 ? (
          <p className="my-auto text-center text-sm text-muted-foreground">
            Speak to the scheduling agent. Mic access is optional: the demo simulates input if you decline.
          </p>
        ) : (
          <LiveTranscript segments={segments} agentName="Agent" />
        )}
      </div>
    </div>
  )
}
hooks/use-simulated-spectrum.ts
"use client"

import * as React from "react"

/** Speech-like spectrum for agent playback, demos, and users who decline mic access. */
export function useSimulatedSpectrum(active: boolean, bands = 24) {
  const [spectrum, setSpectrum] = React.useState<number[]>(() => Array(bands).fill(0))

  React.useEffect(() => {
    if (!active) return
    let raf = 0
    const tick = (t: number) => {
      setSpectrum(
        Array.from({ length: bands }, (_, i) => {
          const center = 1 - Math.abs(i - bands / 2) / (bands / 2)
          const wave = (Math.sin(t / 140 + i * 0.7) + Math.sin(t / 90 + i * 1.3)) / 4 + 0.5
          return Math.max(0, center * wave * (0.6 + Math.random() * 0.4))
        })
      )
      raf = requestAnimationFrame(tick)
    }
    raf = requestAnimationFrame(tick)
    return () => {
      cancelAnimationFrame(raf)
      setSpectrum(Array(bands).fill(0))
    }
  }, [active, bands])

  const level = Math.min(1, (spectrum.reduce((a, b) => a + b, 0) / bands) * 2.2)
  return { spectrum, level }
}