captions.js

Word-by-word captions in React and Next.js

Add TikTok-style animated captions over a video in a React or Next.js app: a small component around captions.js, with live preset switching.

captions.js draws captions on a canvas layered over your <video>. In React that is one effect: create the overlay when the video mounts, update it when props change, destroy it on unmount.

Install

npm install captions.js

A <CaptionedVideo> component

"use client"; // Next.js App Router: the overlay needs the DOM

import { useEffect, useRef } from "react";
import type { Caption, CaptionsInstance, StylePresetName } from "captions.js";

type Props = {
  src: string;
  captions: Caption[];
  preset?: StylePresetName;
};

export function CaptionedVideo({ src, captions, preset = "Karaoke" }: Props) {
  const videoRef = useRef<HTMLVideoElement>(null);
  const instanceRef = useRef<CaptionsInstance | null>(null);

  // Create once. The import runs in the browser only, so SSR never loads canvas code.
  useEffect(() => {
    let cancelled = false;
    import("captions.js").then(({ default: captionsjs, getPreset }) => {
      if (cancelled || !videoRef.current) return;
      instanceRef.current = captionsjs({
        video: videoRef.current,
        preset: getPreset(preset),
        captions,
      });
    });
    return () => {
      cancelled = true;
      instanceRef.current?.destroy();
      instanceRef.current = null;
    };
    // eslint-disable-next-line react-hooks/exhaustive-deps
  }, []);

  // Live updates, no remount.
  useEffect(() => {
    import("captions.js").then(({ getPreset }) => instanceRef.current?.preset(getPreset(preset)));
  }, [preset]);

  useEffect(() => {
    instanceRef.current?.captions(captions);
  }, [captions]);

  return (
    <div style={{ position: "relative" }}>
      <video ref={videoRef} src={src} controls playsInline style={{ width: "100%" }} />
    </div>
  );
}

Use it with word timings from any speech-to-text service:

<CaptionedVideo src="/talk.mp4" captions={words} preset="Focus Box" />

words is an array of { word, startTime, endTime } in seconds. For a Deepgram response call toCaptions(response); for Whisper see Animated captions from Whisper.

Notes

  • Positioning: the overlay sits on top of the video and resizes with it. Keep the <video> inside a positioned wrapper as above.
  • Style picker: presetNames lists every preset, handy for a <select>. Each name has a preview in the styles gallery.
  • Same look on export: the server package renders with the same renderFrame, so what users pick in your UI is exactly what burnCaptions() writes to the MP4. See Animated captions with FFmpeg in Node.js.

For AI agents and LLMs: this page as Markdown · llms.txt · all docs in one file