Skip to content

Voice

The film's code
import manimgx as m


class VoiceHero(m.Scene):
    def construct(self) -> None:
        square = m.MathTex("x^2", font_size=96).shift(2 * m.LEFT)
        derivative = m.MathTex("2x", font_size=96).shift(2 * m.RIGHT)
        arrow = m.Arrow(square.get_right(), derivative.get_left(), buff=0.4)
        script = m.Text(
            "The derivative of [x squared] is [two x].", font_size=32
        ).to_edge(m.DOWN)
        self.add(square, script)
        self.play(m.Indicate(square))
        self.play(m.GrowArrow(arrow), m.Write(derivative))
        self.wait()

say speaks a line with the scene's voice, and plays animations while it speaks. Put the words of each animation in brackets, and give the animations in the same order: each starts with the first of its words. The scene waits until the line ends, so lines don't overlap; change the words, and the animations follow the new words.

The scene's voice speaks through a text-to-speech service. manimgx keeps what it says in a folder named voice/, next to the scene's file, so the scene renders again without the service, and only a changed line goes to it again. What the voice says becomes the video's captions.

say

Code
import manimgx as m


class SayExample(m.Scene):
    def construct(self) -> None:
        x2 = m.MathTex("x^2").shift(2 * m.LEFT)
        dx = m.MathTex("2x").shift(2 * m.RIGHT)
        self.add(x2)
        self.say(
            "The derivative of [x squared] is [two x].",
            m.Indicate(x2),
            m.Write(dx),
        )

Speak a script, playing animations as it is said; the scene waits for both.

Mark in brackets the words an animation belongs to: the animations given play, in order, during the bracketed words, each starting as its words start and lasting as long as they take to say, or its own run time if that is longer (an empty [] starts one where it stands). Without brackets, the animations play during the whole speech. The brackets are not spoken.

self.say(script, *animations, voice=None)
script

What to say, with a bracketed span for each animation (or none).

*animations

The animations, one per span.

voice

Who says it; by default the scene's voice.

Returns The speech, with its words and when each is said.

Source

src/manimgx/scene.py

def say(
    self, script: str, *animations: Animation, voice: Voice | None = None
) -> Speech:
    """Speak a script, playing animations as it is said; the scene waits for both.

    Mark in brackets the words an animation belongs to: the animations given play,
    in order, during the bracketed words, each starting as its words start and
    lasting as long as they take to say, or its own run time if that is longer (an
    empty `[]` starts one where it stands). Without brackets, the animations play
    during the whole speech. The brackets are not spoken.

    Args:
        script: What to say, with a bracketed span for each animation (or none).
        *animations: The animations, one per span.
        voice: Who says it; by default the scene's [`voice`][manimgx.Scene.voice].

    Returns:
        The speech, with its words and when each is said.

    Examples:
        ```python
        import manimgx as m


        class SayExample(m.Scene):
            def construct(self) -> None:
                x2 = m.MathTex("x^2").shift(2 * m.LEFT)
                dx = m.MathTex("2x").shift(2 * m.RIGHT)
                self.add(x2)
                self.say(
                    "The derivative of [x squared] is [two x].",
                    m.Indicate(x2),
                    m.Write(dx),
                )
        ```
    """
    from manimgx.animation.timeline import AnimationGroup, Succession, Wait
    from manimgx.audio import spans, times

    text, found = spans(script)
    if found and len(found) != len(animations):
        raise ValueError(
            f"{len(found)} bracketed spans for {len(animations)} animations: give"
            " one animation per span"
        )
    speech = self.speech(text, voice)
    parts: list[Animation] = [speech]
    for anim, span in zip(
        animations, found or [None] * len(animations), strict=True
    ):
        start, end = (0.0, speech.duration) if span is None else times(speech, span)
        anim.run_time = max(anim.run_time, end - start)
        parts.append(Succession(Wait(start), anim) if start > 0 else anim)
    self.play(AnimationGroup(*parts))
    return speech

speech

What the scene's voice says for text, without playing it.

Spoken once, then kept in voice/ beside the scene's file (named by the text), so the scene renders again without speaking again; replace a file there with your own recording of the same words, and it is used instead. Play the speech, add it (add_sound: it speaks while the scene goes on), or read its words.

self.speech(text, voice=None)
text

What to say.

voice

Who says it; by default the scene's voice.

Returns The speech.

Source

src/manimgx/scene.py

def speech(self, text: str, voice: Voice | None = None) -> Speech:
    """What the scene's voice says for `text`, without playing it.

    Spoken once, then kept in `voice/` beside the scene's file (named by the text), so
    the scene renders again without speaking again; replace a file there with your
    own recording of the same words, and it is used instead. Play the speech, add it
    ([`add_sound`][manimgx.Scene.add_sound]: it speaks while the scene goes on), or
    read its [`words`][manimgx.Speech.words].

    Args:
        text: What to say.
        voice: Who says it; by default the scene's [`voice`][manimgx.Scene.voice].

    Returns:
        The speech.
    """
    from manimgx.audio import Fal, cached

    # the scene's voice, as given: a function set on the class is not a method of it
    own = self.__dict__.get("voice", inspect.getattr_static(type(self), "voice"))
    if isinstance(own, staticmethod):
        own = own.__func__
    return cached(voice or own or Fal(), text, _voice_folder(self))

voice

Who speaks the scene's say: any voice; None for fal.ai's default model (Fal()).

Speech

Speech: a sound that knows what it says and when it says each word.

A voice returns one. It plays as any sound does; Scene.say also times animations by its words.

m.Speech(source, *, text, words=(), rate=None)
source

The audio: a file's path, a file's bytes, or samples.

text

What it says.

words

Each word and when it is said, if the voice knows; otherwise they are estimated from the audio.

rate

The samples' rate; only for samples.

Source

src/manimgx/audio/sound.py

def __init__(
    self,
    source: Source,
    *,
    text: str,
    words: Sequence[Word] = (),
    rate: int | None = None,
) -> None:
    super().__init__(source, rate=rate)
    self.text = text
    self._words = tuple(words)  # the source's: before trims and speed

words

Its words, in order, with when each is said (in seconds from its start): the voice's, or estimated from the audio; a trim or a change of speed moves them with the audio.

speech.words
Source

src/manimgx/audio/sound.py

def words(self) -> tuple[Word, ...]:
    """Its words, in order, with when each is said (in seconds from its start): the
    voice's, or estimated from the audio; a trim or a change of speed moves them with
    the audio."""
    if not self._words:
        self._words = estimate(self.text, _source_samples(self))
    e = self._edit
    if e.start == 0 and e.speed == 1 and e.end is None:
        return self._words
    end = math.inf if e.end is None else e.end
    return tuple(
        Word(w.text, (w.start - e.start) / e.speed, (w.end - e.start) / e.speed)
        for w in self._words
        if e.start <= w.start < end
    )

Voices

The voices manimgx brings are fal.ai's: m.voices.Fal, with the model you choose and its settings. Your editor shows each model's settings.

Fal

A scene's voice, in its class:

voice = m.voices.Fal(
    "fal-ai/minimax/speech-2.8-hd",
    voice_setting={"voice_id": "Calm_Woman", "speed": 1.1},
)

A text-to-speech model on fal.ai, as a voice: Fal(model, **settings), where the settings are the model's own (ElevenV3 for the default).

m.voices.Fal(model='fal-ai/elevenlabs/tts/eleven-v3', /, *, key=None, **settings)
model

The model's ID on fal.

key

fal's API key; by default FAL_KEY.

The model's settings.

Source

src/manimgx/audio/fal.py

def __init__(
    self,
    model: Model = "fal-ai/elevenlabs/tts/eleven-v3",
    /,
    *,
    key: str | None = None,
    **settings: object,
) -> None:
    spec = _MODELS.get(model)
    if spec is None:
        raise ValueError(f"fal: no model {model!r} here: {', '.join(_MODELS)}")
    known = spec.settings.__annotations__
    if unknown := sorted(settings.keys() - known.keys()):
        raise TypeError(
            f"{model} takes {', '.join(known)}; not {', '.join(unknown)}"
        )
    object.__setattr__(self, "model", model)
    object.__setattr__(
        self, "settings", {k: _ordered(v) for k, v in sorted(settings.items())}
    )
    object.__setattr__(self, "key", key)

model

The model's ID on fal.

settings

The model's settings, as given.

key

fal's API key, if given; else FAL_KEY is read as the voice speaks.

Models

ElevenV3

The settings of ElevenLabs' Eleven v3 (fal-ai/elevenlabs/tts/eleven-v3), which times its words. Its text may carry audio tags: [whispers], [laughs], [excited]…

voice

A voice's name or ID in ElevenLabs' library: Aria, Roger, Sarah, Laura, Charlie, George, Callum, River, Liam, Charlotte, Alice, Matilda, Will, Jessica, Eric, Chris, Brian, Daniel, Lily, Bill… (default "Rachel").

stability

How steady the voice is, from 0 (expressive) to 1 (steady) (default 0.5).

language_code

The language, as an ISO 639-1 code ("en", "de"…) (default: the text's).

apply_text_normalization

Whether numbers and the like are spelled out before they are said (default "auto": as the model sees fit).

voice

A voice's name or ID in ElevenLabs' library: Aria, Roger, Sarah, Laura, Charlie, George, Callum, River, Liam, Charlotte, Alice, Matilda, Will, Jessica, Eric, Chris, Brian, Daniel, Lily, Bill… (default "Rachel").

stability

How steady the voice is, from 0 (expressive) to 1 (steady) (default 0.5).

language_code

The language, as an ISO 639-1 code ("en", "de"…) (default: the text's).

apply_text_normalization

Whether numbers and the like are spelled out before they are said (default "auto": as the model sees fit).

MiniMax

The settings of MiniMax Speech 2.8 HD (fal-ai/minimax/speech-2.8-hd). Its text may carry pauses, <#0.5#> (in seconds), and interjections: (laughs), (sighs), (coughs), (clears throat), (gasps), (sniffs), (groans), (yawns).

voice_setting

The voice and how it speaks (MiniMaxVoice).

audio_setting

The audio it makes (MiniMaxAudio).

language_boost

A language or dialect it listens for (default: none).

normalization_setting

How it evens out the loudness (MiniMaxLoudness).

voice_modify

How it changes the voice itself (MiniMaxTimbre).

pronunciation_dict

How it pronounces particular words (MiniMaxPronunciation).

voice_setting

The voice and how it speaks (MiniMaxVoice).

audio_setting

The audio it makes (MiniMaxAudio).

language_boost

A language or dialect it listens for (default: none).

normalization_setting

How it evens out the loudness (MiniMaxLoudness).

voice_modify

How it changes the voice itself (MiniMaxTimbre).

pronunciation_dict

How it pronounces particular words (MiniMaxPronunciation).

MiniMaxVoice

A MiniMax voice and how it speaks.

voice_id

The voice: Wise_Woman, Friendly_Person, Inspirational_girl, Deep_Voice_Man, Calm_Woman, Casual_Guy, Lively_Girl, Patient_Man, Young_Knight, Determined_Man, Lovely_Girl, Decent_Boy, Imposing_Manner, Elegant_Man, Abbess, Sweet_Girl_2, Exuberant_Girl, or a cloned voice's ID (default "Wise_Woman").

speed

How fast, from 0.5 to 2 (default 1).

vol

How loud, from 0.01 to 10 (default 1).

pitch

How high, in semitones from -12 to 12 (default 0).

emotion

The emotion it speaks with (default: as the text reads).

english_normalization

Whether English numbers are read more carefully, a little slower (default False).

voice_id

The voice: Wise_Woman, Friendly_Person, Inspirational_girl, Deep_Voice_Man, Calm_Woman, Casual_Guy, Lively_Girl, Patient_Man, Young_Knight, Determined_Man, Lovely_Girl, Decent_Boy, Imposing_Manner, Elegant_Man, Abbess, Sweet_Girl_2, Exuberant_Girl, or a cloned voice's ID (default "Wise_Woman").

speed

How fast, from 0.5 to 2 (default 1).

vol

How loud, from 0.01 to 10 (default 1).

pitch

How high, in semitones from -12 to 12 (default 0).

emotion

The emotion it speaks with (default: as the text reads).

english_normalization

Whether English numbers are read more carefully, a little slower (default False).

MiniMaxAudio

The audio MiniMax makes.

sample_rate

Samples a second (default 32000).

bitrate

Bits a second, for MP3 (default 128000).

format

The file's format (default "mp3").

channel

Channels: 1 (mono) or 2 (stereo) (default 1).

sample_rate

Samples a second (default 32000).

bitrate

Bits a second, for MP3 (default 128000).

format

The file's format (default "mp3").

channel

Channels: 1 (mono) or 2 (stereo) (default 1).

MiniMaxLoudness

How MiniMax evens out the audio's loudness.

enabled

Whether it does (default True).

target_loudness

The loudness it aims at, in LUFS, from -70 to -10 (default -18).

target_range

The loudness range, in LU, from 0 to 20 (default 8).

target_peak

The highest peak, in dBTP, from -3 to 0 (default -0.5).

enabled

Whether it does (default True).

target_loudness

The loudness it aims at, in LUFS, from -70 to -10 (default -18).

target_range

The loudness range, in LU, from 0 to 20 (default 8).

target_peak

The highest peak, in dBTP, from -3 to 0 (default -0.5).

MiniMaxTimbre

How MiniMax changes the voice itself.

pitch

Higher or lower, from -100 to 100 (default 0).

intensity

More or less energetic, from -100 to 100 (default 0).

timbre

Its tone's color, from -100 to 100 (default 0).

pitch

Higher or lower, from -100 to 100 (default 0).

intensity

More or less energetic, from -100 to 100 (default 0).

timbre

Its tone's color, from -100 to 100 (default 0).

MiniMaxPronunciation

How MiniMax pronounces particular words.

tone_list

Each word and its pronunciation: "text/(pronunciation)" (Chinese tones 1 to 5: "燕少飞/(yan4)(shao3)(fei1)").

tone_list

Each word and its pronunciation: "text/(pronunciation)" (Chinese tones 1 to 5: "燕少飞/(yan4)(shao3)(fei1)").

Gemini

The settings of Google's Gemini 3.8 Flash TTS (google/gemini-3.8-flash-tts). Its text may carry vocal events: <laugh>, <sigh>.

voice

The voice (default "Kore").

style_instructions

How to say it, in words: "Warm and unhurried, like a patient teacher." (default: none).

voice

The voice (default "Kore").

style_instructions

How to say it, in words: "Warm and unhurried, like a patient teacher." (default: none).

Inworld

The settings of Inworld TTS-1.5 Max (fal-ai/inworld-tts), which speaks up to 2,000 characters a line.

voice

The voice, with its language (default "Craig (en)").

sample_rate_hertz

Samples a second (default 48000).

voice

The voice, with its language (default "Craig (en)").

sample_rate_hertz

Samples a second (default 48000).

Qwen

The settings of Alibaba's Qwen3-TTS 1.7B (fal-ai/qwen-3-tts/text-to-speech/1.7b).

voice

The voice (default: the model's).

language

The language (default "Auto": the text's).

prompt

How to say it, in words (not what to say) (default: none).

speaker_voice_embedding_file_url

A cloned voice: the URL of the speaker embedding fal-ai/qwen-3-tts/clone-voice made; it overrides voice and prompt (default: none).

reference_text

What the cloned voice's recording says, which helps it sound like it (default: none).

temperature

How varied the delivery is, above 0 and up to 1 (default 0.9).

top_k

Sampling: from the k likeliest sounds (default 50).

top_p

Sampling: from the likeliest sounds that make up p of the chance, 0 to 1 (default 1).

repetition_penalty

How strongly it avoids repeating itself (default 1.05).

subtalker_dosample

Whether its second stage samples (default True).

subtalker_temperature

Its second stage's temperature, 0 to 1 (default 0.9).

subtalker_top_k

Its second stage's top k (default 50).

subtalker_top_p

Its second stage's top p, 0 to 1 (default 1).

max_new_tokens

The most sound it makes, in its codec's tokens, 1 to 8192 (default 8192 here: fal's own, 200, can cut a long line short).

voice

The voice (default: the model's).

language

The language (default "Auto": the text's).

prompt

How to say it, in words (not what to say) (default: none).

speaker_voice_embedding_file_url

A cloned voice: the URL of the speaker embedding fal-ai/qwen-3-tts/clone-voice made; it overrides voice and prompt (default: none).

reference_text

What the cloned voice's recording says, which helps it sound like it (default: none).

temperature

How varied the delivery is, above 0 and up to 1 (default 0.9).

top_k

Sampling: from the k likeliest sounds (default 50).

top_p

Sampling: from the likeliest sounds that make up p of the chance, 0 to 1 (default 1).

repetition_penalty

How strongly it avoids repeating itself (default 1.05).

subtalker_dosample

Whether its second stage samples (default True).

subtalker_temperature

Its second stage's temperature, 0 to 1 (default 0.9).

subtalker_top_k

Its second stage's top k (default 50).

subtalker_top_p

Its second stage's top p, 0 to 1 (default 1).

max_new_tokens

The most sound it makes, in its codec's tokens, 1 to 8192 (default 8192 here: fal's own, 200, can cut a long line short).

Captions

add_subcaption

Caption the film: content shown from now plus offset, for duration seconds.

The film's captions hold it, with those of its speech; manimgx render writes them beside the video, as subtitles.

self.add_subcaption(content, duration=1, offset=0)
content

The caption's text.

duration

How long it shows, in seconds.

offset

How long after now it begins, in seconds.

Source

src/manimgx/scene.py

def add_subcaption(
    self, content: str, duration: float = 1, offset: float = 0
) -> None:
    """Caption the film: `content` shown from now plus `offset`, for `duration` seconds.

    The film's [`captions`][manimgx.Film.captions] hold it, with those of its
    speech; `manimgx render` writes them beside the video, as subtitles.

    Args:
        content: The caption's text.
        duration: How long it shows, in seconds.
        offset: How long after now it begins, in seconds.
    """
    start = float(clock.now) + offset
    self.film.subcaptions.append(Caption(start, start + duration, content))

A voice of your own

A voice is any function from a text to a Speech: a sound that knows its words, and when each is said. Any text-to-speech service can be one.

Voice

Voice = Callable[[str], Speech]

Anything that speaks: a function from text to Speech. Bring any text-to-speech: wrap its call in a function that returns Speech(audio, text=text).

Word

A word of a speech, and when it is said: from start to end, in seconds from the speech's start.

m.voices.Word(text, start, end)

timed

The words of text, timed by a voice's timed pieces of it, in order (its characters, or its tokens): each word from the start of the piece its first letter is in to the end of the piece its last letter is in. Empty if the pieces don't spell the text's letters (a voice that said "two" for "2"): its words are then estimated.

m.voices.timed(text, pieces)
Source

src/manimgx/audio/sound.py

def timed(text: str, pieces: Sequence[tuple[str, float, float]]) -> tuple[Word, ...]:
    """The words of `text`, timed by a voice's timed pieces of it, in order (its characters,
    or its tokens): each word from the start of the piece its first letter is in to the end
    of the piece its last letter is in. Empty if the pieces don't spell the text's letters
    (a voice that said "two" for "2"): its words are then estimated."""
    # each letter of the pieces: its piece's times
    owner: list[tuple[float, float]] = []
    for piece, a, b in pieces:
        owner += [(a, b)] * sum(c.isalnum() for c in piece)
    letters = [c.lower() for c in text if c.isalnum()]
    spelled = [c.lower() for piece, _, _ in pieces for c in piece if c.isalnum()]
    if letters != spelled:
        return ()
    words, k = [], 0
    for m in _WORD.finditer(text):
        n = sum(c.isalnum() for c in m.group())
        if n:
            words.append(Word(m.group(), owner[k][0], owner[k + n - 1][1]))
            k += n
        elif words:  # a word of no letters ("—"): at the end of the one before
            words.append(Word(m.group(), words[-1].end, words[-1].end))
    return tuple(words)

estimate

When each word of text is said in samples, estimated: the audio's voiced span shared among the words by their letters, with a share more for the pause after a comma or a full stop. Against voices' own times, it is off by about 0.1 s on average.

m.voices.estimate(text, samples)
Source

src/manimgx/audio/sound.py

def estimate(text: str, samples: np.ndarray) -> tuple[Word, ...]:
    """When each word of `text` is said in `samples`, estimated: the audio's voiced span
    shared among the words by their letters, with a share more for the pause after a comma or
    a full stop. Against voices' own times, it is off by about 0.1 s on average."""
    words = _WORD.findall(text)
    if not words:
        return ()
    level = np.abs(samples).max(axis=1) if samples.ndim > 1 else np.abs(samples)
    hop = RATE // 100  # 10 ms
    frames = len(level) // hop
    loud = level[: frames * hop].reshape(frames, hop).max(axis=1) > 0.02
    voiced = np.flatnonzero(loud)
    if not len(voiced):
        total = len(samples) / RATE
        return tuple(
            Word(w, total * i / len(words), total * (i + 1) / len(words))
            for i, w in enumerate(words)
        )
    first, last = voiced[0], voiced[-1] + 1
    # weight: letters, plus a pause after a comma or a full stop
    weights = []
    for w in words:
        weights.append(len(w) + (6 if w[-1] in ".!?:;" else 3 if w[-1] == "," else 1))
    total_weight = sum(weights)
    edges = np.concatenate([[0], np.cumsum(weights)]) / total_weight
    span = (last - first) / 100
    times = first / 100 + edges * span
    return tuple(
        Word(
            w,
            float(times[i]),
            float(times[i + 1] - (weights[i] - len(w)) / total_weight * span),
        )
        for i, w in enumerate(words)
    )

cached

What voice says for text: spoken once, then kept in folder as its audio and a JSON of its words, named by the text and a digest of the voice and the text.

The audio may be replaced (a recording of the same words, in any format the engine reads, under the same name): its words are then timed again, from it.

m.voices.cached(voice, text, folder)
Source

src/manimgx/audio/sound.py

def cached(voice: Voice, text: str, folder: Path) -> Speech:
    """What `voice` says for `text`: spoken once, then kept in `folder` as its audio and a
    JSON of its words, named by the text and a digest of the voice and the text.

    The audio may be replaced (a recording of the same words, in any format the engine
    reads, under the same name): its words are then timed again, from it.
    """
    from manimgx._engine import digest

    identity = _identity(voice)
    key = f"{digest(identity.encode(), b'\0', text.encode()):016x}"
    slug = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")[:40]
    stem = folder / f"{slug}-{key[:10]}"
    meta = stem.with_suffix(".json")
    if meta.exists():
        record = json.loads(meta.read_text(encoding="utf-8"))
        audio = stem.with_suffix(record["suffix"]).read_bytes()
        same = record.get("audio") == f"{digest(audio):016x}"
        words = [Word(*w) for w in record["words"]] if same else []
        return Speech(audio, text=text, words=words)
    speech = voice(text)
    folder.mkdir(parents=True, exist_ok=True)
    suffix, data = _stored(speech)
    stem.with_suffix(suffix).write_bytes(data)
    meta.write_text(
        json.dumps(
            {
                "text": text,
                "voice": identity,
                "suffix": suffix,
                "audio": f"{digest(data):016x}",
                "words": [[w.text, w.start, w.end] for w in speech._words],
            },
            indent=1,
        )
        + "\n",  # a text file, to commit
        encoding="utf-8",
    )
    return speech