diff --git a/api-reference/v2/live/callback/answering-machine.mdx b/api-reference/v2/live/callback/answering-machine.mdx
new file mode 100644
index 0000000..e310aba
--- /dev/null
+++ b/api-reference/v2/live/callback/answering-machine.mdx
@@ -0,0 +1,5 @@
+---
+title: Answering Machine
+description: Payload definition for the callback event `live.answering_machine`.
+openapi-schema: CallbackLiveAnsweringMachineMessage
+---
diff --git a/asyncapi.yaml b/asyncapi.yaml
index 91fbcbd..01584d4 100644
--- a/asyncapi.yaml
+++ b/asyncapi.yaml
@@ -32,6 +32,8 @@ channels:
$ref: "#/components/messages/speechStart"
speechEnd:
$ref: "#/components/messages/speechEnd"
+ answeringMachine:
+ $ref: "#/components/messages/answeringMachine"
translation:
$ref: "#/components/messages/translation"
namedEntityRecognition:
@@ -103,6 +105,12 @@ operations:
$ref: "#/channels/liveTranscription"
messages:
- $ref: "#/channels/liveTranscription/messages/speechEnd"
+ onAnsweringMachine:
+ action: send
+ channel:
+ $ref: "#/channels/liveTranscription"
+ messages:
+ - $ref: "#/channels/liveTranscription/messages/answeringMachine"
onTranslation:
action: send
channel:
@@ -259,6 +267,13 @@ components:
summary: Indicates the end of detected speech activity.
payload:
$ref: "#/components/schemas/SpeechEventMessageEnd"
+ answeringMachine:
+ name: answering_machine
+ title: Answering Machine
+ contentType: application/json
+ summary: Reports whether a human or an answering machine picked up.
+ payload:
+ $ref: "#/components/schemas/AnsweringMachineMessage"
translation:
name: translation
title: Translation
@@ -595,6 +610,65 @@ components:
time: 1.24
channel: 0
+ AnsweringMachineMessage:
+ title: Answering Machine
+ type: object
+ required: [session_id, created_at, type, data]
+ properties:
+ session_id: { $ref: "#/components/schemas/UUID" }
+ created_at: { $ref: "#/components/schemas/ISODateTime" }
+ type: { type: string, const: answering_machine }
+ data:
+ type: object
+ required: [kind, confidence, time]
+ properties:
+ kind:
+ type: string
+ enum: [human, machine]
+ description: Whether a person picked up or the call reached an answering machine.
+ confidence:
+ type: number
+ minimum: 0
+ maximum: 1
+ description: >-
+ How machine-like the call opening was, between 0 and 1. This is a ranking
+ score, not a calibrated probability: compare it against a threshold tuned on
+ your own traffic rather than reading it as a likelihood.
+ time:
+ type: number
+ description: >-
+ Position in the audio stream, in seconds from the start of the session, at
+ which the decision was taken. Measured on the audio timeline and sharing the
+ same reference as speech_start, speech_end and transcript, so all messages can
+ be ordered against each other. Use created_at to correlate with wall-clock logs.
+ channel:
+ type: integer
+ description: Audio channel index (0-based) for which the decision applies.
+ description: Outcome of answering machine detection for the session.
+ description: >-
+ Reports whether a human or an answering machine picked up. Emitted at most once
+ per session, about five seconds after the first speech, and never revised.
+ Not emitted if nobody speaks or if the session ends before the decision.
+ Does not require transcription, so it behaves the same across languages; partial
+ or final transcripts may arrive before it.
+ examples:
+ - session_id: "550e8400-e29b-41d4-a716-446655440000"
+ created_at: "2026-09-09T12:34:11Z"
+ type: answering_machine
+ data:
+ kind: machine
+ confidence: 0.78
+ time: 5.24
+ channel: 0
+ - session_id: "550e8400-e29b-41d4-a716-446655440000"
+ created_at: "2026-09-09T12:34:11Z"
+ type: answering_machine
+ data:
+ kind: human
+ confidence: 0.12
+ time: 5.31
+ channel: 0
+
TranslationMessage:
type: object
required: [session_id, created_at, type, data]
diff --git a/chapters/live-stt/features/answering-machine-detection.mdx b/chapters/live-stt/features/answering-machine-detection.mdx
new file mode 100644
index 0000000..b979d5f
--- /dev/null
+++ b/chapters/live-stt/features/answering-machine-detection.mdx
@@ -0,0 +1,64 @@
+---
+title: "Answering Machine Detection"
+description: "Know within seconds whether a human or a voicemail picked up"
+---
+
+Answering machine detection tells you whether the person you called picked up, or whether you reached a voicemail greeting.
+
+The decision arrives about five seconds after the person you called starts speaking, without waiting for a transcript, so an outbound agent can act on it immediately.
+
+### The message
+
+You receive an `answering_machine` message at most once per session:
+
+```json
+{
+ "session_id": "550e8400-e29b-41d4-a716-446655440000",
+ "created_at": "2026-09-09T12:34:11Z",
+ "type": "answering_machine",
+ "data": {
+ "kind": "machine",
+ "confidence": 0.78,
+ "time": 5.24,
+ "channel": 0
+ }
+}
+```
+
+| Field | Meaning |
+|---|---|
+| `kind` | `human` or `machine` |
+| `confidence` | Score between 0 and 1 indicating how machine-like the call opening was |
+| `time` | Position in the audio stream, in seconds from the start of the session, at which the decision was taken |
+| `channel` | Audio channel index the decision applies to |
+
+
+`time` is measured on the **audio timeline**, not the clock. It uses the same reference as `speech_start`, `speech_end` and `transcript`, so you can order all of them against each other. If you push audio faster than real time, `time` still refers to the audio position, so use `created_at` when you need to correlate with your own logs.
+
+
+
+`confidence` is a **ranking score, not a probability**. A value of `0.7` does not mean "70% chance this is a machine". Use it to compare against a threshold you tune on your own traffic, not as a calibrated likelihood.
+
+
+### Acting on it
+
+Most callers act on `kind` directly:
+
+- **`machine`** — hang up and queue the number for a later attempt, or stay silent and let the greeting finish before leaving a recorded message.
+- **`human`** — start the conversation.
+
+If you want to be more conservative than the default, read `confidence` and require a high score before treating a call as a machine. Hanging up on a real person usually costs more than a few wasted seconds on a voicemail, so a threshold above the default is a common choice.
+
+### Timing and edge cases
+
+The decision fires **at most once per session** and is never revised. Later audio does not change it.
+
+- **The window starts at the first speech.** Ringing, connection delay and any leading silence do not count. The decision is taken once five seconds of audio have been heard after that first speech.
+- **Nobody speaks at all.** No message is emitted: silence never starts the window and is not evidence of a machine. Apply your own timeout if you need to give up on a silent line.
+- **Very short calls.** If the session ends before the five seconds have elapsed, no message is emitted.
+- **Transcripts are not ordered against it.** Partial or final transcripts may arrive before the decision.
+- **No transcription required.** The detection does not depend on what was said, so it behaves the same across languages and adds no transcription latency.
+
+### Limits
+
+Detection is tuned for the opening of an outbound call. It is not designed to detect an answering machine that picks up mid-conversation, and it does not attempt to identify IVR menus or hold music.
diff --git a/docs.json b/docs.json
index 4f6e194..a409986 100644
--- a/docs.json
+++ b/docs.json
@@ -63,6 +63,7 @@
"chapters/live-stt/quickstart",
"chapters/live-stt/audio-intelligence",
"chapters/live-stt/features/endpointing",
+ "chapters/live-stt/features/answering-machine-detection",
"chapters/live-stt/features/partial-transcripts",
"chapters/live-stt/recommended-parameters"
]
@@ -184,6 +185,7 @@
"api-reference/v2/live/callback/transcript",
"api-reference/v2/live/callback/speech-start",
"api-reference/v2/live/callback/speech-end",
+ "api-reference/v2/live/callback/answering-machine",
"api-reference/v2/live/callback/translation",
"api-reference/v2/live/callback/named-entity-recognition",
"api-reference/v2/live/callback/sentiment-analysis"