curl --request GET \
--url https://api.pyai.com/v1/amd/stream \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.pyai.com/v1/amd/stream"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.pyai.com/v1/amd/stream', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pyai.com/v1/amd/stream",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.pyai.com/v1/amd/stream"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.pyai.com/v1/amd/stream")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pyai.com/v1/amd/stream")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"error": {
"message": "<string>",
"type": "<string>",
"code": "invalid_request_error",
"param": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>",
"code": "invalid_request_error",
"param": "<string>"
}
}Answering-machine detection (WebSocket)
Realtime answering-machine detection over a WebSocket. This surface speaks Twilio’s Media Streams protocol natively (start / media / stop frames, G.711 μ-law 8 kHz base64, ~20 ms), so migrating from Twilio AMD is a one-line-TwiML change, point the call’s media at PyAI, keep your carrier and your code.
<Response><Start>
<Stream url="wss://api.pyai.com/v1/amd/stream">
<Parameter name="api_key" value="YOUR_PYAI_KEY"/>
<Parameter name="aggressiveness" value="0.25"/>
<Parameter name="decision_timeout_ms" value="3000"/>
<Parameter name="lead_id" value="lead-42"/>
<Parameter name="webhook" value="https://you/amd-events"/>
</Stream>
</Start>
<!-- your existing call flow continues here -->
</Response>
Use <Start><Stream>, NOT <Connect><Stream>. <Start> forks the audio and TwiML continues to your next verb, so the call still goes where it was going; <Connect> hands the media path to the socket and blocks TwiML until the stream ends, and because AMD is listen-only and never sends audio back the caller would hear dead air and a dialer would never reach the agent. (<Connect> is correct for Omni, which is a two-way voice agent.) Drop machineDetection from the call and keep your carrier.
On a Twilio-originated stream TWILIO owns the socket and relays only media/mark frames, so the pushed amd event does not reach you: a Twilio integration must read the decision from the webhook <Parameter> (or GET /v1/amd/calls/{id}). The socket push is for clients that drive the socket themselves.
From Twilio, authenticate with the api_key <Parameter> shown above (Twilio strips query strings from the <Stream> URL and cannot send headers; PyAI verifies the key from the stream’s start frame before processing any audio, and closes connections that never present a valid key). Server-side clients may instead authenticate at the handshake with the Sec-WebSocket-Protocol: pyai.v1, pyai-key.<API_KEY> subprotocol pair or ?api_key=. Requires the amd:detect scope. Mid-call, PyAI pushes an amd decision event on the socket (and to the per-call TwiML webhook parameter): answered_by (the routing class: human, machine, sit_invalid, unknown), answered_by_twilio (Twilio’s exact AnsweredBy enum for drop-in routing parity), subtype (when available), party_detected, voicemail_ready, confidence, decision_ms, and a human-readable reason. party_detected is true for human or machine classification and false for unknown or invalid-number outcomes. voicemail_ready is false on classification events: detecting voicemail does not establish that recording has started or authorize dropping a message. decision_ms measures processed inbound audio through the decision, not elapsed time from carrier answer. A machine decision can carry a subtype (voicemail, ivr, screening, music); the stored call record folds that subtype into answered_by. Read the stored record with GET /v1/amd/calls/{id} or receive it on the account-wide amd.call.completed webhook (webhook_url in POST /v1/amd/config). The per-call aggressiveness <Parameter> overrides the account default from POST /v1/amd/config.
Decision events also include engine_version (decision-policy identity), rule_id (stable rule identifier), speech_ms (cumulative voiced audio excluding pauses, null if untracked), silence_ms (current contiguous silence), speech_elapsed_ms (audio since first voiced frame including pauses, null without onset), and thresholds (effective window, human dwell, early-yield hold and rule flags). These optional diagnostics also appear on new stored call details and the account-wide completion webhook; older records may omit them. Historical turn-yield reasons called elapsed time including trailing silence speech; use the new structured fields for comparisons. Confidence is a rule score, not a calibrated probability. Temporal human decisions require at least 200 ms of measured speech; isolated pulses cannot qualify just by waiting in silence. English introductions allow 2500 ms of silence for continuation, while complete recognized greetings can qualify after 500 ms of silence. Recognized call-progress tones are excluded from speech evidence. Without an explicit decision_window_ms override, the English 5000 ms base window can extend once by up to 2000 ms for recent speech or pending recognition; thresholds.effective_deadline_ms records the resulting deadline.
English streams may set decision_window_ms in start.customParameters (or a TwiML Parameter) to an integer or decimal integer string from 1000 to 15000. Omitting it keeps the configured default. Non-English overrides and invalid values emit an error event with code invalid_decision_window and close with code 1008. A longer window allows late evidence; it does not force unknown calls to a binary result. Continue real-time media pacing. Replay only the callee channel; never stereo-downmix the rep and callee. Sales Dialer Predictive recordings use channel 0 for the callee, Auto/Dynamic use channel 1; verify the source format before replaying.
Set decision_timeout_ms in start.customParameters to an integer or decimal integer string from 1000 to 15000 (for example 3000 or 5000) to cap elapsed decision time in any supported language. The clock starts when the authenticated start frame and its parameters are accepted, before recognizer startup. Decisive evidence returns a result earlier. At the cutoff, AMD uses decisive evidence already received by that time; otherwise it emits unknown with rule_id=decision_timeout, including when no media arrives or recognition is still pending. It does not force a human or machine guess. This elapsed cap also bounds the final recognition wait and is not extended by the adaptive audio window. decision_window_ms remains a separate processed-audio limit; either limit can finish the decision first. Omit decision_timeout_ms to preserve existing behavior. Invalid values emit invalid_stream_parameters and close with code 1008. Timing diagnostics decision_timeout_ms and decision_elapsed_ms accompany opted-in results in socket events, per-call webhooks, stored details and account completion webhooks. The timeout bounds the decision budget, not network delivery or receiver acknowledgement; allow transport time for your own fallback timer.
Customer correlation fields such as lead_id and campaign_id supplied as additional TwiML Parameters / start.customParameters are returned under custom_parameters in the decision event, both webhook paths and the stored call detail. Fields remain nested and cannot override call_id, answered_by or other AMD output. Names must match [A-Za-z0-9][A-Za-z0-9_.-]{0,63}; values must be strings of at most 1024 UTF-8 bytes. At most 32 correlation fields and 8192 UTF-8 bytes of combined names and values are accepted. Invalid correlation fields emit invalid_stream_parameters. Reserved authentication, routing and AMD configuration parameters and names beginning with _ are excluded from the echo. Do not send secrets as correlation fields. Query parameters on the per-call webhook URL are preserved in that URL; they are not copied into the JSON body.
Business introductions, generic requests for the reason for calling, unfinished requests to record a name, and recording disclosures are not decisive machine evidence by themselves. The default policy also abstains on tonal audio without recognized words; historical music results remain readable. These cases can return unknown while stronger evidence is absent. Screening and IVR may lead to a human: preserve that routing opportunity rather than treating every machine result as authorization to end the call. AMD emits one final classification per stream; it does not predict a later human pickup.
When the decision arrives
Measured over ~1,850 real answered calls (US telephony, 8 kHz μ-law), streamed at real time:
| verdict | typical | 9 in 10 by |
|---|---|---|
human | ~1.4 s | ~3.0 s |
machine | ~2.2 s | ~3.2 s |
These historical latency measurements are not a guarantee for the current policy. The English default window is 5 seconds of processed audio with a bounded extension to 7 seconds for eligible calls. Size your fallback timer beyond the applicable decision window plus transport time, and react to the decision event.
What to do with each verdict
answered_by | subtype | do |
|---|---|---|
human | — | connect the agent |
machine | voicemail | consider a message only after independently establishing recording readiness |
machine | ivr | a phone tree that may reach a person; navigate or route to an agent, and do not drop a message |
machine | screening | an AI screener (iPhone/Google) is relaying to a person; treat as a live-ish path, not voicemail |
machine | music | tonal audio and nothing transcribed — hold music or ringback, but also a greeting we failed to transcribe. Keep waiting; do not read it as a positive hold-music signal, and do not gate a drop on it |
sit_invalid | — | dead/invalid number, stop retrying it |
unknown | silence | answered but nothing came down the line, retry later rather than burning an agent slot |
unknown | — | no decisive evidence; your default decides |
Read subtype on the wire, stored record, or completion webhook. A machine classification alone does not authorize hangup or voicemail drop. Screening and IVR can connect a real person. Music is a historical/legacy classification of tonal audio without text, not proof of an unreachable lead. Consider voicemail actions only for subtype voicemail and after establishing recording readiness separately; voicemail_ready is false on classification events.
Choosing aggressiveness
The aggressiveness dial adjusts evidence timing thresholds. It does not manufacture machine evidence from a deadline: uncertain calls remain unknown at every setting. For live-agent dialers, retain the default and preserve the call when uncertain. Validate human-to-machine errors using all independently reviewed human calls as the denominator; confidence is a rule score, not a calibrated probability.
Billed per answered call (amd.calls), the first 5,000 answered calls/month are free, then $0.004/answered call; AMD bundled with PyAI telephony/Omni is included.
curl --request GET \
--url https://api.pyai.com/v1/amd/stream \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.pyai.com/v1/amd/stream"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.pyai.com/v1/amd/stream', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pyai.com/v1/amd/stream",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.pyai.com/v1/amd/stream"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.pyai.com/v1/amd/stream")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pyai.com/v1/amd/stream")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"error": {
"message": "<string>",
"type": "<string>",
"code": "invalid_request_error",
"param": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>",
"code": "invalid_request_error",
"param": "<string>"
}
}Authorizations
Use Authorization: Bearer pyai_live_... (or pyai_test_...).
Response
WebSocket upgrade, Twilio Media Streams protocol; PyAI emits amd decision events.