Crawl a public website into a knowledge base
curl --request POST \
--url https://api.pyai.com/v1/knowledgebases/{id}/crawls \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"url": "<string>",
"max_pages": 25
}
'import requests
url = "https://api.pyai.com/v1/knowledgebases/{id}/crawls"
payload = {
"url": "<string>",
"max_pages": 25
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({url: '<string>', max_pages: 25})
};
fetch('https://api.pyai.com/v1/knowledgebases/{id}/crawls', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pyai.com/v1/knowledgebases/{id}/crawls",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'max_pages' => 25
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.pyai.com/v1/knowledgebases/{id}/crawls"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.pyai.com/v1/knowledgebases/{id}/crawls")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pyai.com/v1/knowledgebases/{id}/crawls")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}"
response = http.request(request)
puts response.read_body{
"object": "knowledgebase.crawl",
"seed_url": "<string>",
"max_pages": 123,
"discovered": 123,
"accepted": 123,
"reused": 123,
"skipped": [
{
"url": "<string>",
"reason": "<string>"
}
],
"documents": [
{
"id": "<string>",
"url": "<string>",
"title": "<string>",
"reused": true
}
]
}{
"title": "<string>",
"status": 123,
"type": "<string>",
"detail": "<string>",
"request_id": "<string>"
}{
"title": "<string>",
"status": 123,
"type": "<string>",
"detail": "<string>",
"request_id": "<string>"
}Knowledge Bases
Crawl a public website into a knowledge base
Discover same-origin pages from a public seed URL (robots.txt / sitemap, then a shallow link walk), rank them, and register each selected page as a normal URL document. Private, loopback, and metadata addresses are rejected. Defaults to 25 pages, hard-capped at 40. Already-ingested URLs in this knowledge base are reused. Poll document status until indexed before treating the site as ready. Requires the kb:manage scope.
POST
/
v1
/
knowledgebases
/
{id}
/
crawls
Crawl a public website into a knowledge base
curl --request POST \
--url https://api.pyai.com/v1/knowledgebases/{id}/crawls \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"url": "<string>",
"max_pages": 25
}
'import requests
url = "https://api.pyai.com/v1/knowledgebases/{id}/crawls"
payload = {
"url": "<string>",
"max_pages": 25
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({url: '<string>', max_pages: 25})
};
fetch('https://api.pyai.com/v1/knowledgebases/{id}/crawls', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pyai.com/v1/knowledgebases/{id}/crawls",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'max_pages' => 25
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.pyai.com/v1/knowledgebases/{id}/crawls"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.pyai.com/v1/knowledgebases/{id}/crawls")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pyai.com/v1/knowledgebases/{id}/crawls")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"max_pages\": 25\n}"
response = http.request(request)
puts response.read_body{
"object": "knowledgebase.crawl",
"seed_url": "<string>",
"max_pages": 123,
"discovered": 123,
"accepted": 123,
"reused": 123,
"skipped": [
{
"url": "<string>",
"reason": "<string>"
}
],
"documents": [
{
"id": "<string>",
"url": "<string>",
"title": "<string>",
"reused": true
}
]
}{
"title": "<string>",
"status": 123,
"type": "<string>",
"detail": "<string>",
"request_id": "<string>"
}{
"title": "<string>",
"status": 123,
"type": "<string>",
"detail": "<string>",
"request_id": "<string>"
}Authorizations
apiKeyxApiKey
Use Authorization: Bearer pyai_live_... (or pyai_test_...).
Path Parameters
Body
application/json
Response
Pages registered (pending ingestion)
Example:
"knowledgebase.crawl"
Same-origin candidates considered before ranking.
New URL documents registered by this crawl.
Selected URLs that already existed in this knowledge base.
Show child attributes
Show child attributes
Show child attributes
Show child attributes