import os
from qaip import Qaip
client = Qaip(
api_key=os.environ.get("QAIP_API_KEY"), # This is the default and can be omitted
)
crawl = client.crawls.create(
max_depth=1,
max_num_files=1,
name="name",
start_url="start_url",
)
print(crawl.id)curl --request POST \
--url https://developer.qaip.com/api/v1/crawls \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"name": "<string>",
"start_url": "<string>",
"max_depth": 5,
"max_num_files": 50000,
"path_filters": [
"<string>"
],
"content_pattern": [
"<string>"
],
"html_only": false,
"use_browser": false,
"no_canonical_check": false,
"file_extensions": [
"<string>"
],
"rrule": "<string>",
"metadata": {
"records": [
{
"key": "<string>",
"val": "<string>"
}
]
}
}
'const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
start_url: '<string>',
max_depth: 5,
max_num_files: 50000,
path_filters: ['<string>'],
content_pattern: ['<string>'],
html_only: false,
use_browser: false,
no_canonical_check: false,
file_extensions: ['<string>'],
rrule: '<string>',
metadata: {records: [{key: '<string>', val: '<string>'}]}
})
};
fetch('https://developer.qaip.com/api/v1/crawls', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://developer.qaip.com/api/v1/crawls",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'start_url' => '<string>',
'max_depth' => 5,
'max_num_files' => 50000,
'path_filters' => [
'<string>'
],
'content_pattern' => [
'<string>'
],
'html_only' => false,
'use_browser' => false,
'no_canonical_check' => false,
'file_extensions' => [
'<string>'
],
'rrule' => '<string>',
'metadata' => [
'records' => [
[
'key' => '<string>',
'val' => '<string>'
]
]
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://developer.qaip.com/api/v1/crawls"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://developer.qaip.com/api/v1/crawls")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://developer.qaip.com/api/v1/crawls")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"name": "<string>",
"start_url": "<string>",
"status": "unknown",
"ingestion_setting_id": "<string>",
"creation_time": 123,
"start_time": 123,
"end_time": 123,
"error": {
"title": "<string>",
"message": "<string>"
},
"metadata": {
"records": [
{
"key": "<string>",
"val": "<string>",
"type": "string"
}
]
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}Create a web crawl data source
Creates a new web crawl data source and starts ingestion.
Required scope: ingestion:manage
import os
from qaip import Qaip
client = Qaip(
api_key=os.environ.get("QAIP_API_KEY"), # This is the default and can be omitted
)
crawl = client.crawls.create(
max_depth=1,
max_num_files=1,
name="name",
start_url="start_url",
)
print(crawl.id)curl --request POST \
--url https://developer.qaip.com/api/v1/crawls \
--header 'Content-Type: application/json' \
--header 'x-api-key: <api-key>' \
--data '
{
"name": "<string>",
"start_url": "<string>",
"max_depth": 5,
"max_num_files": 50000,
"path_filters": [
"<string>"
],
"content_pattern": [
"<string>"
],
"html_only": false,
"use_browser": false,
"no_canonical_check": false,
"file_extensions": [
"<string>"
],
"rrule": "<string>",
"metadata": {
"records": [
{
"key": "<string>",
"val": "<string>"
}
]
}
}
'const options = {
method: 'POST',
headers: {'x-api-key': '<api-key>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
start_url: '<string>',
max_depth: 5,
max_num_files: 50000,
path_filters: ['<string>'],
content_pattern: ['<string>'],
html_only: false,
use_browser: false,
no_canonical_check: false,
file_extensions: ['<string>'],
rrule: '<string>',
metadata: {records: [{key: '<string>', val: '<string>'}]}
})
};
fetch('https://developer.qaip.com/api/v1/crawls', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://developer.qaip.com/api/v1/crawls",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'start_url' => '<string>',
'max_depth' => 5,
'max_num_files' => 50000,
'path_filters' => [
'<string>'
],
'content_pattern' => [
'<string>'
],
'html_only' => false,
'use_browser' => false,
'no_canonical_check' => false,
'file_extensions' => [
'<string>'
],
'rrule' => '<string>',
'metadata' => [
'records' => [
[
'key' => '<string>',
'val' => '<string>'
]
]
]
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json",
"x-api-key: <api-key>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://developer.qaip.com/api/v1/crawls"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("x-api-key", "<api-key>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://developer.qaip.com/api/v1/crawls")
.header("x-api-key", "<api-key>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://developer.qaip.com/api/v1/crawls")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["x-api-key"] = '<api-key>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"start_url\": \"<string>\",\n \"max_depth\": 5,\n \"max_num_files\": 50000,\n \"path_filters\": [\n \"<string>\"\n ],\n \"content_pattern\": [\n \"<string>\"\n ],\n \"html_only\": false,\n \"use_browser\": false,\n \"no_canonical_check\": false,\n \"file_extensions\": [\n \"<string>\"\n ],\n \"rrule\": \"<string>\",\n \"metadata\": {\n \"records\": [\n {\n \"key\": \"<string>\",\n \"val\": \"<string>\"\n }\n ]\n }\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"name": "<string>",
"start_url": "<string>",
"status": "unknown",
"ingestion_setting_id": "<string>",
"creation_time": 123,
"start_time": 123,
"end_time": 123,
"error": {
"title": "<string>",
"message": "<string>"
},
"metadata": {
"records": [
{
"key": "<string>",
"val": "<string>",
"type": "string"
}
]
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}{
"error": {
"message": "<string>",
"type": "<string>"
}
}Authorizations
API key for authentication
Body
Name of the web crawl data source
200Start URL of the web crawl
2000Maximum crawl depth
1 <= x <= 10Maximum number of files to crawl
1 <= x <= 100000Path filters for crawling. The total number of characters across all elements in the array must be 2000 or fewer.
2000Content patterns for filtering. The total number of characters across all elements in the array must be 2000 or fewer.
2000When true, only HTML files will be downloaded
Whether to use a headless browser for crawling
Whether to disable duplicate filtering by canonical URL
File extensions to include (e.g. ".pdf", ".docx"). For supported file extensions, please refer to https://developer.qaip.com/docs/datasources#%E5%AF%BE%E5%BF%9C%E3%81%97%E3%81%A6%E3%81%84%E3%82%8B%E3%83%95%E3%82%A1%E3%82%A4%E3%83%AB%E5%BD%A2%E5%BC%8F
200010Recurrence rule (RFC 5545 RRULE)
(reserved for future use) Additional metadata for the web crawl data source
Show child attributes
Show child attributes
Response
Successfully created web crawl data source
Web crawl data source ID
Name of the web crawl ingestion setting
Start URL of the web crawl
Job status
unknown, queued, not_started, managed, starting, started, success, failure, canceling, canceled, deleting, delete_job_failure Web crawl ingestion setting ID
Creation time (Unix timestamp in seconds)
Job start time (Unix timestamp in seconds)
Job end time (Unix timestamp in seconds)
Show child attributes
Show child attributes
(reserved for future use) Additional metadata for the web crawl data source
Show child attributes
Show child attributes

