Create a new web crawl job
curl --request POST \
--url https://api.example.com/api/connect/crawling/ \
--header 'Content-Type: application/json' \
--data '
{
"url": "<string>",
"limit": 123,
"excludePaths": [
"<string>"
],
"includePaths": [
"<string>"
],
"allowExternalLinks": true
}
'import requests
url = "https://api.example.com/api/connect/crawling/"
payload = {
"url": "<string>",
"limit": 123,
"excludePaths": ["<string>"],
"includePaths": ["<string>"],
"allowExternalLinks": True
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
url: '<string>',
limit: 123,
excludePaths: ['<string>'],
includePaths: ['<string>'],
allowExternalLinks: true
})
};
fetch('https://api.example.com/api/connect/crawling/', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.example.com/api/connect/crawling/",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'limit' => 123,
'excludePaths' => [
'<string>'
],
'includePaths' => [
'<string>'
],
'allowExternalLinks' => true
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.example.com/api/connect/crawling/"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.example.com/api/connect/crawling/")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.example.com/api/connect/crawling/")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"graphite_project_id": "<string>",
"url": "<string>",
"created_at": "<string>",
"updated_at": "<string>",
"completed_at": "<string>",
"completed_pages": 123,
"error_message": "<string>",
"total_pages": 123,
"tenant_id": "<string>"
}Web Crawling
Create a new web crawl job
POST
/
api
/
connect
/
crawling
/
Create a new web crawl job
curl --request POST \
--url https://api.example.com/api/connect/crawling/ \
--header 'Content-Type: application/json' \
--data '
{
"url": "<string>",
"limit": 123,
"excludePaths": [
"<string>"
],
"includePaths": [
"<string>"
],
"allowExternalLinks": true
}
'import requests
url = "https://api.example.com/api/connect/crawling/"
payload = {
"url": "<string>",
"limit": 123,
"excludePaths": ["<string>"],
"includePaths": ["<string>"],
"allowExternalLinks": True
}
headers = {"Content-Type": "application/json"}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {'Content-Type': 'application/json'},
body: JSON.stringify({
url: '<string>',
limit: 123,
excludePaths: ['<string>'],
includePaths: ['<string>'],
allowExternalLinks: true
})
};
fetch('https://api.example.com/api/connect/crawling/', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.example.com/api/connect/crawling/",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'url' => '<string>',
'limit' => 123,
'excludePaths' => [
'<string>'
],
'includePaths' => [
'<string>'
],
'allowExternalLinks' => true
]),
CURLOPT_HTTPHEADER => [
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.example.com/api/connect/crawling/"
payload := strings.NewReader("{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.example.com/api/connect/crawling/")
.header("Content-Type", "application/json")
.body("{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.example.com/api/connect/crawling/")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Content-Type"] = 'application/json'
request.body = "{\n \"url\": \"<string>\",\n \"limit\": 123,\n \"excludePaths\": [\n \"<string>\"\n ],\n \"includePaths\": [\n \"<string>\"\n ],\n \"allowExternalLinks\": true\n}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"graphite_project_id": "<string>",
"url": "<string>",
"created_at": "<string>",
"updated_at": "<string>",
"completed_at": "<string>",
"completed_pages": 123,
"error_message": "<string>",
"total_pages": 123,
"tenant_id": "<string>"
}Body
application/json
Response
200 - application/json
Default Response
Available options:
idle, cancelled, completed, failed, scraping Was this page helpful?
⌘I