Scrape data and return it directly in the response.
curl --request POST \
--url https://api.brightdata.com/datasets/v3/scrape \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": [
{
"url": "www.linkedin.com/in/bulentakar"
}
],
"custom_output_fields": "url|about.updated_on"
}
'import requests
url = "https://api.brightdata.com/datasets/v3/scrape"
payload = {
"input": [{ "url": "www.linkedin.com/in/bulentakar" }],
"custom_output_fields": "url|about.updated_on"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: [{url: 'www.linkedin.com/in/bulentakar'}],
custom_output_fields: 'url|about.updated_on'
})
};
fetch('https://api.brightdata.com/datasets/v3/scrape', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.brightdata.com/datasets/v3/scrape",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => [
[
'url' => 'www.linkedin.com/in/bulentakar'
]
],
'custom_output_fields' => 'url|about.updated_on'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.brightdata.com/datasets/v3/scrape"
payload := strings.NewReader("{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.brightdata.com/datasets/v3/scrape")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.brightdata.com/datasets/v3/scrape")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}"
response = http.request(request)
puts response.read_body"OK"{
"status": "starting",
"message": "Snapshot is not ready yet, try again in 30s"
}Scraper API
Synchronous requests
Use the Bright Data Web Scraper API to synchronous Requests. POST /datasets/v3/scrape returns scraped data synchronously in a single response.
POST
/
datasets
/
v3
/
scrape
Scrape data and return it directly in the response.
curl --request POST \
--url https://api.brightdata.com/datasets/v3/scrape \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"input": [
{
"url": "www.linkedin.com/in/bulentakar"
}
],
"custom_output_fields": "url|about.updated_on"
}
'import requests
url = "https://api.brightdata.com/datasets/v3/scrape"
payload = {
"input": [{ "url": "www.linkedin.com/in/bulentakar" }],
"custom_output_fields": "url|about.updated_on"
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
input: [{url: 'www.linkedin.com/in/bulentakar'}],
custom_output_fields: 'url|about.updated_on'
})
};
fetch('https://api.brightdata.com/datasets/v3/scrape', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.brightdata.com/datasets/v3/scrape",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'input' => [
[
'url' => 'www.linkedin.com/in/bulentakar'
]
],
'custom_output_fields' => 'url|about.updated_on'
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.brightdata.com/datasets/v3/scrape"
payload := strings.NewReader("{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.brightdata.com/datasets/v3/scrape")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.brightdata.com/datasets/v3/scrape")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"input\": [\n {\n \"url\": \"www.linkedin.com/in/bulentakar\"\n }\n ],\n \"custom_output_fields\": \"url|about.updated_on\"\n}"
response = http.request(request)
puts response.read_body"OK"{
"status": "starting",
"message": "Snapshot is not ready yet, try again in 30s"
}How It Works
This synchronous API endpoint allows users to send a scraping request and receive the results in real-time directly in the response, at the point of request - such as a terminal or application - without the need for external storage or manual downloads. This approach streamlines the data collection process by eliminating additional steps for retrieving results. You can specify the desired output format using the format parameter. If no format is provided, the response will default to JSON.Timeout Limit
Please note that this synchronous request is subject to a 1 minute timeout limit. If the data retrieval process exceeds this limit, the API will return an HTTP 202 response, indicating that the request is still being processed. In such cases, you will receive a snapshot ID to monitor and retrieve the results asynchronously via the Monitor Snapshot and Download Snapshot endpoints. Example response on timeout:202
{
"snapshot_id": "s_xxx",
"message": "Your request is still in progress and cannot be retrieved in this call. Use the provided Snapshot ID to track progress via the Monitor Snapshot endpoint and download it once ready via the Download Snapshot endpoint."
}
Custom inputs
You can add custom fields to the input schema. Whatever you send in those fields is returned in the results for each record. Use this to:- Keep a unified output structure across different scrapers and datasets.
- Pass an
id,row_indexor any internal key so you can match results back to your original input rows.
Authorizations
Use your Bright Data API Key as a Bearer token in the Authorization header.
How to authenticate:
- Obtain your API Key from the Bright Data account settings at https://brightdata.com/cp/setting/users
- Include the API Key in the Authorization header of your requests
- Format:
Authorization: Bearer YOUR_API_KEY
Example:
Authorization: Bearer YOUR_API_KEY
Learn how to get your Bright Data API key: https://docs.brightdata.com/api-reference/authentication
Query Parameters
Dataset ID for which data collection is triggered.
List of output columns, separated by | (e.g., url|about.updated_on). Filters the response to include only the specified fields.
Example:
"url|about.updated_on"
Include errors report with the results.
Specifies the format of the response (default: ndjson).
Available options:
ndjson, json, csv Body
application/json
Response
OK
The response is of type string.
Example:
"OK"
Was this page helpful?