获取页面内容
curl --request GET \
--url https://api.olostep.com/v1/retrieve \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.olostep.com/v1/retrieve"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.olostep.com/v1/retrieve', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.olostep.com/v1/retrieve",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.olostep.com/v1/retrieve"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.olostep.com/v1/retrieve")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.olostep.com/v1/retrieve")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"html_content": "<string>",
"markdown_content": "<string>",
"json_content": "<string>",
"html_hosted_url": "<string>",
"markdown_hosted_url": "<string>",
"json_hosted_url": "<string>",
"size_exceeded": true
}检索
检索内容
检索已处理批次和抓取URL的内容。
GET
/
v1
/
retrieve
获取页面内容
curl --request GET \
--url https://api.olostep.com/v1/retrieve \
--header 'Authorization: Bearer <token>'import requests
url = "https://api.olostep.com/v1/retrieve"
headers = {"Authorization": "Bearer <token>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: 'Bearer <token>'}};
fetch('https://api.olostep.com/v1/retrieve', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.olostep.com/v1/retrieve",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.olostep.com/v1/retrieve"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.olostep.com/v1/retrieve")
.header("Authorization", "Bearer <token>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.olostep.com/v1/retrieve")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = 'Bearer <token>'
response = http.request(request)
puts response.read_body{
"html_content": "<string>",
"markdown_content": "<string>",
"json_content": "<string>",
"html_hosted_url": "<string>",
"markdown_hosted_url": "<string>",
"json_hosted_url": "<string>",
"size_exceeded": true
}授权
Bearer认证头格式为Bearer ,其中是你的认证令牌。
查询参数
要获取的页面内容的 ID。可在 /v1/crawls/{crawl_id}/pages、/v1/scrapes/{scrape_id} 或 /v1/batches/{batch_id}/items 端点的响应中找到
可选数组,用于在生产中仅获取特定格式。如果未提供,将返回所有格式。
可用选项:
html, markdown, json 响应
成功响应页面内容。
页面 HTML 内容(如果请求且可用)。
页面 Markdown 内容(如果请求且可用)。
从解析器返回的页面 JSON 内容(如果请求且可用)。
HTML 的 S3 存储桶 URL。7 天后过期。
Markdown 的 S3 存储桶 URL。7 天后过期。
JSON 的 S3 存储桶 URL。7 天后过期。
如果内容对象的大小超过 6MB 限制。如果为 true,请使用托管的 S3 urls 获取内容。
此页面对您有帮助吗?