curl --request POST \
--url https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
[
{
"id": "07281559-0695-0216-0000-c269be8b7592",
"url": "https://www.etsy.com/"
}
]
'import requests
url = "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content"
payload = [
{
"id": "07281559-0695-0216-0000-c269be8b7592",
"url": "https://www.etsy.com/"
}
]
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify([{id: '07281559-0695-0216-0000-c269be8b7592', url: 'https://www.etsy.com/'}])
};
fetch('https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
[
'id' => '07281559-0695-0216-0000-c269be8b7592',
'url' => 'https://www.etsy.com/'
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content"
payload := strings.NewReader("[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]"
response = http.request(request)
puts response.read_body{
"version": "<string>",
"status_code": 123,
"status_message": "<string>",
"time": "<string>",
"cost": 123,
"tasks_count": 123,
"tasks_error": 123,
"tasks": [
{
"id": "<string>",
"status_code": 123,
"status_message": "<string>",
"time": "<string>",
"cost": 123,
"result_count": 123,
"path": [
"<string>"
],
"data": {},
"result": [
{
"crawl_progress": "<string>",
"crawl_status": {
"max_crawl_pages": 123,
"pages_in_queue": 123,
"pages_crawled": 123
},
"items_count": 123,
"items": [
{
"url": "<string>",
"total_count": 123,
"pages": [
{
"similarity": 123,
"page": [
{
"resource_type": "<string>",
"status_code": 123,
"location": "<string>",
"url": "<string>",
"resource_errors": {
"errors": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
],
"warnings": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
]
},
"size": 123,
"encoded_size": 123,
"total_transfer_size": 123,
"fetch_time": "<string>",
"cache_control": {
"cachable": true,
"ttl": 123
},
"checks": {},
"content_encoding": "<string>",
"media_type": "<string>",
"server": "<string>",
"last_modified": {
"header": "<string>",
"sitemap": "<string>",
"meta_tag": "<string>"
},
"meta": {
"title": "<string>",
"charset": 123,
"follow": true,
"generator": "<string>",
"htags": {},
"description": "<string>",
"favicon": "<string>",
"meta_keywords": "<string>",
"canonical": "<string>",
"internal_links_count": 123,
"external_links_count": 123,
"inbound_links_count": 123,
"images_count": 123,
"images_size": 123,
"scripts_count": 123,
"scripts_size": 123,
"stylesheets_count": 123,
"stylesheets_size": 123,
"title_length": 123,
"description_length": 123,
"render_blocking_scripts_count": 123,
"render_blocking_stylesheets_count": 123,
"cumulative_layout_shift": 123,
"meta_title": "<string>",
"content": {
"plain_text_size": 123,
"plain_text_rate": 123,
"plain_text_word_count": 123,
"automated_readability_index": 123,
"coleman_liau_readability_index": 123,
"dale_chall_readability_index": 123,
"flesch_kincaid_readability_index": 123,
"smog_readability_index": 123,
"description_to_content_consistency": 123,
"title_to_content_consistency": 123,
"meta_keywords_to_content_consistency": 123
},
"deprecated_tags": [
"<string>"
],
"duplicate_meta_tags": [
"<string>"
],
"spell": {
"hunspell_language_code": "<string>",
"misspelled": [
{
"word": "<string>"
}
]
},
"social_media_tags": {},
"broken_html": {
"errors": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
],
"warnings": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
]
}
},
"page_timing": {
"time_to_interactive": 123,
"dom_complete": 123,
"largest_contentful_paint": 123,
"first_input_delay": 123,
"connection_time": 123,
"time_to_secure_connection": 123,
"request_sent_time": 123,
"waiting_time": 123,
"download_time": 123,
"duration_time": 123,
"fetch_start": 123,
"fetch_end": 123
},
"onpage_score": 123,
"total_dom_size": 123,
"custom_js_response": {},
"custom_js_client_exception": "<string>",
"broken_resources": true,
"broken_links": true,
"duplicate_title": true,
"duplicate_description": true,
"duplicate_content": true,
"click_depth": 123,
"is_resource": true,
"url_length": 123,
"relative_url_length": 123
}
]
}
]
}
]
}
]
}
]
}OnPage API 重复内容
一次爬取中与给定 url 内容相似的页面,相似度阈值由 similarity 决定。
curl --request POST \
--url https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
[
{
"id": "07281559-0695-0216-0000-c269be8b7592",
"url": "https://www.etsy.com/"
}
]
'import requests
url = "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content"
payload = [
{
"id": "07281559-0695-0216-0000-c269be8b7592",
"url": "https://www.etsy.com/"
}
]
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify([{id: '07281559-0695-0216-0000-c269be8b7592', url: 'https://www.etsy.com/'}])
};
fetch('https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
[
'id' => '07281559-0695-0216-0000-c269be8b7592',
'url' => 'https://www.etsy.com/'
]
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content"
payload := strings.NewReader("[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "[\n {\n \"id\": \"07281559-0695-0216-0000-c269be8b7592\",\n \"url\": \"https://www.etsy.com/\"\n }\n]"
response = http.request(request)
puts response.read_body{
"version": "<string>",
"status_code": 123,
"status_message": "<string>",
"time": "<string>",
"cost": 123,
"tasks_count": 123,
"tasks_error": 123,
"tasks": [
{
"id": "<string>",
"status_code": 123,
"status_message": "<string>",
"time": "<string>",
"cost": 123,
"result_count": 123,
"path": [
"<string>"
],
"data": {},
"result": [
{
"crawl_progress": "<string>",
"crawl_status": {
"max_crawl_pages": 123,
"pages_in_queue": 123,
"pages_crawled": 123
},
"items_count": 123,
"items": [
{
"url": "<string>",
"total_count": 123,
"pages": [
{
"similarity": 123,
"page": [
{
"resource_type": "<string>",
"status_code": 123,
"location": "<string>",
"url": "<string>",
"resource_errors": {
"errors": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
],
"warnings": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
]
},
"size": 123,
"encoded_size": 123,
"total_transfer_size": 123,
"fetch_time": "<string>",
"cache_control": {
"cachable": true,
"ttl": 123
},
"checks": {},
"content_encoding": "<string>",
"media_type": "<string>",
"server": "<string>",
"last_modified": {
"header": "<string>",
"sitemap": "<string>",
"meta_tag": "<string>"
},
"meta": {
"title": "<string>",
"charset": 123,
"follow": true,
"generator": "<string>",
"htags": {},
"description": "<string>",
"favicon": "<string>",
"meta_keywords": "<string>",
"canonical": "<string>",
"internal_links_count": 123,
"external_links_count": 123,
"inbound_links_count": 123,
"images_count": 123,
"images_size": 123,
"scripts_count": 123,
"scripts_size": 123,
"stylesheets_count": 123,
"stylesheets_size": 123,
"title_length": 123,
"description_length": 123,
"render_blocking_scripts_count": 123,
"render_blocking_stylesheets_count": 123,
"cumulative_layout_shift": 123,
"meta_title": "<string>",
"content": {
"plain_text_size": 123,
"plain_text_rate": 123,
"plain_text_word_count": 123,
"automated_readability_index": 123,
"coleman_liau_readability_index": 123,
"dale_chall_readability_index": 123,
"flesch_kincaid_readability_index": 123,
"smog_readability_index": 123,
"description_to_content_consistency": 123,
"title_to_content_consistency": 123,
"meta_keywords_to_content_consistency": 123
},
"deprecated_tags": [
"<string>"
],
"duplicate_meta_tags": [
"<string>"
],
"spell": {
"hunspell_language_code": "<string>",
"misspelled": [
{
"word": "<string>"
}
]
},
"social_media_tags": {},
"broken_html": {
"errors": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
],
"warnings": [
{
"line": 123,
"column": 123,
"message": "<string>",
"status_code": 123
}
]
}
},
"page_timing": {
"time_to_interactive": 123,
"dom_complete": 123,
"largest_contentful_paint": 123,
"first_input_delay": 123,
"connection_time": 123,
"time_to_secure_connection": 123,
"request_sent_time": 123,
"waiting_time": 123,
"download_time": 123,
"duration_time": 123,
"fetch_start": 123,
"fetch_end": 123
},
"onpage_score": 123,
"total_dom_size": 123,
"custom_js_response": {},
"custom_js_client_exception": "<string>",
"broken_resources": true,
"broken_links": true,
"duplicate_title": true,
"duplicate_description": true,
"duplicate_content": true,
"click_depth": 123,
"is_resource": true,
"url_length": 123,
"relative_url_length": 123
}
]
}
]
}
]
}
]
}
]
}url 内容相似的页面,相似度阈值由 similarity 决定。实测 606 字节。读的是一次已完成的爬取,因此需要 post_dataforseo_on_page_submit 返回的 id,并会带回 crawl_progress 和 crawl_status(max_crawl_pages、pages_in_queue、pages_crawled)——结果偏少时先看这几个字段,爬取还在跑就是本来没那么多。 要找标题或 meta description 重复而不是正文重复,用 post_dataforseo_on_page_duplicate_tags。## 示例
curl -X POST "https://api.aisa.one/apis/v1/dataforseo/on_page/duplicate_content" \
-H "Content-Type: application/json" \
-H "Authorization: Bearer YOUR_API_KEY" \
-d '[{"id": "...", "url": "...", "similarity": "..."}]'
// 把整段复制给 Claude Code、Codex、Cursor 或任意 coding agent。
// 连接、授权、跑通这个例子需要的东西都在里面 ——
// 不用粘贴任何 key,浏览器里点一次 Allow 就好。
{
"operation_id": "post_dataforseo_on_page_duplicate_content",
"arguments": {},
"max_price_usd": 0.05
}
// ── for the agent ─────────────────────────────────────────────────────────
// Register the AIsa MCP server, then call its `use` tool with the JSON above.
// name aisa
// endpoint https://mcp.aisa.one/mcp
// transport streamable-http
// auth OAuth. The server publishes RFC 9728 metadata, so open a
// browser and the user clicks Allow once — nothing to paste.
// If AISA_API_KEY is set, send it as a bearer header instead.
// Use your own MCP mechanism, not a hand-edited config file:
// Codex codex mcp add aisa --url https://mcp.aisa.one/mcp
// Claude Code claude mcp add --transport http -s user aisa https://mcp.aisa.one/mcp
// Cursor / VS Code a "url" entry for https://mcp.aisa.one/mcp in the MCP config
// The server lists five tools over https://mcp.aisa.one/servers worth of APIs:
// search, get_details, use, batch_use, list_categories. search and
// get_details are free; use is billed per call and max_price_usd refuses
// anything above the cap before spending. This operation's full contract —
// every argument, the response shape, the price and the pitfalls — is at
// https://aisa.one/docs/zh/api-reference/dataforseo/post_dataforseo-on-page-duplicate-content.md
// Then run the call and show me the result.
https://mcp.aisa.one/mcp —— Claude Code、
Codex、Cursor、VS Code 都可以。鉴权走 OAuth:客户端打开浏览器,你点一次
Allow,不需要粘贴任何 key。各客户端的具体命令和每次调用的价格见
aisa.one/zh-cn/mcp。授权
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
请求体
page URL
required field
specify the initial page you want to receive duplicate content for
content similarity score
by default, the content is considered duplicate if the value is greater than or equals 6
you can specify any similarity score in the 0-to-10 range
the maximum number of returned pages
optional field
default value: 100
maximum value: 1000
offset in the results array of returned pages
optional field
default value: 0
maximum value: 2000000
if you specify the 10 value, the first ten pages in the results array will be omitted and the data will be provided for the successive pages
user-defined task identifier
optional field
the character limit is 255
you can use this parameter to identify the task and match it with the result
you will find the specified tag value in the data object of the response
[
{
"id": "07281559-0695-0216-0000-c269be8b7592",
"url": "https://www.etsy.com/"
}
]
响应
Successful operation
API 的当前版本
general status code you can find the full list of the response codes here
general informational message you can find the full list of general informational messages here
total execution time, seconds
任务总成本(美元)
tasks 数组中的任务数量
返回错误的 tasks 数组中的任务数量
array of tasks
Show child attributes
Show child attributes