> ## Documentation Index
> Fetch the complete documentation index at: https://aisa.one/docs/llms.txt
> Use this file to discover all available pages before exploring further.

# extract

> Extracts relevant content from specific web URLs.



## OpenAPI

````yaml openapi/zh/parallel.ai.json POST /apis/v1/parallel/extract
openapi: 3.1.0
info:
  description: ''
  title: Parallel.ai
  version: '1'
  x-aisa-capabilities:
    idempotency: Idempotency-Key
    max_price: X-AISA-Max-Price-USD
    quote:
      header: X-AISA-Cost-Mode
      value: quote
  x-aisa-configured-paths:
    - /apis/v1/parallel/chat/completions
    - /apis/v1/parallel/extract
    - /apis/v1/parallel/findall/ingest
    - /apis/v1/parallel/findall/runs/{findall_id}
    - /apis/v1/parallel/findall/runs/{findall_id}/cancel
    - /apis/v1/parallel/findall/runs/{findall_id}/events
    - /apis/v1/parallel/findall/runs/{findall_id}/result
    - /apis/v1/parallel/findall/runs/{findall_id}/schema
    - /apis/v1/parallel/monitors/{monitor_id}
    - /apis/v1/parallel/monitors/{monitor_id}/cancel
    - /apis/v1/parallel/monitors/{monitor_id}/events
    - /apis/v1/parallel/monitors/{monitor_id}/update
    - /apis/v1/parallel/tasks/groups
    - /apis/v1/parallel/tasks/groups/{taskgroup_id}
    - /apis/v1/parallel/tasks/groups/{taskgroup_id}/events
    - /apis/v1/parallel/tasks/groups/{taskgroup_id}/runs/{run_id}
    - /apis/v1/parallel/tasks/runs/{run_id}
    - /apis/v1/parallel/tasks/runs/{run_id}/events
    - /apis/v1/parallel/tasks/runs/{run_id}/input
    - /apis/v1/parallel/tasks/runs/{run_id}/result
    - /apis/v1/parallel/v1beta/tasks/runs/{run_id}/events
  x-aisa-document:
    facts_hash: 7d3bf2025708131927d3136de215ba314114253046b35f239876910b2481cbd9
    generator_version: '2'
    protocol_version: '1'
    schema_version: '1'
    composer_version: '13'
    response_pending: []
    document_hash: sha256:dae9e982424f710c06d9af023168dd26ed96bf5e5b5232e0728c01867d2da7dc
  x-aisa-plans:
    builder: 1
    display_plan: payg
    payg: 1
    similarweb_payg: 1
    team: 1
    version: e70a0959cd1b93b84ae16dffd7a538273ddb9c1d46277b82823c37ff1b6bc0fb
  x-aisa-provider: parallel.ai
  x-aisa-catalogs:
    parallel.ai:
      title: Parallel.ai
      description: ''
      x-aisa-document:
        facts_hash: 7d3bf2025708131927d3136de215ba314114253046b35f239876910b2481cbd9
        generator_version: '2'
        protocol_version: '1'
        schema_version: '1'
      x-aisa-plans:
        builder: 1
        display_plan: payg
        payg: 1
        similarweb_payg: 1
        team: 1
        version: e70a0959cd1b93b84ae16dffd7a538273ddb9c1d46277b82823c37ff1b6bc0fb
      x-aisa-capabilities:
        idempotency: Idempotency-Key
        max_price: X-AISA-Max-Price-USD
        quote:
          header: X-AISA-Cost-Mode
          value: quote
servers:
  - url: https://api.aisa.one
security:
  - bearerAuth: []
paths:
  /apis/v1/parallel/extract:
    post:
      tags:
        - default
      summary: extract
      description: Extracts relevant content from specific web URLs.
      operationId: any_parallel_ai_apis_v1_parallel_extract
      requestBody:
        content:
          application/json:
            schema:
              properties:
                urls:
                  items:
                    type: string
                  type: array
                  title: Urls
                  description: URLs to extract content from. Up to 20 URLs.
                objective:
                  anyOf:
                    - type: string
                    - type: 'null'
                  title: Objective
                  description: >-
                    As in SearchRequest, a natural-language description of the
                    underlying question or goal driving the request. Used
                    together with search_queries to focus excerpts on the most
                    relevant content.
                search_queries:
                  anyOf:
                    - items:
                        type: string
                      type: array
                    - type: 'null'
                  title: Search Queries
                  description: >-
                    Optional keyword search queries, as in SearchRequest. Used
                    together with objective to focus excerpts on the most
                    relevant content.
                max_chars_total:
                  anyOf:
                    - type: integer
                    - type: 'null'
                  title: Max Chars Total
                  description: >-
                    Upper bound on total characters across excerpts from all
                    extracted results.
                session_id:
                  anyOf:
                    - type: string
                      maxLength: 1000
                    - type: 'null'
                  title: Session Id
                  description: >-
                    Session identifier to track calls across separate search and
                    extract calls, to be used as part of a larger task.
                    Specifying it may give better contextual results for
                    subsequent API calls.
                client_model:
                  anyOf:
                    - type: string
                    - type: 'null'
                  title: Client Model
                  description: >-
                    The model generating this request and consuming the results.
                    Enables optimizations and tailors default settings for the
                    model's capabilities.
                  examples:
                    - claude-opus-4-7
                    - gpt-5.4
                    - gemini-3.1-pro
                advanced_settings:
                  anyOf:
                    - properties:
                        fetch_policy:
                          anyOf:
                            - properties:
                                max_age_seconds:
                                  anyOf:
                                    - type: integer
                                    - type: 'null'
                                  title: Max Age Seconds
                                  description: >-
                                    Maximum age of cached content in seconds to
                                    trigger a live fetch. Minimum value 600
                                    seconds (10 minutes).
                                  examples:
                                    - 86400
                                timeout_seconds:
                                  anyOf:
                                    - type: number
                                    - type: 'null'
                                  title: Timeout Seconds
                                  description: >-
                                    Timeout in seconds for fetching live content
                                    if unavailable in cache.
                                  examples:
                                    - 60
                                disable_cache_fallback:
                                  type: boolean
                                  title: Disable Cache Fallback
                                  description: >-
                                    If false, fallback to cached content older
                                    than max-age if live fetch fails or times
                                    out. If true, returns an error instead.
                                  default: false
                              type: object
                              title: FetchPolicy
                              description: Policy for live fetching web results.
                            - type: 'null'
                          description: >-
                            Fetch policy: determines when to return cached
                            content from the index (faster) vs fetching live
                            content (fresher). Default is to use a dynamic
                            policy based on the search objective and url. Note:
                            enabling live fetch significantly increases extract
                            latency because it requires fetching content from
                            source websites.
                        excerpt_settings:
                          anyOf:
                            - properties:
                                max_chars_per_result:
                                  anyOf:
                                    - type: integer
                                    - type: 'null'
                                  title: Max Chars Per Result
                                  description: >-
                                    Optional upper bound on the total number of
                                    characters to include per url. Excerpts may
                                    contain fewer characters than this limit to
                                    maximize relevance and token efficiency.
                              additionalProperties: false
                              type: object
                              title: V1ExcerptSettings
                              description: >-
                                Optional settings for returning relevant
                                excerpts.
                            - type: 'null'
                          description: >-
                            Controls excerpt sizes. Provide excerpt settings for
                            fine-grained control, or omit to use defaults.
                        full_content:
                          anyOf:
                            - properties:
                                max_chars_per_result:
                                  anyOf:
                                    - type: integer
                                    - type: 'null'
                                  title: Max Chars Per Result
                                  description: >-
                                    Optional limit on the number of characters
                                    to include in the full content for each url.
                                    Full content always starts at the beginning
                                    of the page and is truncated at the limit if
                                    necessary.
                              type: object
                              title: FullContentSettings
                              description: Optional settings for returning full content.
                            - type: boolean
                          title: Full Content
                          description: >-
                            Controls full content extraction. Set to true to
                            enable with defaults, false to disable, or provide
                            FullContentSettings for fine-grained control.
                          default: false
                      additionalProperties: false
                      type: object
                      title: AdvancedExtractSettings
                      description: >-
                        Advanced extract configuration.


                        These settings may impact result quality and latency
                        unless used carefully.

                        See
                        https://docs.parallel.ai/search/advanced-extract-settings
                        for more info.
                    - type: 'null'
                  description: >-
                    Advanced configuration for fetch policy, excerpt settings,
                    and full content settings. May impact result quality and
                    latency unless used carefully. When omitted, excerpts are
                    enabled and full content is disabled by default.
              additionalProperties: false
              type: object
              required:
                - urls
              title: V1ExtractRequest
              description: Extract request.
        required: true
      responses:
        '200':
          description: Successful Response
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResponseB67C75A8Feba_V1ExtractResponse'
              example:
                extract_id: extract_8a911eb27c7a4afaa20d0d9dc98d07c0
                results:
                  - url: https://www.example.com
                    title: Example Title
                    excerpts:
                      - Excerpted text ...
                    full_content: Full content ...
                errors:
                  - url: https://www.example.com
                    error_type: fetch_error
                    http_status_code: 500
                    content: Error fetching content from https://www.example.com
                session_id: session_8a911eb27c7a4afaa20d0d9dc98d07c0
        default:
          content:
            application/json:
              schema: {}
          description: >-
            Error response; upstream passthrough responses may use provider
            formats
components:
  schemas:
    ResponseB67C75A8Feba_V1ExtractResponse:
      properties:
        extract_id:
          type: string
          title: Extract Id
          description: Extract request ID, e.g. `extract_cad0a6d2dec046bd95ae900527d880e7`
        results:
          items:
            $ref: '#/components/schemas/ResponseB67C75A8Feba_V1ExtractResult'
          type: array
          title: Results
          description: Successful extract results.
        errors:
          items:
            $ref: '#/components/schemas/ResponseB67C75A8Feba_ExtractError'
          type: array
          title: Errors
          description: 'Extract errors: requested URLs not in the results.'
        warnings:
          anyOf:
            - items:
                $ref: '#/components/schemas/ResponseB67C75A8Feba_Warning'
              type: array
            - type: 'null'
          title: Warnings
          description: Warnings for the extract request, if any.
        usage:
          anyOf:
            - items:
                $ref: '#/components/schemas/ResponseB67C75A8Feba_UsageItem'
              type: array
            - type: 'null'
          title: Usage
          description: Usage metrics for the extract request.
        session_id:
          type: string
          title: Session Id
          description: >-
            Session identifier. Echoed back from the request if provided,
            otherwise generated by the server. Should be passed to future search
            and extract calls made by the agent as part of the same larger task.
          examples:
            - session_8a911eb27c7a4afaa20d0d9dc98d07c0
      type: object
      required:
        - extract_id
        - results
        - errors
        - session_id
      title: V1ExtractResponse
      description: Extract response.
    ResponseB67C75A8Feba_V1ExtractResult:
      properties:
        url:
          type: string
          title: Url
          description: URL associated with the search result.
        title:
          anyOf:
            - type: string
            - type: 'null'
          title: Title
          description: Title of the webpage, if available.
        publish_date:
          anyOf:
            - type: string
            - type: 'null'
          title: Publish Date
          description: Publish date of the webpage in YYYY-MM-DD format, if available.
        excerpts:
          items:
            type: string
          type: array
          title: Excerpts
          description: Relevant excerpted content from the URL, formatted as markdown.
        full_content:
          anyOf:
            - type: string
            - type: 'null'
          title: Full Content
          description: Full content from the URL formatted as markdown, if requested.
      type: object
      required:
        - url
        - excerpts
      title: V1ExtractResult
      description: Extract result for a single URL.
    ResponseB67C75A8Feba_ExtractError:
      properties:
        url:
          type: string
          title: Url
        error_type:
          type: string
          title: Error Type
          description: Error type.
        http_status_code:
          anyOf:
            - type: integer
            - type: 'null'
          title: Http Status Code
          description: HTTP status code, if available.
        content:
          anyOf:
            - type: string
            - type: 'null'
          title: Content
          description: Content returned for http client or server errors, if any.
      type: object
      required:
        - url
        - error_type
        - http_status_code
        - content
      title: ExtractError
      description: Extract error details.
    ResponseB67C75A8Feba_Warning:
      properties:
        type:
          type: string
          enum:
            - spec_validation_warning
            - input_validation_warning
            - warning
          title: Type
          description: >-
            Type of warning. Note that adding new warning types is considered a
            backward-compatible change.
          examples:
            - spec_validation_warning
            - input_validation_warning
        message:
          type: string
          title: Message
          description: Human-readable message.
        detail:
          anyOf:
            - additionalProperties: true
              type: object
            - type: 'null'
          title: Detail
          description: Optional detail supporting the warning.
      type: object
      required:
        - type
        - message
      title: Warning
      description: Human-readable message for a task.
    ResponseB67C75A8Feba_UsageItem:
      properties:
        name:
          type: string
          title: Name
          description: Name of the SKU.
          examples:
            - sku_search
            - sku_extract_excerpts
        count:
          type: integer
          title: Count
          description: Count of the SKU.
          examples:
            - 1
      type: object
      required:
        - name
        - count
      title: UsageItem
      description: Usage item for a single operation.
  securitySchemes:
    bearerAuth:
      scheme: bearer
      type: http

````

This documentation is built and hosted on [Mintlify](https://mintlify.com), a developer documentation platform.