openapi: 3.2.0 info: title: Octen Ai Extract API version: 1.0.0 description: 'Operations tagged Extract across 2 of this provider''s published API definitions: octen-ai-openapi.json, octen-ai-openapi.yml. Each path carries the servers of the definition it was published in.' servers: - url: https://api.octen.ai security: - bearerAuth: [] - apiKeyAuth: [] tags: - name: Extract paths: /extract: post: summary: Extract description: Extracts clean markdown content from URLs. Supports batch processing, query-focused highlights, page classification, and multimedia resources. operationId: extract requestBody: required: true content: application/json: schema: $ref: '#/components/schemas/ExtractRequest' examples: basic: summary: Basic Extract value: urls: - https://octen.ai/ - https://docs.octen.ai/api-reference/search intentQuery: summary: Intent-focused Highlights with Query value: urls: - https://www.who.int/news-room/fact-sheets/detail/influenza-(seasonal) query: vaccination guidelines withMedia: summary: With Multimedia Resources value: urls: - https://octen.ai/ include_images: true include_videos: true include_audio: true advancedMode: summary: Advanced Mode value: urls: - https://x.com/kzouapt mode: advanced withLinks: summary: With Page Links value: urls: - https://octen.ai/ include_links: scope: prefer_internal max_links: 200 responses: '200': description: Successful extraction response content: application/json: schema: $ref: '#/components/schemas/ExtractResponse' example: code: 0 msg: success request_id: req_abc123def456 data: results: - url: https://octen.ai/ status: success resolved_mode: standard title: Octen | The Search infrastructure for AI full_content: '# Search infrastructure for AI Real-time indexing | Low latency | High reliability Start Building | View Docs ## Search Beyond Text Beyond text queries: Octen''s multimodal search understands images and videos alongside text...' highlights: null time_published: null time_last_crawled: '2026-07-03T09:35:43Z' page_structure: primary: Index Page secondary: Home Page category: primary: Computers, Electronics & Technology secondary: Search Engines favicon: https://octen.ai/favicon.ico cover_image: url: https://octen.ai/_next/static/media/octen-cover.dc74905e.png images: - url: https://octen.ai/_next/static/media/multi-modal-img.ccffddff.png - url: https://octen.ai/_next/static/media/showcase-multimodal1.449ae003.png videos: - url: https://octen.ai/static/video/multimodal-video.mp4 links: - url: https://octen.ai/pricing anchor_text: Pricing is_external: false - url: https://docs.octen.ai/overview/welcome anchor_text: View Docs is_external: false - url: https://github.com/Octen-Team/octen-skills anchor_text: GitHub is_external: true - url: https://docs.octen.ai/api-reference/search status: success resolved_mode: standard title: Search - Octen full_content: '# Search API Octen Search API enables ranked web results with query-focused highlights, time filtering, and multimodal assets...' highlights: null time_published: '2024-10-15T00:00:00Z' time_last_crawled: '2026-07-03T09:34:29Z' page_structure: primary: Content Page secondary: Code category: primary: Computers, Electronics & Technology secondary: Search Engines favicon: https://docs.octen.ai/favicon.ico cover_image: url: https://octen.ai/_next/static/media/octen-cover.dc74905e.png meta: usage: total_urls: 2 successful_urls: 2 successful_by_mode: standard_urls: 2 advanced_urls: 0 latency: 1832 warning: '' '400': description: Invalid params — Returned when a required parameter is missing or invalid. content: application/json: schema: $ref: '#/components/schemas/ErrorResponse' example: code: 400 msg: Invalid params. Missing parameter urls request_id: req_abc123def456 '401': $ref: '#/components/responses/Unauthorized' '403': $ref: '#/components/responses/InsufficientBalance' '429': $ref: '#/components/responses/RateLimited' '500': $ref: '#/components/responses/InternalError' tags: - Extract servers: - url: https://api.octen.ai components: schemas: ExtractPageStructure: type: object description: Detected page structure for an extraction result. properties: primary: type: string enum: - Index Page - Content Page - No Main Content description: Top-level page type. secondary: type: string nullable: true description: Sub-type within the primary structure. 50+ possible values. `null` when no sub-type applies. ExtractResponse: type: object properties: code: type: integer description: Business status code. 0 indicates success. msg: type: string description: A message describing the result. request_id: type: string description: The unique identifier for this request. data: $ref: '#/components/schemas/ExtractData' meta: $ref: '#/components/schemas/ExtractMeta' ExtractRequest: type: object required: - urls description: Request body for the Extract API. properties: urls: type: array items: type: string description: 'List of URLs to extract content from. Maximum URLs per request: 20. Maximum length per URL: 2048. Failed URLs are not billed.' example: - https://example.com/article-1 - https://example.com/article-2 mode: type: string enum: - standard - advanced - auto default: standard description: Processing mode. `standard` prioritizes speed, `advanced` prioritizes success rate on hard-to-reach pages, and `auto` picks one per URL. `advanced` and `auto` can take longer, so raise `timeout` accordingly. query: type: string maxLength: 500 description: Intent-focused keywords. When provided, returns query-relevant highlights per URL; otherwise returns the complete page content. max_age_seconds: type: integer default: 86400 minimum: 300 maximum: 31536000 description: Maximum age (in seconds) of cached content. URLs whose cached version exceeds this threshold will be re-fetched. Values outside the allowed range are adjusted to the nearest bound. format: type: string enum: - markdown - text default: markdown description: Format of the returned content. timeout: type: integer default: 30 minimum: 1 maximum: 60 description: Per-URL extraction timeout in seconds. Values outside the allowed range are adjusted to the nearest bound. include_images: type: boolean default: false description: Whether to return image URLs detected on the page. include_videos: type: boolean default: false description: Whether to return video URLs detected on the page. include_audio: type: boolean default: false description: Whether to return audio URLs detected on the page. include_links: type: object description: Controls whether to return links detected on the page. properties: scope: type: string enum: - prefer_internal - prefer_external default: prefer_internal description: Which links to prioritize. `prefer_internal` favors links within the page's registered domain; `prefer_external` favors external links. Prioritized links come first, and links of the same kind keep their order on the page. max_links: type: integer default: 200 minimum: 1 maximum: 1000 description: Maximum number of links to return per URL. ExtractMediaResource: type: object description: A single multimedia resource detected on the page. properties: url: type: string description: Resource URL. ExtractData: type: object description: The main extract response payload. properties: results: type: array description: Extraction result for each requested URL. Order matches the input urls array. items: $ref: '#/components/schemas/ExtractResult' ExtractCategory: type: object description: Detected content category for an extraction result. properties: primary: type: string description: Top-level content category. 23 possible values. secondary: type: string nullable: true description: Sub-category within the primary category. 160+ possible values. `null` when no sub-category applies. ExtractLink: type: object description: A single link found on the page. properties: url: type: string description: Absolute URL of the link target. anchor_text: type: string description: The link's visible text on the page. Empty when the link has no visible text. is_external: type: boolean description: Whether the link points to a different registered domain than the page. ErrorResponse: type: object properties: code: type: integer description: Business status code. Non-zero values indicate an error. msg: type: string description: A message describing the error. request_id: type: string description: Unique identifier for the request. required: - code - msg - request_id ExtractResult: type: object description: 'A single extraction result. Batch requests may return 200 OK overall while individual items fail; failed items are marked with `status: "failed"` and an `error_message`, and are not billed.' properties: url: type: string description: The requested URL. status: type: string enum: - success - failed description: Extraction status for this URL. resolved_mode: type: string enum: - standard - advanced description: The mode this URL is billed at. For `auto`, the mode chosen for this URL. Returned when `status` is `success`. title: type: string nullable: true description: Page title, extracted from HTML `