openapi: 3.2.0 info: title: AIML Ocr API version: 1.0.0 servers: - url: https://api.aimlapi.com tags: - name: OCR paths: /v1/ocr: post: operationId: _v1_ocr requestBody: required: true content: application/json: schema: anyOf: - type: object properties: model: type: string enum: - gc-document-ai - google/gc-document-ai document: anyOf: - type: string format: uri - type: string description: The document file to be processed by the OCR model. mimeType: type: string enum: - application/pdf - image/gif - image/tiff - image/jpeg - image/png - image/bmp - image/webp - text/html description: The MIME type of the document. pages: anyOf: - type: object properties: type: type: string enum: - start start: type: integer minimum: 1 required: - type - start - type: object properties: type: type: string enum: - end end: type: integer minimum: 1 required: - type - end - type: object properties: type: type: string enum: - range start: type: integer minimum: 1 end: type: integer minimum: 2 required: - type - start - end - type: object properties: type: type: string enum: - indices indices: type: array items: type: integer minimum: 1 maxItems: 15 required: - type - indices description: Specific pages you wants to process required: - model - document title: gc-document-ai, google/gc-document-ai - type: object properties: model: type: string enum: - glm-ocr - zhipu/glm-ocr document: oneOf: - type: object properties: type: type: string enum: - document_url description: Type of document. document_url: type: string format: uri description: 'URL of a document file to be processed by the OCR model. Supported file formats: PDF ≤ 50MB.' required: - type - document_url - type: object properties: type: type: string enum: - image_url description: Image URL. image_url: type: string format: uri description: 'URL of a single image to be processed by the OCR model. Supported file formats: JPG, PNG. Single image ≤10MB.' required: - type - image_url description: Document to run OCR. pages: anyOf: - type: string - type: array items: type: integer description: Specific pages to process, e.g. "3", "0-2", [0, 3, 4]. include_image_base64: type: boolean description: Include base64 images in response. image_limit: type: integer description: Max images to extract. image_min_size: type: integer description: Minimum height and width of image to extract return_crop_images: type: boolean description: Whether to return screenshot information. need_layout_visualization: type: boolean description: Whether to return detailed layout image result information. required: - model - document title: glm-ocr, zhipu/glm-ocr - type: object properties: model: type: string enum: - test/dummy-ocr document: oneOf: - type: object properties: type: type: string enum: - document_url document_url: type: string format: uri required: - type - document_url - type: object properties: type: type: string enum: - image_url image_url: type: string format: uri required: - type - image_url pages: anyOf: - type: string - type: array items: type: integer - {} include_image_base64: type: - boolean - 'null' test: type: object properties: delay: type: number pages: type: integer minimum: 1 maximum: 20 errorStatus: type: number required: - model - document title: test/dummy-ocr - type: object properties: model: type: string enum: - mistral-ocr-latest - mistral/mistral-ocr-latest - mistral-ocr-2512 - mistral/mistral-ocr-2512 - mistral-ocr-4-0 - mistral/mistral-ocr-4-0 - mistral-ocr-3 - mistral/mistral-ocr-3 - mistral-ocr-4 - mistral/mistral-ocr-4 document: oneOf: - type: object properties: type: type: string enum: - document_url description: Type of document. document_url: type: string format: uri description: Document URL. required: - type - document_url - type: object properties: type: type: string enum: - image_url description: Image URL. image_url: type: string format: uri description: Type of document. required: - type - image_url description: Document to run OCR pages: anyOf: - type: string - type: array items: type: integer - {} description: Specific pages you wants to process example: '"3" or "0-2" or [0, 3, 4]' include_image_base64: type: - boolean - 'null' description: Include base64 images in response image_limit: type: - integer - 'null' description: Max images to extract image_min_size: type: - integer - 'null' description: Minimum height and width of image to extract bbox_annotation_format: type: - object - 'null' properties: type: type: string enum: - json_schema json_schema: type: object properties: name: type: string schema: type: object additionalProperties: {} description: type: - string - 'null' strict: type: - boolean - 'null' required: - name - schema required: - type - json_schema description: JSON schema to structure the annotation of each extracted bounding box (figures, charts, images). Using any annotation format switches the request to the annotated-page rate. document_annotation_format: type: - object - 'null' properties: type: type: string enum: - json_schema json_schema: type: object properties: name: type: string schema: type: object additionalProperties: {} description: type: - string - 'null' strict: type: - boolean - 'null' required: - name - schema required: - type - json_schema description: JSON schema to extract structured data from the whole document. Using any annotation format switches the request to the annotated-page rate. document_annotation_prompt: type: - string - 'null' description: Optional high-level prompt to guide and instruct how the document is annotated. required: - model - document title: mistral-ocr-latest, mistral/mistral-ocr-latest, mistral-ocr-2512, mistral/mistral-ocr-2512, mistral-ocr-4-0, mistral/mistral-ocr-4-0, mistral-ocr-3, mistral/mistral-ocr-3, mistral-ocr-4, mistral/mistral-ocr-4 responses: '200': content: application/json: schema: type: object properties: pages: type: array items: type: object properties: index: type: integer description: The page index in a PDF document starting from 0 markdown: type: string description: The markdown string response of the page images: type: array items: type: object properties: id: type: string description: Image ID for extracted image in a page top_left_x: type: - integer - 'null' description: X coordinate of top-left corner of the extracted image top_left_y: type: - integer - 'null' description: Y coordinate of top-left corner of the extracted image bottom_right_x: type: - integer - 'null' description: X coordinate of bottom-right corner of the extracted image bottom_right_y: type: - integer - 'null' description: Y coordinate of bottom-right corner of the extracted image image_base64: type: - string - 'null' format: uri description: Base64 string of the extracted image required: - id - top_left_x - top_left_y - bottom_right_x - bottom_right_y description: List of all extracted images in the page dimensions: type: - object - 'null' properties: dpi: type: integer description: Dots per inch of the page-image. height: type: integer description: Height of the image in pixels. width: type: integer description: Width of the image in pixels. required: - dpi - height - width description: The dimensions of the PDF page's screenshot image required: - index - markdown - images - dimensions description: List of OCR info for pages model: type: string description: The model used to generate the OCR. document_annotation: type: - string - 'null' description: Structured annotation of the whole document as a JSON string, returned when document_annotation_format is provided. usage_info: type: object properties: pages_processed: type: integer description: Number of pages processed doc_size_bytes: type: - integer - 'null' description: Document size in bytes required: - pages_processed - doc_size_bytes description: Usage info for the OCR request. meta: type: - object - 'null' properties: usage: type: - object - 'null' properties: credits_used: type: number description: The number of tokens consumed during generation. example: 120000 usd_spent: type: number description: The total amount of money spent by the user in USD. example: 0.06 required: - credits_used - usd_spent description: Additional details about the generation. required: - pages - model - usage_info tags: - OCR summary: V1 ocr x-summary-source: derived