AIMLAPI Ocr API
The Ocr API from AIMLAPI — 1 operation(s) for ocr.
The Ocr API from AIMLAPI — 1 operation(s) for ocr.
Every API here is available over the APIs.io API and to AI agents over MCP.
One button, every client — Claude, Cursor, VS Code and the rest.
https://apis.io/mcp
find_apisBrowse and filter every API in the catalog.get_api_artifactsOne API's artifacts, grouped by type.get_openapiThe primary OpenAPI for this API.find_similar_apisAPIs that look like this one.apis_io_searchSTART HERE — APIs, providers and tags for one query, each with its total.resolveTurn a domain, URL or GitHub org into the provider it belongs to.find_cohortsEvery scored population of providers in the catalog.curl "https://apis.io/api/v1/apis/aimlapi-ocr-api"
curl "https://apis.io/api/v1/apis?limit=25"
Discovery needs no key. Ratings and market analysis are Pro.
Free tier, no form to fill in. Signing in shares your email address with us — we store it to create your key and to recognise you if you sign in with another provider. See our Privacy Policy and Terms.
A second provider on the same verified email joins the account you already have.
openapi: 3.2.0
info:
title: AIML Ocr API
version: 1.0.0
servers:
- url: https://api.aimlapi.com
tags:
- name: OCR
paths:
/v1/ocr:
post:
operationId: _v1_ocr
requestBody:
required: true
content:
application/json:
schema:
anyOf:
- type: object
properties:
model:
type: string
enum:
- gc-document-ai
- google/gc-document-ai
document:
anyOf:
- type: string
format: uri
- type: string
description: The document file to be processed by the OCR model.
mimeType:
type: string
enum:
- application/pdf
- image/gif
- image/tiff
- image/jpeg
- image/png
- image/bmp
- image/webp
- text/html
description: The MIME type of the document.
pages:
anyOf:
- type: object
properties:
type:
type: string
enum:
- start
start:
type: integer
minimum: 1
required:
- type
- start
- type: object
properties:
type:
type: string
enum:
- end
end:
type: integer
minimum: 1
required:
- type
- end
- type: object
properties:
type:
type: string
enum:
- range
start:
type: integer
minimum: 1
end:
type: integer
minimum: 2
required:
- type
- start
- end
- type: object
properties:
type:
type: string
enum:
- indices
indices:
type: array
items:
type: integer
minimum: 1
maxItems: 15
required:
- type
- indices
description: Specific pages you wants to process
required:
- model
- document
title: gc-document-ai, google/gc-document-ai
- type: object
properties:
model:
type: string
enum:
- glm-ocr
- zhipu/glm-ocr
document:
oneOf:
- type: object
properties:
type:
type: string
enum:
- document_url
description: Type of document.
document_url:
type: string
format: uri
description: 'URL of a document file to be processed by the OCR model. Supported file formats: PDF ≤ 50MB.'
required:
- type
- document_url
- type: object
properties:
type:
type: string
enum:
- image_url
description: Image URL.
image_url:
type: string
format: uri
description: 'URL of a single image to be processed by the OCR model. Supported file formats: JPG, PNG. Single image ≤10MB.'
required:
- type
- image_url
description: Document to run OCR.
pages:
anyOf:
- type: string
- type: array
items:
type: integer
description: Specific pages to process, e.g. "3", "0-2", [0, 3, 4].
include_image_base64:
type: boolean
description: Include base64 images in response.
image_limit:
type: integer
description: Max images to extract.
image_min_size:
type: integer
description: Minimum height and width of image to extract
return_crop_images:
type: boolean
description: Whether to return screenshot information.
need_layout_visualization:
type: boolean
description: Whether to return detailed layout image result information.
required:
- model
- document
title: glm-ocr, zhipu/glm-ocr
- type: object
properties:
model:
type: string
enum:
- test/dummy-ocr
document:
oneOf:
- type: object
properties:
type:
type: string
enum:
- document_url
document_url:
type: string
format: uri
required:
- type
- document_url
- type: object
properties:
type:
type: string
enum:
- image_url
image_url:
type: string
format: uri
required:
- type
- image_url
pages:
anyOf:
- type: string
- type: array
items:
type: integer
- {}
include_image_base64:
type:
- boolean
- 'null'
test:
type: object
properties:
delay:
type: number
pages:
type: integer
minimum: 1
maximum: 20
errorStatus:
type: number
required:
- model
- document
title: test/dummy-ocr
- type: object
properties:
model:
type: string
enum:
- mistral-ocr-latest
- mistral/mistral-ocr-latest
- mistral-ocr-2512
- mistral/mistral-ocr-2512
- mistral-ocr-4-0
- mistral/mistral-ocr-4-0
- mistral-ocr-3
- mistral/mistral-ocr-3
- mistral-ocr-4
- mistral/mistral-ocr-4
document:
oneOf:
- type: object
properties:
type:
type: string
enum:
- document_url
description: Type of document.
document_url:
type: string
format: uri
description: Document URL.
required:
- type
- document_url
- type: object
properties:
type:
type: string
enum:
- image_url
description: Image URL.
image_url:
type: string
format: uri
description: Type of document.
required:
- type
- image_url
description: Document to run OCR
pages:
anyOf:
- type: string
- type: array
items:
type: integer
- {}
description: Specific pages you wants to process
example: '"3" or "0-2" or [0, 3, 4]'
include_image_base64:
type:
- boolean
- 'null'
description: Include base64 images in response
image_limit:
type:
- integer
- 'null'
description: Max images to extract
image_min_size:
type:
- integer
- 'null'
description: Minimum height and width of image to extract
bbox_annotation_format:
type:
- object
- 'null'
properties:
type:
type: string
enum:
- json_schema
json_schema:
type: object
properties:
name:
type: string
schema:
type: object
additionalProperties: {}
description:
type:
- string
- 'null'
strict:
type:
- boolean
- 'null'
required:
- name
- schema
required:
- type
- json_schema
description: JSON schema to structure the annotation of each extracted bounding box (figures, charts, images). Using any annotation format switches the request to the annotated-page rate.
document_annotation_format:
type:
- object
- 'null'
properties:
type:
type: string
enum:
- json_schema
json_schema:
type: object
properties:
name:
type: string
schema:
type: object
additionalProperties: {}
description:
type:
- string
- 'null'
strict:
type:
- boolean
- 'null'
required:
- name
- schema
required:
- type
- json_schema
description: JSON schema to extract structured data from the whole document. Using any annotation format switches the request to the annotated-page rate.
document_annotation_prompt:
type:
- string
- 'null'
description: Optional high-level prompt to guide and instruct how the document is annotated.
required:
- model
- document
title: mistral-ocr-latest, mistral/mistral-ocr-latest, mistral-ocr-2512, mistral/mistral-ocr-2512, mistral-ocr-4-0, mistral/mistral-ocr-4-0, mistral-ocr-3, mistral/mistral-ocr-3, mistral-ocr-4, mistral/mistral-ocr-4
responses:
'200':
content:
application/json:
schema:
type: object
properties:
pages:
type: array
items:
type: object
properties:
index:
type: integer
description: The page index in a PDF document starting from 0
markdown:
type: string
description: The markdown string response of the page
images:
type: array
items:
type: object
properties:
id:
type: string
description: Image ID for extracted image in a page
top_left_x:
type:
- integer
- 'null'
description: X coordinate of top-left corner of the extracted image
top_left_y:
type:
- integer
- 'null'
description: Y coordinate of top-left corner of the extracted image
bottom_right_x:
type:
- integer
- 'null'
description: X coordinate of bottom-right corner of the extracted image
bottom_right_y:
type:
- integer
- 'null'
description: Y coordinate of bottom-right corner of the extracted image
image_base64:
type:
- string
- 'null'
format: uri
description: Base64 string of the extracted image
required:
- id
- top_left_x
- top_left_y
- bottom_right_x
- bottom_right_y
description: List of all extracted images in the page
dimensions:
type:
- object
- 'null'
properties:
dpi:
type: integer
description: Dots per inch of the page-image.
height:
type: integer
description: Height of the image in pixels.
width:
type: integer
description: Width of the image in pixels.
required:
- dpi
- height
- width
description: The dimensions of the PDF page's screenshot image
required:
- index
- markdown
- images
- dimensions
description: List of OCR info for pages
model:
type: string
description: The model used to generate the OCR.
document_annotation:
type:
- string
- 'null'
description: Structured annotation of the whole document as a JSON string, returned when document_annotation_format is provided.
usage_info:
type: object
properties:
pages_processed:
type: integer
description: Number of pages processed
doc_size_bytes:
type:
- integer
- 'null'
description: Document size in bytes
required:
- pages_processed
- doc_size_bytes
description: Usage info for the OCR request.
meta:
type:
- object
- 'null'
properties:
usage:
type:
- object
- 'null'
properties:
credits_used:
type: number
description: The number of tokens consumed during generation.
example: 120000
usd_spent:
type: number
description: The total amount of money spent by the user in USD.
example: 0.06
required:
- credits_used
- usd_spent
description: Additional details about the generation.
required:
- pages
- model
- usage_info
tags:
- OCR
summary: V1 ocr
x-summary-source: derived