openapi: 3.0.3 info: title: Spider Cloud Crawling API version: v1 description: 'Spider Cloud is a Rust-based, AI-friendly web scraping and crawling platform. The REST API accepts and returns JSON (also XML, CSV, JSONL via the content-type header) and authenticates with a Bearer API key. Every account is allowed up to 10,000 core API requests per minute. ' contact: name: Spider Cloud Support url: https://spider.cloud email: support@spider.cloud license: name: Proprietary url: https://spider.cloud servers: - url: https://api.spider.cloud description: Spider Cloud production API security: - bearerAuth: [] tags: - name: Crawling description: Recursively crawl entire websites and collect every page. paths: /crawl: post: tags: - Crawling summary: Crawl a Website operationId: crawl description: Recursively crawl an entire website and return every page as clean markdown, JSON, HTML, or another supported format. requestBody: required: true content: application/json: schema: $ref: '#/components/schemas/CrawlRequest' responses: '200': description: Crawl result payload. content: application/json: schema: $ref: '#/components/schemas/CrawlResponse' '401': $ref: '#/components/responses/Unauthorized' '429': $ref: '#/components/responses/Throttled' components: schemas: CrawlResponse: type: object properties: pages: type: array items: type: object properties: url: type: string status: type: integer content: type: string metadata: type: object additionalProperties: true Error: type: object properties: error: type: string code: type: string message: type: string CrawlRequest: type: object required: - url properties: url: type: string description: The seed URL to crawl. limit: type: integer description: Maximum number of pages to crawl. depth: type: integer description: Maximum crawl depth from the seed URL. return_format: type: string enum: - markdown - html - text - json - raw respect_robots_txt: type: boolean default: true proxy_enabled: type: boolean stealth: type: boolean chrome: type: boolean description: Use a full headless browser instead of HTTP-only fetcher. responses: Unauthorized: description: Missing or invalid API key. content: application/json: schema: $ref: '#/components/schemas/Error' Throttled: description: Rate limit exceeded (10,000 RPM per account). content: application/json: schema: $ref: '#/components/schemas/Error' securitySchemes: bearerAuth: type: http scheme: bearer bearerFormat: API Key