Docs To Markdown
Pricing
from $1.00 / 1,000 results
Docs To Markdown
Pricing
from $1.00 / 1,000 results
You can access the Docs To Markdown programmatically from your own applications by using the Apify API. You can also choose the language preference from below. To use the Apify API, you’ll need an Apify account and your API token, found in API & Integrations in Apify Console.
{ "openapi": "3.0.1", "info": { "version": "0.0", "x-build-id": "Fa2K3blTtfhDmsj9o" }, "servers": [ { "url": "https://api.apify.com/v2" } ], "paths": { "/acts/excellent_mustang~docs-to-markdown/run-sync-get-dataset-items": { "post": { "operationId": "run-sync-get-dataset-items-excellent_mustang-docs-to-markdown", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } }, "/acts/excellent_mustang~docs-to-markdown/runs": { "post": { "operationId": "runs-sync-excellent_mustang-docs-to-markdown", "x-openai-isConsequential": false, "summary": "Executes an Actor and returns information about the initiated run in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/runsResponseSchema" } } } } } } }, "/acts/excellent_mustang~docs-to-markdown/run-sync": { "post": { "operationId": "run-sync-excellent_mustang-docs-to-markdown", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } } }, "components": { "schemas": { "inputSchema": { "type": "object", "required": [ "startUrls" ], "properties": { "startUrls": { "title": "Documentation URLs", "type": "array", "description": "Where to start crawling. Paste the top of the docs section you want, for example https://docs.example.com/guide. By default the crawl stays inside that path, so a run started at /guide will not wander into /blog or /pricing.", "default": [ { "url": "https://docs.apify.com/academy" } ], "items": { "type": "object", "required": [ "url" ], "properties": { "url": { "type": "string", "title": "URL of a web page", "format": "uri" } } } }, "maxPages": { "title": "Max pages", "minimum": 1, "maximum": 10000, "type": "integer", "description": "Hard ceiling on pages fetched. This is a real limit, not a hint: the crawl stops the moment it is reached. Kept deliberately low by default so an exploratory run cannot turn into a surprise bill.", "default": 25 }, "maxCrawlDepth": { "title": "Max crawl depth", "minimum": 0, "maximum": 20, "type": "integer", "description": "How many link hops from a start URL to follow. 0 crawls only the start URLs themselves.", "default": 3 }, "crawlWholeDomain": { "title": "Crawl the whole domain", "type": "boolean", "description": "Ignore the start URL's path and allow any page on the same domain. Leave this off to keep a docs crawl inside the docs.", "default": false }, "urlPrefixes": { "title": "URL prefixes (advanced)", "type": "array", "description": "Override the automatic scope with an explicit list of prefixes. A page is crawled only if its URL starts with one of them.", "default": [], "items": { "type": "string" } }, "excludeUrlPatterns": { "title": "Exclude URL patterns", "type": "array", "description": "Regular expressions. Any URL matching one of them is skipped. Useful for changelogs, tag pages, or the /v1/ copy of versioned docs you do not want duplicated in your vector database.", "default": [], "items": { "type": "string" } }, "useSitemap": { "title": "Also load URLs from sitemap.xml", "type": "boolean", "description": "Read /sitemap.xml and queue every in-scope URL it lists, in addition to following links. Sitemap URLs still obey the scope prefixes and the max pages limit.", "default": false }, "minCoverageRatio": { "title": "Minimum coverage ratio", "minimum": 0, "maximum": 1, "type": "number", "description": "Coverage is the share of a page's available content text that survived extraction. Below this value the page is flagged with a warning, and the extractor retries with a more permissive strategy before giving up. Raise it to be stricter about partial extractions.", "default": 0.25 }, "skipLowCoveragePages": { "title": "Drop pages that fail the coverage check", "type": "boolean", "description": "Leave off to keep every page with its warning attached, which is usually what you want while you are still tuning. Turn on to keep the dataset clean for a production ingest.", "default": false }, "includeHtml": { "title": "Include raw HTML", "type": "boolean", "description": "Add the original HTML of each page to the output. Useful for auditing a page the coverage check flagged. Makes the dataset much larger.", "default": false }, "concurrency": { "title": "Concurrency", "minimum": 1, "maximum": 20, "type": "integer", "description": "How many pages to fetch in parallel. Lower this if the documentation site rate-limits you.", "default": 8 }, "requestTimeoutSecs": { "title": "Request timeout", "minimum": 5, "maximum": 120, "type": "integer", "description": "Seconds to wait for a single page before giving up on it and moving on.", "default": 25 }, "maxRunSecs": { "title": "Max run time", "minimum": 0, "maximum": 36000, "type": "integer", "description": "Stop crawling after this many seconds and write out whatever has been extracted so far. 0 means no limit beyond the platform run timeout. A second guardrail against a crawl that never ends.", "default": 0 }, "userAgent": { "title": "User agent", "type": "string", "description": "Override the browser User-Agent header sent with every request.", "default": "" }, "proxyConfiguration": { "title": "Proxy configuration", "type": "object", "description": "Optional. Route requests through Apify Proxy. Most public documentation sites do not need it.", "default": { "useApifyProxy": false } } } }, "runsResponseSchema": { "type": "object", "properties": { "data": { "type": "object", "properties": { "id": { "type": "string" }, "actId": { "type": "string" }, "userId": { "type": "string" }, "startedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "finishedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "status": { "type": "string", "example": "READY" }, "meta": { "type": "object", "properties": { "origin": { "type": "string", "example": "API" }, "userAgent": { "type": "string" } } }, "stats": { "type": "object", "properties": { "inputBodyLen": { "type": "integer", "example": 2000 }, "rebootCount": { "type": "integer", "example": 0 }, "restartCount": { "type": "integer", "example": 0 }, "resurrectCount": { "type": "integer", "example": 0 }, "computeUnits": { "type": "integer", "example": 0 } } }, "options": { "type": "object", "properties": { "build": { "type": "string", "example": "latest" }, "timeoutSecs": { "type": "integer", "example": 300 }, "memoryMbytes": { "type": "integer", "example": 1024 }, "diskMbytes": { "type": "integer", "example": 2048 } } }, "buildId": { "type": "string" }, "defaultKeyValueStoreId": { "type": "string" }, "defaultDatasetId": { "type": "string" }, "defaultRequestQueueId": { "type": "string" }, "buildNumber": { "type": "string", "example": "1.0.0" }, "containerUrl": { "type": "string" }, "usage": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "integer", "example": 1 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 } } }, "usageTotalUsd": { "type": "number", "example": 0.00005 }, "usageUsd": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "number", "example": 0.00005 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 } } } } } } } } }}OpenAPI is a standard for designing and describing RESTful APIs, allowing developers to define API structure, endpoints, and data formats in a machine-readable way. It simplifies API development, integration, and documentation.
OpenAPI is effective when used with AI agents and GPTs by standardizing how these systems interact with various APIs, for reliable integrations and efficient communication.
By defining machine-readable API specifications, OpenAPI allows AI models like GPTs to understand and use varied data sources, improving accuracy. This accelerates development, reduces errors, and provides context-aware responses, making OpenAPI a core component for AI applications.
You can download the OpenAPI definitions for Docs To Markdown from the options below:
If you’d like to learn more about how OpenAPI powers GPTs, read our blog post.
You can also check out our other API clients: