RAG Dataset Quality Auditor
Pricing
from $0.50 / 1,000 document auditeds
RAG Dataset Quality Auditor
Find duplicate, stale, incomplete, repetitive, and poorly chunked documents before they weaken RAG retrieval.
RAG Dataset Quality Auditor
Pricing
from $0.50 / 1,000 document auditeds
Find duplicate, stale, incomplete, repetitive, and poorly chunked documents before they weaken RAG retrieval.
You can access the RAG Dataset Quality Auditor programmatically from your own applications by using the Apify API. You can also choose the language preference from below. To use the Apify API, you’ll need an Apify account and your API token, found in API & Integrations in Apify Console.
{ "openapi": "3.0.1", "info": { "version": "0.2", "x-build-id": "gwd95YbEO6VY3Rlkh" }, "servers": [ { "url": "https://api.apify.com/v2" } ], "paths": { "/acts/gifted_wagon~rag-dataset-quality-auditor/run-sync-get-dataset-items": { "post": { "operationId": "run-sync-get-dataset-items-gifted_wagon-rag-dataset-quality-auditor", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } }, "/acts/gifted_wagon~rag-dataset-quality-auditor/runs": { "post": { "operationId": "runs-sync-gifted_wagon-rag-dataset-quality-auditor", "x-openai-isConsequential": false, "summary": "Executes an Actor and returns information about the initiated run in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/runsResponseSchema" } } } } } } }, "/acts/gifted_wagon~rag-dataset-quality-auditor/run-sync": { "post": { "operationId": "run-sync-gifted_wagon-rag-dataset-quality-auditor", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } } }, "components": { "schemas": { "inputSchema": { "type": "object", "properties": { "datasetId": { "title": "Dataset to audit", "type": "string", "description": "Select an existing Apify dataset. Each item should contain a text-like field such as text, markdown, content, or body." }, "documents": { "title": "Inline documents", "type": "array", "description": "Optional documents pasted directly. Each item may contain id, url, title, text, updatedAt, and metadata.", "items": { "type": "object", "additionalProperties": true, "properties": { "id": { "title": "Document ID", "type": "string", "description": "Stable source identifier.", "editor": "textfield" }, "url": { "title": "Source URL", "type": "string", "description": "Canonical source URL.", "editor": "textfield" }, "title": { "title": "Document title", "type": "string", "description": "Descriptive title used by retrieval consumers.", "editor": "textfield" }, "text": { "title": "Document text", "type": "string", "description": "Plain-text content to audit.", "editor": "textarea" }, "updatedAt": { "title": "Updated at", "type": "string", "description": "ISO 8601 freshness timestamp.", "editor": "textfield" }, "metadata": { "title": "Metadata", "type": "object", "description": "Optional source metadata retained in the input only.", "editor": "json" } } } }, "maxDocuments": { "title": "Maximum documents", "minimum": 1, "maximum": 5000, "type": "integer", "description": "Maximum number of documents to read and charge for across both sources.", "default": 500 }, "minWords": { "title": "Minimum useful words", "minimum": 1, "maximum": 10000, "type": "integer", "description": "Documents below this threshold receive a too-short issue.", "default": 80 }, "maxWords": { "title": "Maximum useful words", "minimum": 50, "maximum": 100000, "type": "integer", "description": "Documents above this threshold receive a too-long issue and should usually be chunked.", "default": 2000 }, "staleAfterDays": { "title": "Stale after days", "minimum": 0, "maximum": 3650, "type": "integer", "description": "Documents with a valid updatedAt older than this threshold receive a stale-content issue. Set 0 to disable.", "default": 365 }, "nearDuplicateThreshold": { "title": "Near-duplicate similarity", "minimum": 0.7, "maximum": 1, "type": "number", "description": "Jaccard similarity threshold for candidate documents identified by the fingerprint index.", "default": 0.92 }, "contentFields": { "title": "Dataset content fields", "type": "array", "description": "First matching field is used as document text when reading a dataset.", "items": { "type": "string" }, "default": [ "text", "markdown", "content", "body" ] }, "titleFields": { "title": "Dataset title fields", "type": "array", "description": "First matching field is used as the document title when reading a dataset.", "items": { "type": "string" }, "default": [ "title", "name", "heading" ] }, "urlFields": { "title": "Dataset URL fields", "type": "array", "description": "First matching field is used as the canonical source URL when reading a dataset.", "items": { "type": "string" }, "default": [ "url", "sourceUrl", "canonicalUrl" ] }, "updatedAtFields": { "title": "Dataset freshness fields", "type": "array", "description": "First matching field is used as the freshness timestamp when reading a dataset.", "items": { "type": "string" }, "default": [ "updatedAt", "lastModified", "modifiedAt", "publishedAt" ] }, "includeContentPreview": { "title": "Include content preview", "type": "boolean", "description": "Include a short normalized preview in each result. Leave off for sensitive corpora.", "default": false }, "contentPreviewCharacters": { "title": "Preview characters", "minimum": 50, "maximum": 2000, "type": "integer", "description": "Maximum number of normalized document characters to include when previews are enabled.", "default": 280 }, "payload": { "title": "Integration payload", "type": "object", "description": "Automatically supplied by Apify when this Actor is connected to another Actor run.", "additionalProperties": true } } }, "runsResponseSchema": { "type": "object", "properties": { "data": { "type": "object", "properties": { "id": { "type": "string" }, "actId": { "type": "string" }, "userId": { "type": "string" }, "startedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "finishedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "status": { "type": "string", "example": "READY" }, "meta": { "type": "object", "properties": { "origin": { "type": "string", "example": "API" }, "userAgent": { "type": "string" } } }, "stats": { "type": "object", "properties": { "inputBodyLen": { "type": "integer", "example": 2000 }, "rebootCount": { "type": "integer", "example": 0 }, "restartCount": { "type": "integer", "example": 0 }, "resurrectCount": { "type": "integer", "example": 0 }, "computeUnits": { "type": "integer", "example": 0 } } }, "options": { "type": "object", "properties": { "build": { "type": "string", "example": "latest" }, "timeoutSecs": { "type": "integer", "example": 300 }, "memoryMbytes": { "type": "integer", "example": 1024 }, "diskMbytes": { "type": "integer", "example": 2048 } } }, "buildId": { "type": "string" }, "defaultKeyValueStoreId": { "type": "string" }, "defaultDatasetId": { "type": "string" }, "defaultRequestQueueId": { "type": "string" }, "buildNumber": { "type": "string", "example": "1.0.0" }, "containerUrl": { "type": "string" }, "usage": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "integer", "example": 1 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 } } }, "usageTotalUsd": { "type": "number", "example": 0.00005 }, "usageUsd": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "number", "example": 0.00005 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 } } } } } } } } }}OpenAPI is a standard for designing and describing RESTful APIs, allowing developers to define API structure, endpoints, and data formats in a machine-readable way. It simplifies API development, integration, and documentation.
OpenAPI is effective when used with AI agents and GPTs by standardizing how these systems interact with various APIs, for reliable integrations and efficient communication.
By defining machine-readable API specifications, OpenAPI allows AI models like GPTs to understand and use varied data sources, improving accuracy. This accelerates development, reduces errors, and provides context-aware responses, making OpenAPI a core component for AI applications.
You can download the OpenAPI definitions for RAG Dataset Quality Auditor from the options below:
If you’d like to learn more about how OpenAPI powers GPTs, read our blog post.
You can also check out our other API clients: