Dataset Deduplicator & Cleaner
Pricing
Pay per usage
Dataset Deduplicator & Cleaner
Remove duplicate rows from an Apify Dataset or inline JSON. Deduplicate by selected fields, trim strings, and write the cleaned rows to a new Dataset.
Dataset Deduplicator & Cleaner
Pricing
Pay per usage
Remove duplicate rows from an Apify Dataset or inline JSON. Deduplicate by selected fields, trim strings, and write the cleaned rows to a new Dataset.
You can access the Dataset Deduplicator & Cleaner programmatically from your own applications by using the Apify API. You can also choose the language preference from below. To use the Apify API, you’ll need an Apify account and your API token, found in API & Integrations in Apify Console.
{ "openapi": "3.0.1", "info": { "version": "0.1", "x-build-id": "ez1JVk1itozls72LC" }, "servers": [ { "url": "https://api.apify.com/v2" } ], "paths": { "/acts/starshaped_bullsnake~dataset-deduplicator-cleaner/run-sync-get-dataset-items": { "post": { "operationId": "run-sync-get-dataset-items-starshaped_bullsnake-dataset-deduplicator-cleaner", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } }, "/acts/starshaped_bullsnake~dataset-deduplicator-cleaner/runs": { "post": { "operationId": "runs-sync-starshaped_bullsnake-dataset-deduplicator-cleaner", "x-openai-isConsequential": false, "summary": "Executes an Actor and returns information about the initiated run in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK", "content": { "application/json": { "schema": { "$ref": "#/components/schemas/runsResponseSchema" } } } } } } }, "/acts/starshaped_bullsnake~dataset-deduplicator-cleaner/run-sync": { "post": { "operationId": "run-sync-starshaped_bullsnake-dataset-deduplicator-cleaner", "x-openai-isConsequential": false, "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.", "tags": [ "Run Actor" ], "requestBody": { "required": true, "content": { "application/json": { "schema": { "$ref": "#/components/schemas/inputSchema" } } } }, "parameters": [ { "name": "token", "in": "query", "required": true, "schema": { "type": "string" }, "description": "Enter your Apify token here" } ], "responses": { "200": { "description": "OK" } } } } }, "components": { "schemas": { "inputSchema": { "type": "object", "properties": { "sourceDatasetId": { "title": "Source Dataset", "type": "string", "description": "Optional Apify Dataset to clean. If empty, the inline sample rows below are used." }, "rows": { "title": "Inline rows", "type": "array", "description": "JSON rows to clean when no Source Dataset is selected. The defaults intentionally contain one duplicate for a quick smoke test.", "items": { "type": "object" }, "default": [ { "id": "A001", "name": " Alpha ", "score": 10 }, { "id": "A002", "name": "Beta", "score": 20 }, { "id": "A001", "name": "Alpha duplicate", "score": 99 } ] }, "keyFields": { "title": "Deduplication key fields", "type": "array", "description": "Rows with the same values in these fields are treated as duplicates. Leave empty to compare the whole row.", "items": { "type": "string" }, "default": [ "id" ] }, "keep": { "title": "Which duplicate to keep", "enum": [ "first", "last" ], "type": "string", "description": "Choose whether the first or last row wins when duplicate keys are found.", "default": "first" }, "trimStrings": { "title": "Trim text values", "type": "boolean", "description": "Remove leading and trailing whitespace from top-level string fields before deduplication and output.", "default": true }, "caseInsensitiveKeys": { "title": "Case-insensitive text keys", "type": "boolean", "description": "When enabled, text values used in deduplication keys are compared without letter-case differences.", "default": false }, "maxItems": { "title": "Maximum source rows", "minimum": 1, "maximum": 250000, "type": "integer", "description": "Safety limit for rows read and processed in one run.", "default": 10000 } } }, "runsResponseSchema": { "type": "object", "properties": { "data": { "type": "object", "properties": { "id": { "type": "string" }, "actId": { "type": "string" }, "userId": { "type": "string" }, "startedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "finishedAt": { "type": "string", "format": "date-time", "example": "2025-01-08T00:00:00.000Z" }, "status": { "type": "string", "example": "READY" }, "meta": { "type": "object", "properties": { "origin": { "type": "string", "example": "API" }, "userAgent": { "type": "string" } } }, "stats": { "type": "object", "properties": { "inputBodyLen": { "type": "integer", "example": 2000 }, "rebootCount": { "type": "integer", "example": 0 }, "restartCount": { "type": "integer", "example": 0 }, "resurrectCount": { "type": "integer", "example": 0 }, "computeUnits": { "type": "integer", "example": 0 } } }, "options": { "type": "object", "properties": { "build": { "type": "string", "example": "latest" }, "timeoutSecs": { "type": "integer", "example": 300 }, "memoryMbytes": { "type": "integer", "example": 1024 }, "diskMbytes": { "type": "integer", "example": 2048 } } }, "buildId": { "type": "string" }, "defaultKeyValueStoreId": { "type": "string" }, "defaultDatasetId": { "type": "string" }, "defaultRequestQueueId": { "type": "string" }, "buildNumber": { "type": "string", "example": "1.0.0" }, "containerUrl": { "type": "string" }, "usage": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "integer", "example": 1 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 }, "PROXY_UNBLOCKER_UNITS": { "type": "integer", "example": 0 } } }, "usageTotalUsd": { "type": "number", "example": 0.00005 }, "usageUsd": { "type": "object", "properties": { "ACTOR_COMPUTE_UNITS": { "type": "integer", "example": 0 }, "DATASET_READS": { "type": "integer", "example": 0 }, "DATASET_WRITES": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_READS": { "type": "integer", "example": 0 }, "KEY_VALUE_STORE_WRITES": { "type": "number", "example": 0.00005 }, "KEY_VALUE_STORE_LISTS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_READS": { "type": "integer", "example": 0 }, "REQUEST_QUEUE_WRITES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_INTERNAL_GBYTES": { "type": "integer", "example": 0 }, "DATA_TRANSFER_EXTERNAL_GBYTES": { "type": "integer", "example": 0 }, "PROXY_RESIDENTIAL_TRANSFER_GBYTES": { "type": "integer", "example": 0 }, "PROXY_SERPS": { "type": "integer", "example": 0 }, "PROXY_UNBLOCKER_UNITS": { "type": "integer", "example": 0 } } } } } } } } }}OpenAPI is a standard for designing and describing RESTful APIs, allowing developers to define API structure, endpoints, and data formats in a machine-readable way. It simplifies API development, integration, and documentation.
OpenAPI is effective when used with AI agents and GPTs by standardizing how these systems interact with various APIs, for reliable integrations and efficient communication.
By defining machine-readable API specifications, OpenAPI allows AI models like GPTs to understand and use varied data sources, improving accuracy. This accelerates development, reduces errors, and provides context-aware responses, making OpenAPI a core component for AI applications.
You can download the OpenAPI definitions for Dataset Deduplicator & Cleaner from the options below:
If you’d like to learn more about how OpenAPI powers GPTs, read our blog post.
You can also check out our other API clients: