{
  "openapi": "3.0.1",
  "info": {
    "title": "Data.gov Dataset Search API: US Government Open Data Catalog",
    "description": "Search the US federal open-data catalog by keyword, agency, government level, tag, theme or file format and get one row per dataset: title, publisher, description, licence, issue and harvest dates, popularity and every resource download URL. Dictionary modes list the agencies and the tags.",
    "version": "0.1",
    "x-build-id": "fkIEZmSn8bsOdIRj4"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/yadroo~data-gov-datasets/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-yadroo-data-gov-datasets",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/yadroo~data-gov-datasets/runs": {
      "post": {
        "operationId": "runs-sync-yadroo-data-gov-datasets",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/yadroo~data-gov-datasets/run-sync": {
      "post": {
        "operationId": "run-sync-yadroo-data-gov-datasets",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "properties": {
          "mode": {
            "title": "What to return",
            "enum": [
              "search",
              "dataset",
              "organizations",
              "keywords"
            ],
            "type": "string",
            "description": "`search` runs the catalog search and writes a row per dataset. `dataset` looks up the exact catalog slugs listed below, one row each, and marks a slug the catalog does not hold with `found: false` instead of failing the run. `organizations` writes the publishing organizations of the catalog (121 on 29.09.2026) with their slug and government level - that slug is what *Publishing organization* takes. `keywords` writes the tag list with the number of datasets behind each tag, so you can see what a filter will return before you pay for rows. The two dictionary modes ignore the filters and cost one request.",
            "default": "search"
          },
          "query": {
            "title": "Search words",
            "type": "string",
            "description": "Full-text search over dataset title, description and tags, e.g. `air quality`, `electric vehicle`, `medicare spending`, `broadband`. Empty = no text condition: with *Sort rows by* set to `popularity` that returns the most-used datasets of the whole catalog, which is the cheap way to explore before you narrow down. Used in `search` mode only."
          },
          "slugs": {
            "title": "Catalog slugs",
            "type": "array",
            "description": "Used in `dataset` mode only: the slugs to look up, e.g. `air-quality`, `motor-vehicle-collisions-crashes`. The slug is the last part of a catalog dataset page URL (`https://catalog.data.gov/dataset/air-quality` -> `air-quality`); a whole page URL is accepted and the slug is read out of it. One request per slug. In `search` mode this field is ignored.",
            "items": {
              "type": "string"
            }
          },
          "organizationSlug": {
            "title": "Publishing organization",
            "type": "string",
            "description": "Keep datasets of one organization, by its catalog slug: `census`, `noaa`, `nasa`, `epa`, `hhs`, `usda`, `energy`, `dot`, `dol`, `doi`, `ed`, `treasury`, `doj`, `hud` for federal departments, `california`, `washington` for states, `nyc-ny` for a city. Run the actor once in `organizations` mode for the complete list - a slug that is not in it matches nothing and yields an empty run, not an error."
          },
          "organizationTypes": {
            "title": "Level of government",
            "type": "array",
            "description": "Keep datasets published by organizations of these levels; several values = OR. The catalog is far from federal-only - city and state portals harvest into it too, so this is the filter that separates \"what does Washington publish\" from \"what does my state publish\". Empty = every level.",
            "items": {
              "type": "string",
              "enum": [
                "Federal Government",
                "State Government",
                "City Government",
                "County Government",
                "University",
                "Tribal",
                "Non-Profit"
              ],
              "enumTitles": [
                "Federal Government - departments and agencies",
                "State Government",
                "City Government",
                "County Government",
                "University",
                "Tribal",
                "Non-Profit"
              ]
            }
          },
          "keywords": {
            "title": "Tags (all of them)",
            "type": "array",
            "description": "Keep datasets tagged with every one of these catalog tags, e.g. [\"finance\", \"budget\"] returns datasets that carry both. Tags are lowercase free text the publisher chose, so check the spelling in `keywords` mode first: the catalog's most common tags are geographic boilerplate (`county or equivalent entity`, `state fips code`) rather than topics.",
            "items": {
              "type": "string"
            }
          },
          "themes": {
            "title": "Themes",
            "type": "array",
            "description": "Keep datasets filed under these themes; several values = OR. Themes come from the publisher's own metadata and are matched without regard to case, so `Education` and `education` are the same value. Values seen live on 29.09.2026: `Environment`, `Education`, `Health`, `Agriculture`, `Energy`, `Management/Operations`, `geospatial`. A theme is a coarser bucket than a tag and many datasets carry none.",
            "items": {
              "type": "string"
            }
          },
          "publisher": {
            "title": "Source portal",
            "type": "string",
            "description": "Keep datasets harvested from one source portal, matched exactly, e.g. `data.cityofnewyork.us`. This is finer than *Publishing organization*: a single organization often feeds the catalog from several portals, and the field on every row tells you which portal a dataset really came from."
          },
          "spatialFilter": {
            "title": "Geospatial datasets",
            "enum": [
              "any",
              "geospatial",
              "non-geospatial"
            ],
            "type": "string",
            "description": "Large parts of the catalog are map layers and boundary files: a popularity search inside one mapping agency comes back as shapefiles almost entirely. Pick `non-geospatial` when you want tables and statistics, `geospatial` when you want the map layers, and expect a narrow combination of both this and an organization filter to return nothing.",
            "default": "any"
          },
          "onlyWithDownloads": {
            "title": "Only datasets with downloadable files",
            "type": "boolean",
            "description": "Keep only datasets the catalog marks as having at least one downloadable file. Part of the catalog describes datasets that are behind a request form, an interactive map or an API you have to sign up for; switch this on when the run has to end in files you can actually fetch.",
            "default": false
          },
          "formats": {
            "title": "File formats",
            "type": "array",
            "description": "Keep datasets that offer at least one file in these formats, e.g. [\"csv\"], [\"json\", \"geojson\"], [\"xlsx\"]. The format of a file is read from its declared media type and from the file extension of its download link, so `text/csv` and a `.csv` link both count as `csv`. The catalog search itself has no format filter, so this one runs here, on the rows that came back: a run with a narrow format list can end with fewer rows than *Max rows* while still costing the requests it made. Empty = no format condition.",
            "items": {
              "type": "string"
            }
          },
          "sinceHours": {
            "title": "Harvested in the last N hours",
            "minimum": 1,
            "maximum": 8760,
            "type": "integer",
            "description": "Keep datasets whose catalog record was last refreshed from the publisher inside this window, in UTC. Pair it with *Sort rows by* = `last harvested` to watch what the catalog picked up overnight. Careful: this is the moment the catalog re-read the publisher's metadata, not the moment the data changed - a re-harvest with unchanged content updates it too."
          },
          "onlyNew": {
            "title": "Only datasets not seen before",
            "type": "boolean",
            "description": "Remember the catalog identifiers this input has already produced in the actor's key-value store and write only datasets that were not there on the previous run. The first run writes everything it finds, later runs write what appeared since. Built for a schedule: a daily run over a search you care about, paying only for the new entries.",
            "default": false
          },
          "sortBy": {
            "title": "Sort rows by",
            "enum": [
              "relevance",
              "popularity",
              "lastHarvested"
            ],
            "type": "string",
            "description": "`relevance` ranks by how well a dataset matches the search words and is the sensible default when you typed some. Without search words use `popularity`, which orders by the catalog's own page-visit counter and puts the datasets people actually use on top. `lastHarvested` orders by the moment the catalog last refreshed the record, newest first - the ordering the monitoring filter above is built for.",
            "default": "relevance"
          },
          "includeResources": {
            "title": "Include the file list",
            "type": "boolean",
            "description": "Add `resources`, the full list of files and endpoints of a dataset with title, format, media type, download link and, where the publisher gives one, a link to the column description. The list arrives in the same request, so it costs nothing extra - switch it off only to keep rows small when you already have `formats`, `resourceCount` and `primaryDownloadUrl`.",
            "default": true
          },
          "includeRawDcat": {
            "title": "Include the raw metadata record",
            "type": "boolean",
            "description": "Add `dcat`, the publisher's metadata record exactly as the catalog stores it (DCAT-US vocabulary: `@type`, `accessLevel`, `identifier`, `issued`, `modified`, `distribution`, `license`, `theme` and whatever else that publisher filled in). For pipelines that want the original vocabulary instead of our flat row. Contact names and mailbox addresses of the publishing office are removed from this record, like everywhere else in the output.",
            "default": false
          },
          "maxDescriptionChars": {
            "title": "Description length limit",
            "minimum": 0,
            "maximum": 20000,
            "type": "integer",
            "description": "Cut `description` after this many characters at a word boundary and set `descriptionTruncated` when something was cut. Catalog descriptions range from one line to several pages of HTML; the field arrives as plain text with markup removed. `0` leaves the description out of the row altogether.",
            "default": 1200
          },
          "maxItems": {
            "title": "Max rows",
            "minimum": 1,
            "maximum": 5000,
            "type": "integer",
            "description": "Stop after this many rows. The catalog held 570,118 datasets on 29.09.2026 and a broad search matches tens of thousands of them, so this cap - not the filters - is what decides what a run costs. Rows are fetched in pages of up to 1000, and the run stops as soon as the cap is reached.",
            "default": 25
          },
          "fields": {
            "title": "Output fields",
            "type": "array",
            "description": "Keep only these fields, in this order, e.g. [\"title\", \"organizationName\", \"formats\", \"primaryDownloadUrl\", \"url\"]. Empty = every field the mode produces.",
            "items": {
              "type": "string"
            }
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}