{
  "data": {
    "id": "kwHS2PKHK4gyKjVZ8",
    "userId": "HIMU0xktu1JeZb4Xl",
    "name": "hugging-face-datasets-scraper",
    "username": "fetch_cat",
    "description": "Scrape public Hugging Face dataset metadata, tags, access flags, downloads, likes, and recency signals.",
    "isPublic": true,
    "createdAt": "2026-07-17T08:38:27.535Z",
    "modifiedAt": "2026-07-29T20:58:03.942Z",
    "taggedBuilds": {
      "latest": {
        "buildId": "mxbG50OCaCghzWGus",
        "finishedAt": "2026-07-29T20:58:03.942Z",
        "buildNumberInt": 100003,
        "buildNumber": "0.1.3"
      }
    },
    "stats": {
      "totalBuilds": 3,
      "totalRuns": 50,
      "totalUsers": 2,
      "totalUsers7Days": 1,
      "totalUsers30Days": 1,
      "totalUsers90Days": 1,
      "lastRunStartedAt": "2026-08-26T16:28:49.604Z",
      "actorReviewCount": 0,
      "actorReviewRating": 0,
      "bookmarkCount": 0,
      "publicActorRunStats30Days": {
        "ABORTED": 0,
        "FAILED": 0,
        "SUCCEEDED": 30,
        "TIMED-OUT": 0,
        "TOTAL": 30
      }
    },
    "versions": [
      {
        "versionNumber": "0.1",
        "sourceType": "SOURCE_FILES",
        "buildTag": "latest"
      }
    ],
    "defaultRunOptions": {
      "build": "latest",
      "timeoutSecs": 300,
      "memoryMbytes": 512
    },
    "exampleRunInput": {
      "body": "{\n  \"searchQueries\": [\"finance\"],\n  \"maxItems\": 10,\n  \"sort\": \"downloads\",\n  \"dedupe\": true\n}",
      "contentType": "application/json; charset=utf-8"
    },
    "categories": [
      "AI",
      "DEVELOPER_TOOLS",
      "AUTOMATION"
    ],
    "isDeprecated": false,
    "title": "Hugging Face Datasets Scraper",
    "pictureUrl": "https://apify-image-uploads-prod.s3.us-east-1.amazonaws.com/HIMU0xktu1JeZb4Xl-actor-kwHS2PKHK4gyKjVZ8-jlxFZJjBUm-actor-icon.png",
    "seoTitle": "Hugging Face Datasets Scraper - CSV, JSON, API",
    "seoDescription": "Export public Hugging Face dataset metadata, tags, licenses, downloads, likes, access flags, and recency signals to CSV, JSON, Excel, or API fast.",
    "pricingInfos": [
      {
        "pricingModel": "PAY_PER_EVENT",
        "startedAt": "2026-07-17T08:42:13.681Z",
        "pricingPerEvent": {
          "actorChargeEvents": {
            "start": {
              "eventTitle": "Start",
              "eventDescription": "One-time fee per run",
              "eventPriceUsd": 0.005,
              "isOneTimeEvent": true
            },
            "item": {
              "eventTitle": "Dataset metadata record",
              "eventDescription": "Per Hugging Face dataset metadata record produced",
              "eventTieredPricingUsd": {
                "FREE": {
                  "tieredEventPriceUsd": 0.000575
                },
                "BRONZE": {
                  "tieredEventPriceUsd": 0.0005
                },
                "SILVER": {
                  "tieredEventPriceUsd": 0.00039
                },
                "GOLD": {
                  "tieredEventPriceUsd": 0.0003
                },
                "PLATINUM": {
                  "tieredEventPriceUsd": 0.0002
                },
                "DIAMOND": {
                  "tieredEventPriceUsd": 0.00014
                }
              },
              "isPrimaryEvent": true
            }
          }
        },
        "createdAt": "2026-07-17T08:42:14.115Z",
        "apifyMarginPercentage": 0.2
      }
    ],
    "notice": "NONE",
    "isCritical": false,
    "isGeneric": false,
    "hasNoDataset": false,
    "isSourceCodeHidden": true,
    "standbyUrl": null,
    "actorPermissionLevel": "LIMITED_PERMISSIONS",
    "readmeSummary": "## Hugging Face Datasets Scraper\n\nA metadata extraction Actor that searches the Hugging Face datasets catalog by keywords or exact dataset identifiers and exports structured, audit-friendly dataset metadata rows. It queries Hugging Face catalog endpoints, ranks and filters search results, preserves raw tags, and parses those tags into standardized fields such as license, tasks, languages, size/format, libraries, and modalities. The Actor captures identifiers and URLs, owner/author and repository names, human-readable descriptions, popularity and trending metrics, timestamps and revision metadata, access flags, optional README/card text, and run provenance and diagnostics — producing one structured record per dataset suitable for discovery, comparison, enrichment, and monitoring workflows.\n\n## Use cases\n\n- Discover datasets for model training, retrieval-augmented generation (RAG), or fine-tuning by keyword search and ranked sorting (downloads, likes, recency, trending).  \n- Compare and audit dataset licenses, task labels, languages, formats, and other tag-derived metadata for compliance review.  \n- Enrich a list of known dataset identifiers with current catalog metadata (owner, description, popularity, tags, timestamps).  \n- Monitor dataset updates and trends over time by scheduling repeated metadata exports and comparing recency and revision fields.  \n- Prepare spreadsheet-friendly exports for license review and dataset selection by exporting raw tags plus parsed convenience fields.  \n- Find domain-specific datasets (examples: finance, medical imaging, legal documents, instruction tuning, NLP, computer vision) and filter or rank results for targeted discovery.  \n- Curate trending or popular datasets in specific modalities (e.g., vision, NLP) for research analysts and ML platform teams.",
    "deploymentKey": "ssh-rsa AAAAB3NzaC1yc2EAAAADAQABAAABAQCe62XezMBq94Ns5tzkZdWTp19xHT8PK0X8YhCqZoDcpEOK/y5KdflStn+23r067V2Y8JTK9KpE5XeRSsWg/PhORIE5QrJSY7+QMU4UsQBPPbr+lijwKvuth/zx7Xj11oS6i6frkg29WTJYIKmagDewKgxqssvIqlLz3EAYGvlXn55z3MmXNjiUQ3+JBLvSg73sYtqQywNZtaAGjX7vXUSizguBR40foowslAsMhwPrxXYZwAW7cDzks3hzQzhXrIK4QSSLEK5mkFndcHCIOku/TYXOTqVnwmWy0esAdkNxeNRFs3G6Kv7J24WlSvXnQnl5J7jwn6AxePiSOs5aBTTN \n"
  }
}