{
  "openapi": "3.0.1",
  "info": {
    "title": "Hugging Face Scraper",
    "description": "Scrape Hugging Face models, datasets, Spaces & papers: downloads, likes, parameters, license, tasks, inference providers and model cards. Search or paste any URL.",
    "version": "1.0",
    "x-build-id": "M1bsg2Fdjr7pXNGZJ"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/dtrungtin~huggingface-scraper/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-dtrungtin-huggingface-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/dtrungtin~huggingface-scraper/runs": {
      "post": {
        "operationId": "runs-sync-dtrungtin-huggingface-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/dtrungtin~huggingface-scraper/run-sync": {
      "post": {
        "operationId": "run-sync-dtrungtin-huggingface-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "properties": {
          "resourceType": {
            "title": "What to scrape",
            "enum": [
              "models",
              "datasets",
              "spaces",
              "papers"
            ],
            "type": "string",
            "description": "Which part of the Hugging Face Hub to search. Ignored when you provide URLs below.",
            "default": "models"
          },
          "searchTerms": {
            "title": "Search terms",
            "type": "array",
            "description": "Full-text search, e.g. <code>llama</code>, <code>whisper</code> or <code>sentiment</code>. Each term is scraped separately. Leave empty to browse everything that matches the filters. For papers, searches titles and abstracts; leave empty for the Daily Papers feed.",
            "items": {
              "type": "string"
            }
          },
          "sort": {
            "title": "Sort by",
            "enum": [
              "trending",
              "downloads",
              "likes",
              "created",
              "modified"
            ],
            "type": "string",
            "description": "Order of results. Spaces cannot be sorted by downloads. Papers support Trending; any other value lists the newest papers.",
            "default": "trending"
          },
          "maxItems": {
            "title": "Max results per search or URL",
            "minimum": 0,
            "type": "integer",
            "description": "Maximum number of results saved for each search term or URL. Set to 0 for no limit (a full crawl of all models is over 2 million results).",
            "default": 100
          },
          "startUrls": {
            "title": "Hugging Face URLs",
            "type": "array",
            "description": "Optional. Paste any huggingface.co URL and its data is scraped instead of the search above. Supported: listing pages with filters (e.g. <code>https://huggingface.co/models?pipeline_tag=text-generation&sort=trending</code>), model / dataset / Space pages, user or organization profiles, collections, and papers pages (daily, trending, a date, or a single paper).",
            "items": {
              "type": "object",
              "required": [
                "url"
              ],
              "properties": {
                "url": {
                  "type": "string",
                  "title": "URL of a web page",
                  "format": "uri"
                }
              }
            }
          },
          "author": {
            "title": "Author or organization",
            "type": "string",
            "description": "Only results published by this user or organization, e.g. <code>meta-llama</code>, <code>google</code>, <code>Qwen</code>."
          },
          "task": {
            "title": "Task",
            "enum": [
              "any-to-any",
              "audio-classification",
              "audio-text-to-text",
              "audio-to-audio",
              "automatic-speech-recognition",
              "depth-estimation",
              "document-question-answering",
              "feature-extraction",
              "fill-mask",
              "image-classification",
              "image-feature-extraction",
              "image-segmentation",
              "image-text-to-image",
              "image-text-to-text",
              "image-text-to-video",
              "image-to-3d",
              "image-to-image",
              "image-to-text",
              "image-to-video",
              "keypoint-detection",
              "mask-generation",
              "object-detection",
              "question-answering",
              "reinforcement-learning",
              "sentence-similarity",
              "summarization",
              "table-question-answering",
              "tabular-classification",
              "tabular-regression",
              "text-classification",
              "text-generation",
              "text-ranking",
              "text-to-3d",
              "text-to-image",
              "text-to-speech",
              "text-to-video",
              "token-classification",
              "translation",
              "unconditional-image-generation",
              "video-classification",
              "video-text-to-text",
              "video-to-video",
              "visual-document-retrieval",
              "visual-question-answering",
              "zero-shot-classification",
              "zero-shot-image-classification",
              "zero-shot-object-detection"
            ],
            "type": "string",
            "description": "Models: pipeline task. Datasets: task category. Not available for Spaces."
          },
          "library": {
            "title": "Library",
            "type": "string",
            "description": "Models: e.g. <code>transformers</code>, <code>gguf</code>, <code>diffusers</code>, <code>mlx</code>, <code>onnx</code>. Datasets: e.g. <code>datasets</code>, <code>pandas</code>, <code>polars</code>."
          },
          "language": {
            "title": "Language",
            "type": "string",
            "description": "ISO language code, e.g. <code>en</code>, <code>vi</code>, <code>fr</code>, <code>zh</code>."
          },
          "license": {
            "title": "License",
            "type": "string",
            "description": "License id, e.g. <code>apache-2.0</code>, <code>mit</code>, <code>cc-by-4.0</code>, <code>llama3.1</code>."
          },
          "tags": {
            "title": "Tags",
            "type": "array",
            "description": "Any Hugging Face tags that results must have, e.g. <code>conversational</code>, <code>base_model:finetune:Qwen/Qwen3-8B</code>, <code>size_categories:1K&lt;n&lt;10K</code>, <code>mcp-server</code>.",
            "items": {
              "type": "string"
            }
          },
          "minParameters": {
            "title": "Min parameters (models)",
            "pattern": "^(\\d+(\\.\\d+)?\\s*[KMBTkmbt]?)?$",
            "type": "string",
            "description": "Minimum model size, e.g. <code>500M</code>, <code>7B</code>."
          },
          "maxParameters": {
            "title": "Max parameters (models)",
            "pattern": "^(\\d+(\\.\\d+)?\\s*[KMBTkmbt]?)?$",
            "type": "string",
            "description": "Maximum model size, e.g. <code>3B</code>, <code>70B</code>."
          },
          "spaceSdk": {
            "title": "Space SDK",
            "enum": [
              "gradio",
              "streamlit",
              "docker",
              "static"
            ],
            "type": "string",
            "description": "Only Spaces built with this SDK."
          },
          "papersDate": {
            "title": "Papers date",
            "pattern": "^(\\d{4}-\\d{2}-\\d{2}|\\d{4}-W\\d{2}|\\d{4}-\\d{2})?$",
            "type": "string",
            "description": "Daily Papers for a day (<code>2026-09-22</code>), ISO week (<code>2026-W38</code>) or month (<code>2026-09</code>). Leave empty for the latest papers."
          },
          "minDownloads": {
            "title": "Min downloads (last 30 days)",
            "minimum": 0,
            "type": "integer",
            "description": "Skip results with fewer downloads. Fastest when sorting by Most downloads."
          },
          "minLikes": {
            "title": "Min likes / upvotes",
            "minimum": 0,
            "type": "integer",
            "description": "Skip results with fewer likes (upvotes for papers). Fastest when sorting by Most likes."
          },
          "createdAfter": {
            "title": "Created after",
            "type": "string",
            "description": "Only results created (papers: published) after this date. Also accepts a relative period such as <code>7 days</code>, which is useful for scheduled monitoring runs. Fastest when sorting by Recently created."
          },
          "includeReadme": {
            "title": "Include README / model card text",
            "type": "boolean",
            "description": "Adds the full README (model card, dataset card or Space description) as Markdown. Great for LLM/RAG pipelines. Makes runs slower (one extra request per result).",
            "default": false
          },
          "includeFiles": {
            "title": "Include file list",
            "type": "boolean",
            "description": "Adds the list of files in each repository (weights, configs, notebooks...). For datasets this needs one extra request per result.",
            "default": false
          },
          "includeCardData": {
            "title": "Include raw card metadata",
            "type": "boolean",
            "description": "Adds the raw YAML metadata of the model/dataset card (base model, datasets, metrics, widget examples...). Can be large.",
            "default": false
          },
          "hfToken": {
            "title": "Hugging Face access token",
            "type": "string",
            "description": "Optional. A free <a href='https://huggingface.co/settings/tokens' target='_blank'>read token</a> gives you a higher rate limit and access to your own private or gated repos. Stored encrypted."
          },
          "proxyConfiguration": {
            "title": "Proxy configuration",
            "type": "object",
            "description": "Not needed in most cases: the Hugging Face API is public. Enable only if you hit rate limits on very large runs without a token.",
            "default": {
              "useApifyProxy": false
            }
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}