{
  "openapi": "3.0.1",
  "info": {
    "title": "Substack Scraper – Posts, Comments & Substack Newsletter API",
    "description": "Scrape any Substack newsletter via its own newsletter API into JSON/CSV: posts with full text, comments, or the category leaderboard itself (rank, subscriber counts, pricing), no post scraping needed. No login, no browser, no start fee. Full text from $0.002, leaderboard rows from $0.0005.",
    "version": "0.1",
    "x-build-id": "UaBG9WmrlEEPQGIVB"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/fetchsmith~substack-scraper/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-fetchsmith-substack-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/fetchsmith~substack-scraper/runs": {
      "post": {
        "operationId": "runs-sync-fetchsmith-substack-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/fetchsmith~substack-scraper/run-sync": {
      "post": {
        "operationId": "run-sync-fetchsmith-substack-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "properties": {
          "publicationUrls": {
            "title": "Publication URLs or handles",
            "type": "array",
            "description": "Substack publications to scrape. Accepts any form: \"astralcodexten\", \"astralcodexten.substack.com\", or a custom domain like \"https://www.bigtechnology.com\". Custom domains and redirects are handled automatically.",
            "default": [
              "https://astralcodexten.substack.com"
            ],
            "items": {
              "type": "string"
            }
          },
          "postUrls": {
            "title": "Individual post URLs",
            "type": "array",
            "description": "Scrape specific posts instead of (or in addition to) whole publications, e.g. https://www.bigtechnology.com/p/some-post.",
            "items": {
              "type": "string"
            }
          },
          "discoverCategories": {
            "title": "Discover by category (no URLs needed)",
            "type": "array",
            "description": "Start from a topic instead of a URL list: Substack category slugs (e.g. technology, business, finance, culture, us-politics, food). Publications come back in Substack leaderboard order and are scraped like any other publication. Subcategory slugs work too. Combine with maxPublicationsPerCategory.",
            "default": [],
            "items": {
              "type": "string"
            }
          },
          "maxPublicationsPerCategory": {
            "title": "Max publications per category",
            "minimum": 1,
            "maximum": 100,
            "type": "integer",
            "description": "How many top publications to take from each category in discoverCategories.",
            "default": 10
          },
          "discoverType": {
            "title": "Discovery: newsletters or podcasts",
            "enum": [
              "all",
              "newsletter",
              "podcast"
            ],
            "type": "string",
            "description": "Filter discovered publications by type. Only applies to discoverCategories. Note: Substack's category leaderboards are almost entirely `newsletter` publications — a live sweep of the whole technology leaderboard (300 publications) found zero podcast-type ones — so \"Podcasts only\" usually returns nothing outside the `podcast` category. If you want podcast *episodes*, use contentType=podcast instead: newsletter-type publications do publish podcast posts.",
            "default": "all"
          },
          "leaderboardTier": {
            "title": "Leaderboard ranking",
            "enum": [
              "all",
              "free",
              "paid"
            ],
            "type": "string",
            "description": "Which Substack leaderboard ranking to pull publications from for discoverCategories: overall, or paid-only (Substack ranks paid as its own list, e.g. substack.com/leaderboard/technology/paid). \"free\" is accepted for backwards compatibility but is NOT a real Substack leaderboard tier — verified live across 3 categories and every page depth, it always returns the exact same list as \"all\" (which mixes free and paid-tier publications), never a free-only ranking. Use \"paid\" to find monetized newsletters; there is no upstream free-only equivalent.",
            "default": "all"
          },
          "leaderboardOnly": {
            "title": "Leaderboard only (no post scraping)",
            "type": "boolean",
            "description": "Return one row per publication from the discoverCategories leaderboard itself — rank, subscriber counts, author, subscription prices — instead of scraping any posts. Cheaper and faster than a full run when you only want market-research / newsletter-discovery data, not article content. Requires discoverCategories; publicationUrls/postUrls are ignored in this mode.",
            "default": false
          },
          "searchQuery": {
            "title": "Search within the publication",
            "type": "string",
            "description": "Only return posts matching this keyword inside each publication's archive. Leave empty to get the newest posts."
          },
          "includeBodyText": {
            "title": "Include full article text",
            "type": "boolean",
            "description": "Fetch each post's full body and return it as clean plain text. Costs one extra request per post; free posts return the whole article, paywalled posts return the public preview.",
            "default": true
          },
          "includeBodyHtml": {
            "title": "Include raw HTML body",
            "type": "boolean",
            "description": "Also return the original body_html (large; useful if you want to keep formatting, images and links).",
            "default": false
          },
          "includeComments": {
            "title": "Include comments",
            "type": "boolean",
            "description": "Also return each post's comments (whole thread, flattened, with parent IDs). Comments are charged as results like posts.",
            "default": false
          },
          "maxCommentsPerPost": {
            "title": "Max comments per post",
            "minimum": 1,
            "maximum": 1000,
            "type": "integer",
            "description": "Cap on comments returned per post.",
            "default": 50
          },
          "includePublicationInfo": {
            "title": "Include publication info",
            "type": "boolean",
            "description": "Add publication-level fields to every post row: free subscriber count, paid-subscriber band, bestseller tier, author name/handle/bio, subscription plan prices, podcast flag, language and first-post date. Costs one extra request per publication (cached), not per post.",
            "default": false
          },
          "contentType": {
            "title": "Content type",
            "enum": [
              "all",
              "newsletter",
              "podcast",
              "thread"
            ],
            "type": "string",
            "description": "Filter posts by type: regular newsletter posts, podcast episodes, or threads (Substack Notes-style discussion posts).",
            "default": "all"
          },
          "audienceFilter": {
            "title": "Free / paid posts",
            "enum": [
              "all",
              "free",
              "paid"
            ],
            "type": "string",
            "description": "Return all posts, only free (public) posts, or only paywalled subscriber posts.",
            "default": "all"
          },
          "publishedAfter": {
            "title": "Published after (ISO date)",
            "type": "string",
            "description": "e.g. 2026-01-01. Posts older than this are skipped."
          },
          "publishedBefore": {
            "title": "Published before (ISO date)",
            "type": "string",
            "description": "e.g. 2026-09-01. Posts newer than this are skipped."
          },
          "minReactionCount": {
            "title": "Min reactions (likes)",
            "minimum": 0,
            "type": "integer",
            "description": "Only return posts with at least this many reactions. Read directly from the archive listing, so it costs nothing extra and applies before any post is charged."
          },
          "minCommentCount": {
            "title": "Min comments",
            "minimum": 0,
            "type": "integer",
            "description": "Only return posts with at least this many comments. Useful for finding a publication's most-discussed posts."
          },
          "minRestackCount": {
            "title": "Min restacks",
            "minimum": 0,
            "type": "integer",
            "description": "Only return posts with at least this many restacks (Substack's share/reshare action)."
          },
          "minWordCount": {
            "title": "Min word count",
            "minimum": 0,
            "type": "integer",
            "description": "Only return posts with at least this many words."
          },
          "maxWordCount": {
            "title": "Max word count",
            "minimum": 0,
            "type": "integer",
            "description": "Only return posts with at most this many words. Useful for finding short posts/links roundups vs. long essays."
          },
          "maxPostsPerPublication": {
            "title": "Max posts per publication",
            "minimum": 1,
            "maximum": 5000,
            "type": "integer",
            "description": "How deep to go into each publication's archive. This counts posts SCANNED, before audienceFilter/publishedAfter/publishedBefore are applied — so when you use those filters (or searchQuery, which returns matches in relevance order, not date order), raise this well above the number of results you want, or matching posts deeper in the archive are never looked at.",
            "default": 50
          },
          "maxResults": {
            "title": "Max results (total)",
            "minimum": 1,
            "maximum": 10000,
            "type": "integer",
            "description": "Overall cap across all publications, posts and comments. This is what you pay for.",
            "default": 200
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}