{
  "openapi": "3.0.1",
  "info": {
    "title": "Weibo Scraper — Hot Search, Posts, Comments & Creator Feeds",
    "description": "Sina Weibo (微博) scraper API, no login: keyword search, trending posts, hot-search board (微博热搜), comments, user profiles and post details, plus Black Cat (黑猫投诉) consumer complaints. Creator timelines (user_posts) need your cookie. Nine modes for social listening and China market research.",
    "version": "1.1",
    "x-build-id": "e2xr6WVaJUQbY4wBS"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/zhorex~weibo-scraper/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-zhorex-weibo-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/zhorex~weibo-scraper/runs": {
      "post": {
        "operationId": "runs-sync-zhorex-weibo-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/zhorex~weibo-scraper/run-sync": {
      "post": {
        "operationId": "run-sync-zhorex-weibo-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "required": [
          "mode"
        ],
        "properties": {
          "mode": {
            "title": "Mode",
            "enum": [
              "search",
              "hot_timeline",
              "hot_search",
              "post_comments",
              "hot_search_delta",
              "user_posts",
              "user_profile",
              "complaints",
              "post_details"
            ],
            "type": "string",
            "description": "What to scrape from Weibo",
            "default": "search"
          },
          "searchQuery": {
            "title": "Search query",
            "type": "string",
            "description": "Search query in Chinese or English. Chinese keywords yield better results. Examples: 人工智能, 新能源汽车, iPhone"
          },
          "autoLocalize": {
            "title": "Auto-localize brand names (search mode)",
            "type": "boolean",
            "description": "When on, a Latin brand name (e.g. 'Nike') also searches its Chinese name ('耐克'); the names take turns page by page and results are merged and deduped, so both share maxResults (no extra cost). Add your own variants via Search aliases below.",
            "default": true
          },
          "searchAliases": {
            "title": "Search aliases (search mode)",
            "type": "array",
            "description": "Extra term variants to also search and merge with your query — e.g. add '耐克' for Nike, or alternate spellings. Useful for brands not in the built-in dictionary.",
            "items": {
              "type": "string"
            }
          },
          "searchDateFrom": {
            "title": "Search date window — start (Beijing time, optional, search mode only)",
            "type": "string",
            "description": "<b>Search mode only. Optional — leave both date fields empty and the Actor behaves exactly as before.</b> Earliest post date to return, as ISO <code>2026-09-01</code> (= 00:00:00 that day) or <code>2026-09-01T08:00:00</code>. <b>Dates are Beijing time (UTC+08:00) unless you write an offset</b> — the same zone every delivered row's <code>createdAtIso</code> carries, so the calendar days you ask for are the days you can group by. Write <code>2026-09-01T00:00:00Z</code> if you want UTC; an offset you write is always honoured. Weibo's search API has <b>no lower-bound parameter</b> — only an end time — so the Actor starts at <b>Search date window — end</b> and pages <i>backwards</i> until the posts it receives are older than this start date. Posts outside the window are dropped before delivery, so <b>you are never charged for them</b>. The run's status message reports the window in both zones and the oldest and newest post actually delivered, and says plainly when Weibo stopped before reaching this date (page limits, run time, or your charge limit) — a window is a <i>sample</i> of the period, never a guaranteed complete archive of it. An unparseable date, or a start at or after the end, <b>fails the run</b> with an explanatory message and charges nothing, rather than silently scraping a different period."
          },
          "searchDateTo": {
            "title": "Search date window — end (Beijing time, optional, search mode only)",
            "type": "string",
            "description": "<b>Search mode only. Optional.</b> Latest post date to return, as ISO <code>2026-09-03</code> (= 23:59:59 that day, so a plain date range covers whole days) or <code>2026-09-03T18:00:00</code>. <b>Dates are Beijing time (UTC+08:00) unless you write an offset</b>, matching every row's <code>createdAtIso</code>. This is the point Weibo starts paging back from (<code>endtime</code>, the one date control the search API honours). Leave it empty to start from now. Posts newer than this are dropped before delivery and <b>not charged</b>. Works with <b>maxResults</b> and your run charge limit, which still cap the run. In <b>deltaMode</b> the dates still filter the rows, but the deeper history walk stays off (as it always is in deltaMode), so a delta run only checks Weibo's first result window against your dates — the status message says so. These two fields apply to <b>search mode only</b>; in any other mode the run tells you in its status message that they did not filter it."
          },
          "userIds": {
            "title": "User IDs or profile URLs",
            "type": "array",
            "description": "Weibo user IDs (numeric) or profile URLs (weibo.com/u/<id> or weibo.com/<id>). <b>user_profile</b> returns one profile row per ID — screen name, followers, following, post count, verification, bio, avatar, account creation date — with <b>no login</b>, charged per profile delivered; an ID Weibo does not resolve (deleted, banned or not a user) is not charged, and duplicates are fetched once. The run status names up to 10 IDs per reason; the full list of IDs not delivered, and why, is saved in the run's key-value store as PROFILE_REPORT. Custom addresses such as weibo.com/name are not resolved: use the numeric ID. <b>user_posts</b>: NOTE (measured 2026-08-28): Weibo now serves user timelines only to logged-in sessions — anonymous requests get an empty timeline from both datacenter and residential IPs. For user_posts, add your cookieString below (log in at weibo.com, F12 > Network, copy the Cookie header). The prefill shows the ID format. All other modes work with no login.",
            "items": {
              "type": "string"
            }
          },
          "postIds": {
            "title": "Post IDs or URLs",
            "type": "array",
            "description": "Weibo post IDs or post URLs, for <b>post_comments</b> and <b>post_details</b>. Accepted forms: the numeric ID (<code>5346708137705813</code>), the short code Weibo shows in post links (<code>RjB5HwkDj</code>), <code>https://weibo.com/&lt;uid&gt;/&lt;id or code&gt;</code>, <code>https://m.weibo.cn/detail/&lt;id&gt;</code> and <code>https://m.weibo.cn/status/&lt;id or code&gt;</code>. <b>post_details</b> returns one row per post — its current text, reposts, comments and likes, author, date, images and video — with <b>no login</b>, charged per post delivered; the same post pasted twice in any two forms is fetched, and charged, once. A post Weibo does not show (deleted, hidden or never existed) gives no row and is not charged; the run status names up to 10 per reason, and the full list of posts not delivered, and why, is saved in the run's key-value store as POST_DETAILS_REPORT.",
            "items": {
              "type": "string"
            }
          },
          "complaintCompanyIds": {
            "title": "Black Cat company IDs or company URLs (complaints mode)",
            "type": "array",
            "description": "<b>complaints mode only.</b> Companies on Black Cat (黑猫投诉, tousu.sina.com.cn — Sina's consumer-complaint platform) whose complaints you want, one row per complaint, with <b>no login</b>. Black Cat ranks each list by recent activity (a complaint filed weeks ago that just got a reply sits next to today's), not by filing date. Paste the company page URL (<code>https://tousu.sina.com.cn/company/view/?couid=2092643773</code>) or just the number after <code>couid=</code>. Lists are read in the order given and share <b>maxResults</b>; duplicates are read once. Black Cat shows anonymous visitors the first <b>500 complaints (50 pages) of each list</b> — the run says so when a company has more. An ID Black Cat rejects or returns nothing for produces no row and is not charged; the status names it, and the key-value store record COMPLAINTS_REPORT lists every company not delivered in full. <b>Leave empty</b> for Black Cat's public latest-complaints feed across all companies (what China is complaining about right now).",
            "items": {
              "type": "string"
            }
          },
          "maxResults": {
            "title": "Max results (raise for bulk monitoring)",
            "minimum": 1,
            "maximum": 5000,
            "type": "integer",
            "description": "Maximum results to return (applies per user in user_posts mode; user_profile ignores it and returns one row per user ID; in complaints mode it caps the complaints of the whole run, shared across the company IDs in order; in post_details mode it caps the posts of the run, one row per post). Typical patterns: 20-50 for a quick lookup, 100-300 for daily brand / sentiment monitoring, 500 for a full keyword sweep or AI-training pull. Higher values automatically walk more result pages. See the Pricing tab for per-result cost.",
            "default": 100
          },
          "includeComments": {
            "title": "Also scrape comments for each post (no cookie needed)",
            "type": "boolean",
            "description": "In <b>search</b>, <b>user posts</b> and <b>hot timeline</b> modes, also pull the comment thread of every post found and return each comment as its own row (tagged <code>mode: post_comment</code>, linked by <code>postId</code>). Comments are where the opinion is — the post is the prompt, the replies are the sentiment. <b>No login or cookie required.</b> Best paired with <b>hot timeline</b>: measured 2026-08-05, 5 trending posts returned 89 rows with maxComments 20 and 114 with 50 (~18-23x), because those posts actually carry threads — a broad keyword search mostly returns posts with no replies. In <b>search</b>, the newest posts rarely have replies yet: set <b>searchDateTo</b> 2 or more days back (Beijing time) and the same keyword returns older posts, which have had time to collect them. Posts with no thread are skipped automatically, so you are never billed for an empty round-trip. <b>Tip for volume:</b> a narrow, high-engagement query returns far bigger threads than a broad one — a generic keyword mostly returns posts with zero comments. This multiplies rows and cost — 100 posts × 50 comments is ~5,100 rows instead of 100 — so cap it with <b>Max comments per post</b>. Off by default.",
            "default": false
          },
          "fetchFullText": {
            "title": "Full text of long posts",
            "type": "boolean",
            "description": "Weibo's list endpoints send only the first ~150 characters of a long post (about half of search results). With this on, each cut-off post is fetched in full at no extra charge — same rows, same price. Turn it off only if you need the fastest possible run.",
            "default": true
          },
          "geoRollup": {
            "title": "Province rollup — where in China the conversation is",
            "type": "boolean",
            "description": "Adds one extra row per location (Chinese provinces; overseas countries flagged <code>isOverseas</code>), computed from the location Weibo already attaches to each post (<code>发布于 广东</code>). Each rollup row carries <code>province</code>, <code>provinceEn</code>, <code>postCount</code>, <code>sharePct</code>, <code>promotedCount</code> and, when sentiment is on, <code>sentimentMean</code> — so you can see whether a story is national or contained to one province, and which regions are driving it. <b>No extra requests and no extra scraping time:</b> it is computed from data the run already fetched. Rollup rows bill as ordinary results. <b>Honest sampling:</b> if fewer than 50 posts in the run carry a location, no province rows are emitted and none are charged; between 50 and 200 they ship with <code>lowConfidence: true</code>. Every row states its own <code>sampleSize</code>. Works with search, hot timeline and user posts. Off by default.",
            "default": false
          },
          "adFilter": {
            "title": "Promoted posts — keep, drop, or study them",
            "enum": [
              "all",
              "organic_only",
              "promo_only"
            ],
            "type": "string",
            "description": "Weibo mixes paid promotion into keyword results, and those posts used to arrive indistinguishable from public opinion. Every row now carries <code>isPromoted</code>, and this decides what to do with it. <b>organic_only</b> excludes promoted posts, so a brand or sentiment study is not diluted by advertising — and because the filter runs before billing, <b>you are not charged for the rows it removes</b>, which means your bill goes down. <b>promo_only</b> flips it into a China paid-social monitor: which advertisers are buying against your keyword. <b>all</b> is the default and changes nothing. <b>Note on counts:</b> filtering happens after fetching, so asking for 100 posts with organic_only returns fewer than 100 — raise <code>maxResults</code> if you need a fixed number back. The promoted share varies a lot by query and hour (measured 0 of 53 on one search, 1 of 10 on one hot timeline), so no fixed percentage is claimed.",
            "default": "all"
          },
          "maxComments": {
            "title": "Max comments per post",
            "minimum": 1,
            "maximum": 1000,
            "type": "integer",
            "description": "Maximum comments to fetch per post (post_comments mode)",
            "default": 20
          },
          "includeAuthorProfiles": {
            "title": "Author profiles — one profile row per post author (charged per row)",
            "type": "boolean",
            "description": "Off by default. In <b>search</b>, <b>hot timeline</b> and <b>user posts</b> modes, after the posts are collected the run also fetches the public profile of each distinct post author (no login) and adds it as its own row, tagged <code>mode: author_profile</code>, with the same fields as a <b>user profile</b> row: followers, following, post count, verification, bio, location, creation date. Join it to the posts on <code>authorId</code> = <code>userId</code>. Weibo's search sends no follower count with a post, so where a profile comes back it also fills the post's <code>authorFollowers</code> / <code>authorFollowing</code> that Weibo left null; an author whose profile request fails gets no row and no charge, and their posts keep <code>null</code>. <b>Cost:</b> each profile row is one result, charged like a post ($0.035), so 100 posts by 60 different authors come to 160 results. At most 50 authors per run, one every 2 s, each once, within about 3 minutes (the rows are delivered after the profiles, so a run takes up to ~3 minutes longer), and never more than your run's charge limit can pay for. In <b>deltaMode</b> only the authors of the run's NEW posts are profiled, so an author can be charged again on a later run that brings a new post of theirs.",
            "default": false
          },
          "cookieString": {
            "title": "Cookie string (optional)",
            "type": "string",
            "description": "Only needed for user_posts. Leave empty for search, hot_timeline, hot_search, hot_search_delta, post_comments and user_profile — a stale cookie here replaces the anonymous session. complaints (Black Cat) does not use it. Log in at weibo.com, F12 > Network, copy the full Cookie request header."
          },
          "sentimentAnalysis": {
            "title": "Sentiment analysis",
            "type": "boolean",
            "description": "Tag each post/comment (and complaint, in complaints mode — scored on its title and summary) with lexicon-based Chinese sentiment — polarity + a -1.0…+1.0 score. Optional add-on — for Chinese text it loads a model, so enable with run memory set to 512 MB or more.",
            "default": false
          },
          "deltaMode": {
            "title": "Delta mode — skip rows an earlier run already delivered",
            "type": "boolean",
            "description": "Off by default. When on, a run in <b>search</b>, <b>user posts</b> or <b>hot timeline</b> returns only rows that no earlier run with the same Delta state key delivered: posts are deduped by post ID, comment rows (includeComments) by their own comment ID, and rows an earlier run delivered are skipped and not charged. In search mode the deeper history walk is off, so a delta run reads only Weibo's first result window. (Hot Search Delta mode has its own built-in delta and ignores this toggle.) <b>Know the trade before you switch it on:</b> a full run returns the whole current result set and bills for it; a delta run returns only what has appeared since the last run. On fast-moving boards (trending, hot search, popular) that is still a healthy batch every run. On a narrow keyword or a single creator it can be a handful of rows, or none on a quiet day — which is the feature working, not a broken run. Pick delta when you want to be told what is NEW; leave it off when you want the data itself. In <b>complaints</b> mode it works per company list (or the public feed), by complaint number: a run returns only complaints no earlier run with the same Delta state key delivered, and reads each list to its end or to Black Cat's 500-complaint depth, because complaints already delivered can sit above new ones (they are skipped, not charged).",
            "default": false
          },
          "deltaStateKey": {
            "title": "Delta state key (delta modes)",
            "type": "string",
            "description": "Names an independent tracking stream for the delta features (Delta mode in search/user_posts/hot_timeline/complaints, and Hot Search Delta). Keep the default for a single scheduled monitor; use distinct keys (e.g. 'nike-weekly', 'hourly') to run several independent delta streams without collision. State persists across runs, so each scheduled run is compared with the previous one.",
            "default": "default"
          },
          "followUpComments": {
            "title": "Follow-up comments on earlier posts (search + Delta mode, charged per row)",
            "type": "boolean",
            "description": "Off by default. Only for <b>search</b> mode with <b>Delta mode</b> on. A delta search reads each post minutes after it is published, usually before anyone has replied, and the post then drops out of the newest results, so its replies never reach the monitor. With this on, the posts each delta run delivers are remembered with their posting time (in the Delta state key's record), and a later delta run re-reads each one's comment thread <b>once</b>: on the first run at least 24 hours after it was posted, before it is 7 days old, newest posts first. Only comments never delivered under this Delta state key come back, as ordinary comment rows (<code>mode: post_comment</code>, linked by <code>postId</code>) marked <code>isFollowUp: true</code>, up to <b>Max comments per post</b> per post, from at most 6 pages of each thread (about 120 new top-level comments). <b>Cost:</b> each comment row is one result, charged like a post ($0.035); a thread with no new comments adds nothing. At most 300 threads, 400 comment requests and about 4 minutes of re-reading per run, 2 s apart, and never more rows than your run's charge limit can pay for; posts not reached keep their place for the next run. The run's own rows are delivered after the re-reads, so a run takes up to ~4 minutes longer: keep the schedule interval longer than a run lasts. It starts with the posts delivered from the first run with this on. Does not run with adFilter = promo_only (comment rows carry no promotion label).",
            "default": false
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}