{
  "openapi": "3.0.1",
  "info": {
    "title": "AI Web Scraper with Your Own OpenAI or Claude Key",
    "description": "Extract structured data from any web page with AI. List the fields you want in plain English or paste a JSON schema, and get clean JSON rows back. Uses your own OpenAI, Anthropic, Gemini or OpenRouter key: no token markup, $4 per 1,000 pages, failed pages free.",
    "version": "0.1",
    "x-build-id": "vE8UwxgG7vsbEhSwn"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/oldjard~ai-web-scraper/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-oldjard-ai-web-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/oldjard~ai-web-scraper/runs": {
      "post": {
        "operationId": "runs-sync-oldjard-ai-web-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/oldjard~ai-web-scraper/run-sync": {
      "post": {
        "operationId": "run-sync-oldjard-ai-web-scraper",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "required": [
          "startUrls"
        ],
        "properties": {
          "startUrls": {
            "title": "Start URLs",
            "type": "array",
            "description": "The web pages to extract data from, one per line. Paste as many as you like; duplicates are removed. Turn on link following below to also visit pages linked from these.",
            "items": {
              "type": "object",
              "required": [
                "url"
              ],
              "properties": {
                "url": {
                  "type": "string",
                  "title": "URL of a web page",
                  "format": "uri"
                }
              }
            }
          },
          "fields": {
            "title": "Fields to extract",
            "type": "array",
            "description": "One field per line, in plain English: <code>name</code>, <code>name: what it is</code> or <code>name (type): what it is</code>. Types: text (default), number, integer, boolean, url, date, list. Example: <code>price (number): the current price, without the currency sign</code>. Each field becomes a column. Leave empty if you paste a JSON schema below.",
            "items": {
              "type": "string"
            }
          },
          "jsonSchema": {
            "title": "…or a JSON schema",
            "type": "object",
            "description": "Instead of the field list, paste a JSON Schema for one record (<code>{\"type\": \"object\", \"properties\": {...}}</code>). A schema with <code>\"type\": \"array\"</code> switches on list mode. Nested objects and arrays are supported. Overrides the field list."
          },
          "instructions": {
            "title": "Instructions (optional)",
            "type": "string",
            "description": "Extra guidance for the model, e.g. \"Prices in USD. Skip sponsored listings.\" With no fields and no schema, describe what you want here and the model picks the field names (less consistent across pages)."
          },
          "extractionMode": {
            "title": "Rows per page",
            "enum": [
              "single",
              "list"
            ],
            "type": "string",
            "description": "<b>One row per page</b> fills your fields once per page. <b>One row per item</b> finds every matching item on the page (each product in a category page, each job in a list) and returns one row per item.",
            "default": "single"
          },
          "llmProvider": {
            "title": "AI provider",
            "enum": [
              "openai",
              "anthropic",
              "google",
              "openrouter"
            ],
            "type": "string",
            "description": "Whose model reads the pages. Any OpenAI-compatible service (Groq, Together, DeepSeek, Mistral, a self-hosted server) works too: pick OpenAI and set the Base URL below.",
            "default": "openai"
          },
          "apiKey": {
            "title": "API key",
            "type": "string",
            "description": "Your API key for the provider above. Without a key the actor runs a free preview of the first 3 pages (the cleaned text the model would read, no extracted fields)."
          },
          "model": {
            "title": "Model",
            "type": "string",
            "description": "Leave empty to use the newest cheap, fast model your key can use (for example OpenAI's luna or nano tier, Claude Haiku, Gemini Flash-Lite). Or name any model your key can use, e.g. <code>gpt-6-sol</code>, <code>claude-sonnet-5-5</code>, <code>gemini-3.8-flash</code>, or for OpenRouter <code>anthropic/claude-sonnet-5.5</code>."
          },
          "baseUrl": {
            "title": "Base URL (OpenAI-compatible services)",
            "type": "string",
            "description": "Only for OpenAI-compatible services or proxies, e.g. <code>https://api.groq.com/openai/v1</code>. Must be https. Leave empty for the provider's own API."
          },
          "renderMode": {
            "title": "Browser rendering",
            "enum": [
              "off",
              "auto",
              "always"
            ],
            "type": "string",
            "description": "<b>Auto</b> (default) loads each page with a fast HTTP request and opens it in headless Chrome only if it is blocked or is an empty JavaScript app. Each page opened in Chrome adds one <i>browser-render</i> charge. For runs with many browser pages, give the run 2 GB of memory or more.",
            "default": "auto"
          },
          "contentScope": {
            "title": "Page content",
            "enum": [
              "auto",
              "full"
            ],
            "type": "string",
            "description": "What the model reads. <b>Main content</b> saves tokens and keeps the model focused. Choose <b>Whole page</b> if a field lives in the header, footer or menu.",
            "default": "auto"
          },
          "cssSelector": {
            "title": "CSS selector (optional)",
            "type": "string",
            "description": "Only read the part of the page that matches this CSS selector, e.g. <code>#product</code> or <code>.job-listing</code>. Cuts tokens a lot on big pages. Pages where it matches nothing are reported and not charged."
          },
          "includeLinks": {
            "title": "Keep link URLs",
            "type": "boolean",
            "description": "Give the model the URL of every link, so it can fill URL fields. Turn off to save tokens if you need no URLs.",
            "default": true
          },
          "includeImages": {
            "title": "Keep image URLs",
            "type": "boolean",
            "description": "Give the model image URLs and alt text, so it can fill image fields.",
            "default": true
          },
          "maxCrawlDepth": {
            "title": "Follow links: depth",
            "minimum": 0,
            "maximum": 5,
            "type": "integer",
            "description": "0 = only the start URLs. 1 = also the pages they link to (same website only), and so on. Use the include pattern below to stay on the pages you want, e.g. product pages.",
            "default": 0
          },
          "maxPagesPerStartUrl": {
            "title": "Follow links: max pages per start URL",
            "minimum": 1,
            "maximum": 10000,
            "type": "integer",
            "description": "Stop following links from a start URL after this many pages (the start page counts).",
            "default": 10
          },
          "linkIncludeGlobs": {
            "title": "Follow links: only URLs matching",
            "type": "array",
            "description": "Glob patterns over the full URL; <code>*</code> matches within one path segment and <code>**</code> matches anything. Example: <code>https://shop.example.com/products/*</code>. Empty = every same-site link.",
            "items": {
              "type": "string"
            }
          },
          "linkExcludeGlobs": {
            "title": "Follow links: skip URLs matching",
            "type": "array",
            "description": "Glob patterns for links to skip, e.g. <code>**/login*</code> or <code>**?sort=*</code>.",
            "items": {
              "type": "string"
            }
          },
          "maxPages": {
            "title": "Max pages in total",
            "minimum": 0,
            "type": "integer",
            "description": "Stop after this many pages across the whole run (0 = no limit). You can also set a maximum cost in the run options.",
            "default": 0
          },
          "maxTokensPerChunk": {
            "title": "Max tokens per chunk",
            "minimum": 1000,
            "maximum": 200000,
            "type": "integer",
            "description": "Pages longer than this (estimated) are split into chunks, each sent to the model separately and the results merged. Lower it to cut cost per call; raise it for models with big context windows.",
            "default": 8000
          },
          "maxChunksPerPage": {
            "title": "Max chunks per page",
            "minimum": 1,
            "maximum": 50,
            "type": "integer",
            "description": "Read at most this many chunks of a very long page. With one row per page the actor also stops early once every field is filled.",
            "default": 3
          },
          "maxOutputTokens": {
            "title": "Max output tokens",
            "minimum": 256,
            "maximum": 64000,
            "type": "integer",
            "description": "The most tokens the model may write per call (reasoning models count their thinking here too). Raise it for listing pages with many items.",
            "default": 8000
          },
          "maxConcurrency": {
            "title": "Pages in parallel",
            "minimum": 1,
            "maximum": 20,
            "type": "integer",
            "description": "How many pages are processed at once. Lower it if your provider answers with rate-limit errors. Each website is still visited politely, at most 2 requests at a time.",
            "default": 5
          },
          "requestTimeoutSecs": {
            "title": "Page timeout (seconds)",
            "minimum": 5,
            "maximum": 180,
            "type": "integer",
            "description": "Give up on a page that hasn't answered within this many seconds. Reported as a timeout and not charged.",
            "default": 30
          },
          "includeMarkdown": {
            "title": "Add page text to the output",
            "type": "boolean",
            "description": "Add a <code>#markdown</code> column with the cleaned page text the model read. Handy for checking results.",
            "default": false
          },
          "dryRun": {
            "title": "Preview mode (no AI)",
            "type": "boolean",
            "description": "Convert the pages to clean markdown and estimate their tokens without calling the model or needing a key. Use it to tune the CSS selector and token settings before a big run. Charged per page like extraction.",
            "default": false
          },
          "canaryExpectations": {
            "title": "Monitoring: expected text",
            "type": "object",
            "description": "Optional, for scheduled health checks: a map of URL to text that must appear in that page's result, e.g. <code>{\"https://example.com/\": [\"Example Domain\"]}</code>. The run fails if more than 20% of the checks miss."
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}