{
  "openapi": "3.0.1",
  "info": {
    "title": "AI Data Extractor & AI Web Scraper: Pages to Structured JSON",
    "description": "Turn web pages, or another Actor's dataset, into structured JSON with an LLM: an AI web scraper for pages you already have. Give a JSON schema or plain-English fields; every answer is validated, and you pay $5 per 1,000 valid pages plus the model's tokens. No browser.",
    "version": "1.0",
    "x-build-id": "PUv9hyeDmKOn3S72N"
  },
  "servers": [
    {
      "url": "https://api.apify.com/v2"
    }
  ],
  "paths": {
    "/acts/humble-echidna~ai-extract/run-sync-get-dataset-items": {
      "post": {
        "operationId": "run-sync-get-dataset-items-humble-echidna-ai-extract",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for its completion, and returns Actor's dataset items in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    },
    "/acts/humble-echidna~ai-extract/runs": {
      "post": {
        "operationId": "runs-sync-humble-echidna-ai-extract",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor and returns information about the initiated run in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK",
            "content": {
              "application/json": {
                "schema": {
                  "$ref": "#/components/schemas/runsResponseSchema"
                }
              }
            }
          }
        }
      }
    },
    "/acts/humble-echidna~ai-extract/run-sync": {
      "post": {
        "operationId": "run-sync-humble-echidna-ai-extract",
        "x-openai-isConsequential": false,
        "summary": "Executes an Actor, waits for completion, and returns the OUTPUT from Key-value store in response.",
        "tags": [
          "Run Actor"
        ],
        "requestBody": {
          "required": true,
          "content": {
            "application/json": {
              "schema": {
                "$ref": "#/components/schemas/inputSchema"
              }
            }
          }
        },
        "parameters": [
          {
            "name": "token",
            "in": "query",
            "required": true,
            "schema": {
              "type": "string"
            },
            "description": "Enter your Apify token here"
          }
        ],
        "responses": {
          "200": {
            "description": "OK"
          }
        }
      }
    }
  },
  "components": {
    "schemas": {
      "inputSchema": {
        "type": "object",
        "properties": {
          "urls": {
            "title": "Web page URLs",
            "type": "array",
            "description": "The pages to extract from, one per line, e.g. https://books.toscrape.com/catalogue/a-light-in-the-attic_1000/index.html; a missing https:// is added. Each page is read once, no crawling. Pages the site's robots.txt disallows or opts out of AI crawlers are skipped, as are pages that need JavaScript to show their content. Ignored when datasetId is set.",
            "default": [
              "https://books.toscrape.com/catalogue/a-light-in-the-attic_1000/index.html"
            ],
            "items": {
              "type": "string"
            }
          },
          "datasetId": {
            "title": "Or: page URLs from a dataset (chain after another actor)",
            "type": "string",
            "description": "Default empty. One of your Apify datasets, picked here or given by id, e.g. a Google Maps or web scraper's results; to chain this actor after another in an integration, `{{resource.defaultDatasetId}}`. Each item's page URL is read like a line of urls (urls is ignored). Each row carries sourceTitle, sourcePlaceId and sourceIndex from its item. Reads at most 20,000 items and 10,000 URLs, read-only."
          },
          "datasetUrlField": {
            "title": "Field with the page URL (dataset)",
            "type": "string",
            "description": "Only with datasetId: the item field that holds the page URL, e.g. `website`, or a dotted path such as `metadata.url`. Leave empty (the default) to find it automatically: the first of url, pageUrl, link, website and loadedUrl that has a web address in the first 100 items, then the same names one level down. Google Maps links are never used."
          },
          "fields": {
            "title": "Fields to extract (plain English)",
            "type": "string",
            "description": "What to pull from each page, comma-separated or one per line, e.g. `product name, price, currency, in stock`. Add a hint after a colon: `price: the sale price, not the list price`. Each becomes a camelCase key in data (productName, price, ...) holding text, a number, true/false, a list of texts or null. Ignored when schema is set. At most 50 fields.",
            "default": "title, price, currency, in stock, number available, UPC"
          },
          "schema": {
            "title": "Or: JSON Schema for one record",
            "type": "object",
            "description": "Default empty. A JSON Schema for the object to extract from each page, with \"type\": \"object\" at the top, e.g. {\"type\": \"object\", \"properties\": {\"price\": {\"type\": \"number\"}, \"services\": {\"type\": \"array\", \"items\": {\"type\": \"string\"}}}}. Every answer is validated against it; one that doesn't match is returned with valid false and errors, and not charged. Overrides fields."
          },
          "instructions": {
            "title": "Extra instructions for the model",
            "type": "string",
            "description": "Default empty. Optional guidance for the model, e.g. `Prices are in EUR unless the page says otherwise` or `Only list services the business offers, not blog topics`. At most 4,000 characters."
          },
          "model": {
            "title": "Model (OpenRouter id)",
            "type": "string",
            "description": "The model that reads each page, as an OpenRouter model id. Default `openai/gpt-4.1-mini`: accurate and cheap for extraction. Cheaper: `openai/gpt-4.1-nano` or `google/gemini-2.5-flash-lite`; harder pages: `openai/gpt-5.4-mini`. It must support structured outputs. Its tokens are billed to your Apify account by Apify's OpenRouter actor, at its prices.",
            "default": "openai/gpt-4.1-mini"
          },
          "pageContent": {
            "title": "What the model reads",
            "enum": [
              "fullPage",
              "mainContent"
            ],
            "type": "string",
            "description": "`fullPage` (default): the whole visible page as Markdown except scripts, forms and navigation menus, so prices, stock and footer contact details are included. `mainContent`: only the article or main content, as page-to-markdown returns it; fewer tokens, best for articles and docs. Either way the page's schema.org JSON-LD is sent too.",
            "default": "fullPage"
          },
          "maxTokensPerPage": {
            "title": "Max tokens per page",
            "minimum": 500,
            "maximum": 100000,
            "type": "integer",
            "description": "Hard cap on the page text sent to the model, in tokens (about 4 characters each). Default 8000. A longer page keeps its start and its end (where footers hold addresses) and has truncated true. From 500 to 100000. Lower it to cut token cost on long pages.",
            "default": 8000
          },
          "maxPages": {
            "title": "Max pages per run",
            "minimum": 1,
            "maximum": 10000,
            "type": "integer",
            "description": "Stop after this many pages are extracted, e.g. 20. From 1 to 10000; leave empty (the default) for no limit. Pages that fail or don't match the schema don't count. The run also stops cleanly at the maximum cost per run you set in the run options, whichever comes first."
          }
        }
      },
      "runsResponseSchema": {
        "type": "object",
        "properties": {
          "data": {
            "type": "object",
            "properties": {
              "id": {
                "type": "string"
              },
              "actId": {
                "type": "string"
              },
              "userId": {
                "type": "string"
              },
              "startedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "finishedAt": {
                "type": "string",
                "format": "date-time",
                "example": "2025-01-08T00:00:00.000Z"
              },
              "status": {
                "type": "string",
                "example": "READY"
              },
              "meta": {
                "type": "object",
                "properties": {
                  "origin": {
                    "type": "string",
                    "example": "API"
                  },
                  "userAgent": {
                    "type": "string"
                  }
                }
              },
              "stats": {
                "type": "object",
                "properties": {
                  "inputBodyLen": {
                    "type": "integer",
                    "example": 2000
                  },
                  "rebootCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "restartCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "resurrectCount": {
                    "type": "integer",
                    "example": 0
                  },
                  "computeUnits": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "options": {
                "type": "object",
                "properties": {
                  "build": {
                    "type": "string",
                    "example": "latest"
                  },
                  "timeoutSecs": {
                    "type": "integer",
                    "example": 300
                  },
                  "memoryMbytes": {
                    "type": "integer",
                    "example": 1024
                  },
                  "diskMbytes": {
                    "type": "integer",
                    "example": 2048
                  }
                }
              },
              "buildId": {
                "type": "string"
              },
              "defaultKeyValueStoreId": {
                "type": "string"
              },
              "defaultDatasetId": {
                "type": "string"
              },
              "defaultRequestQueueId": {
                "type": "string"
              },
              "buildNumber": {
                "type": "string",
                "example": "1.0.0"
              },
              "containerUrl": {
                "type": "string"
              },
              "usage": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "integer",
                    "example": 1
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              },
              "usageTotalUsd": {
                "type": "number",
                "example": 0.00005
              },
              "usageUsd": {
                "type": "object",
                "properties": {
                  "ACTOR_COMPUTE_UNITS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATASET_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "KEY_VALUE_STORE_WRITES": {
                    "type": "number",
                    "example": 0.00005
                  },
                  "KEY_VALUE_STORE_LISTS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_READS": {
                    "type": "integer",
                    "example": 0
                  },
                  "REQUEST_QUEUE_WRITES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_INTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "DATA_TRANSFER_EXTERNAL_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_RESIDENTIAL_TRANSFER_GBYTES": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_SERPS": {
                    "type": "integer",
                    "example": 0
                  },
                  "PROXY_UNBLOCKER_UNITS": {
                    "type": "integer",
                    "example": 0
                  }
                }
              }
            }
          }
        }
      }
    }
  }
}