{
  "data": {
    "id": "aYG0l9s7dbB7j3gbS",
    "userId": "ZscMwFR5H7eCtWtyh",
    "name": "website-content-crawler",
    "username": "apify",
    "description": "Crawl websites and extract text content to feed AI models, LLM applications, vector databases, or RAG pipelines. The Actor supports rich formatting using Markdown, cleans the HTML, downloads files, and integrates well with 🦜🔗 LangChain, LlamaIndex, and the wider LLM ecosystem.",
    "restartOnError": true,
    "isPublic": true,
    "createdAt": "2023-03-27T15:06:04.992Z",
    "modifiedAt": "2026-09-11T16:01:17.953Z",
    "taggedBuilds": {
      "beta": {
        "buildId": "Y7IIeQmLbnSJaxw9s",
        "finishedAt": "2026-09-08T14:47:56.667Z",
        "buildNumberInt": 100361,
        "buildNumber": "0.1.361"
      },
      "version-0": {
        "buildId": "u8gClHFAIyDHCQq0J",
        "finishedAt": "2026-09-08T16:05:17.236Z",
        "buildNumberInt": 300097,
        "buildNumber": "0.3.97"
      },
      "canary": {
        "buildId": "Ckj0LaU5DcvuAaVMP",
        "finishedAt": "2026-08-20T11:28:27.416Z",
        "buildNumberInt": 200129,
        "buildNumber": "0.2.129"
      },
      "test": {
        "buildId": "WATn47GcMoHu9qkGp",
        "finishedAt": "2026-09-11T16:01:17.953Z",
        "buildNumberInt": 9900077,
        "buildNumber": "0.99.77"
      }
    },
    "stats": {
      "totalBuilds": 834,
      "totalRuns": 42508238,
      "totalUsers": 156155,
      "totalUsers7Days": 5032,
      "totalUsers30Days": 10995,
      "totalUsers90Days": 20542,
      "lastRunStartedAt": "2026-09-13T07:00:45.699Z",
      "totalMetamorphs": 493,
      "publicActorRunStats30Days": {
        "ABORTED": 11379,
        "FAILED": 23299,
        "SUCCEEDED": 3032970,
        "TIMED-OUT": 71958,
        "TOTAL": 3139606
      },
      "actorReviewCount": 233,
      "actorReviewRating": 4.611053865140532,
      "bookmarkCount": 2694
    },
    "versions": [
      {
        "versionNumber": "0.0",
        "sourceType": "SOURCE_FILES",
        "buildTag": "beta"
      },
      {
        "versionNumber": "0.1",
        "sourceType": "SOURCE_FILES",
        "buildTag": "beta"
      },
      {
        "versionNumber": "0.2",
        "sourceType": "SOURCE_FILES",
        "buildTag": "canary"
      },
      {
        "versionNumber": "0.3",
        "sourceType": "SOURCE_FILES",
        "buildTag": "version-0"
      },
      {
        "versionNumber": "0.99",
        "sourceType": "GIT_REPO",
        "buildTag": "test"
      }
    ],
    "defaultRunOptions": {
      "build": "version-0",
      "timeoutSecs": 360000,
      "memoryMbytes": 8192,
      "restartOnError": true
    },
    "exampleRunInput": {
      "body": "{ \"helloWorld\": 123 }",
      "contentType": "application/json; charset=utf-8"
    },
    "categories": [
      "AI",
      "DEVELOPER_TOOLS"
    ],
    "isDeprecated": false,
    "title": "Website Content Crawler",
    "pictureUrl": "https://apify-image-uploads-prod.s3.amazonaws.com/aYG0l9s7dbB7j3gbS/PfToENkJZxahzPDu3-CleanShot_2023-03-28_at_10.40.20_2x.png",
    "seoTitle": null,
    "seoDescription": null,
    "notice": "NONE",
    "isCritical": true,
    "isGeneric": false,
    "hasNoDataset": false,
    "isSourceCodeHidden": true,
    "actorStandby": {
      "isEnabled": false,
      "disableStandbyFieldsOverride": false,
      "tenancy": "SINGLE_TENANT",
      "maxRequestsPerActorRun": 4,
      "desiredRequestsPerActorRun": 3,
      "idleTimeoutSecs": 300,
      "build": "latest",
      "memoryMbytes": 1024,
      "shouldPassActorInput": false,
      "isConsoleAuthEnabled": false,
      "isTokenlessEnabled": false
    },
    "standbyUrl": null,
    "actorPermissionLevel": "LIMITED_PERMISSIONS",
    "readmeSummary": "## Website Content Crawler\n\nWebsite Content Crawler is an Apify Actor for deep crawling and content extraction from websites, producing cleaned textual content, structured Markdown or HTML, document file downloads, page metadata, screenshots, and optional per-page AI summaries for LLM ingestion, semantic search, and Retrieval-Augmented Generation (RAG). It is built on Crawlee and supports adaptive crawling modes (headless browser rendering and raw HTTP requests), browser fingerprinting and proxy use to mitigate anti-scraping protections, JavaScript-rendered page handling (including infinite scroll and clickable-element expansion), sitemap discovery, cookie-based authentication for pages behind login, DOM transformation to remove navigation/header/footer/modals/cookie warnings, and targeted extraction to retain only the main article content. The Actor can download common document formats (PDF, DOC/DOCX, XLS/XLSX, CSV) linked from pages and can generate concise AI summaries of page content via an LLM; outputs include cleaned text/Markdown/HTML, document assets, and metadata such as author, language, and publish date. Integrations and workflows commonly pair the crawler output with LangChain, LlamaIndex, vector databases (Pinecone, Qdrant), and OpenAI-based tooling for embedding, indexing, and query interfaces.\n\n## Use cases\n\n- Extracting documentation, knowledge bases, help sites, or blogs to feed LLMs and AI applications.\n- Building knowledge-grounded customer support chatbots that answer from a website's content.\n- Creating personalized content and adapting LLM output to a customer's tone by training or fine-tuning on crawled site content.\n- Enabling Retrieval-Augmented Generation (RAG) pipelines and semantic search by producing text and metadata for embedding and indexing.\n- Summarization, translation, proofreading, and bulk content transformation workflows for large collections of web pages.\n- Populating or updating custom GPTs and other LLM agents with website-specific knowledge.",
    "deploymentKey": "ssh-rsa AAAAB3NzaC1yc2EAAAADAQABAAABAQCe3jwR6yJ3C2l7b+V4hx4X5s+Z3XFrGmau9Yy0GC82VkTDpCKgEflGqdaLJIFmeWT1UHDwOIrCWSwk4W4DEVYQwJE3sgkoUFBGbBo0opLc/vyZFSnvjyleUwGt0vxZg2EyEgH1i9+aRirUHVNAat92Au1viClnEih1lffP0WyN4VuVgrQO0Qf+UcPfq87Adeb9UK9ZbwcWx/yhnUMwIHbWnohshyWPCWJPfiDEUFcdmFs3JfLlQZJ6y79g9zw3AFppJEIeVJhK3bHIa4mfLND049SWJxhR/prvLrCk6ZbOI2D19/CeYJNmCGxSBjPKeohIFwBvShnQZhkXIN5YkZAz \n"
  }
}