{
  "data": {
    "id": "4QpHVhrVdrw4S1m8C",
    "userId": "HIMU0xktu1JeZb4Xl",
    "name": "pdf-text-markdown-extractor",
    "username": "fetch_cat",
    "description": "Extract public PDF URLs into clean text, Markdown, page content, OCR state, and document metadata for AI and automation workflows.",
    "isPublic": true,
    "createdAt": "2026-08-27T17:29:49.117Z",
    "modifiedAt": "2026-08-28T23:53:11.839Z",
    "taggedBuilds": {
      "latest": {
        "buildId": "9MpaHvkayJuGwE8bZ",
        "finishedAt": "2026-08-28T22:21:55.153Z",
        "buildNumberInt": 100014,
        "buildNumber": "0.1.14"
      }
    },
    "stats": {
      "totalBuilds": 15,
      "totalRuns": 57,
      "totalUsers": 2,
      "totalUsers7Days": 1,
      "totalUsers30Days": 1,
      "totalUsers90Days": 1,
      "lastRunStartedAt": "2026-09-20T08:08:41.952Z",
      "actorReviewCount": 0,
      "actorReviewRating": 0,
      "bookmarkCount": 0,
      "publicActorRunStats30Days": {
        "ABORTED": 0,
        "FAILED": 0,
        "SUCCEEDED": 23,
        "TIMED-OUT": 0,
        "TOTAL": 23
      }
    },
    "versions": [
      {
        "versionNumber": "0.0",
        "sourceType": "SOURCE_FILES",
        "buildTag": "latest"
      },
      {
        "versionNumber": "0.1",
        "sourceType": "SOURCE_FILES",
        "buildTag": "latest"
      }
    ],
    "defaultRunOptions": {
      "build": "latest",
      "timeoutSecs": 300,
      "memoryMbytes": 512
    },
    "exampleRunInput": {
      "body": "{\"pdfUrls\":[\"https://www.w3.org/WAI/ER/tests/xhtml/testfiles/resources/pdf/dummy.pdf\"],\"includeMarkdown\":true,\"includePages\":true,\"maxPages\":20}",
      "contentType": "application/json; charset=utf-8"
    },
    "categories": [
      "AI",
      "DEVELOPER_TOOLS",
      "AUTOMATION"
    ],
    "isDeprecated": false,
    "title": "PDF Text and Markdown Extractor",
    "pictureUrl": "https://apify-image-uploads-prod.s3.us-east-1.amazonaws.com/HIMU0xktu1JeZb4Xl-actor-4QpHVhrVdrw4S1m8C-AMfXpYAzye-actor-icon.png",
    "seoTitle": "PDF Text Extractor and PDF to Markdown API",
    "seoDescription": "Extract public PDF URLs into clean text, Markdown, per-page content, OCR state, and metadata for RAG, AI agents, research, and document automation.",
    "pricingInfos": [
      {
        "pricingModel": "PAY_PER_EVENT",
        "startedAt": "2026-08-27T17:30:14.000Z",
        "pricingPerEvent": {
          "actorChargeEvents": {
            "start": {
              "eventTitle": "Actor start",
              "eventDescription": "Charged once when a valid extraction run starts.",
              "eventPriceUsd": 0.005,
              "isOneTimeEvent": true
            },
            "pdf-processed": {
              "eventTitle": "PDF processed",
              "eventDescription": "Charged for each PDF with usable extracted text. Failed downloads and unreadable documents are not charged.",
              "eventPriceUsd": 0.005,
              "isPrimaryEvent": true
            }
          }
        },
        "isPriceChangeNotificationSuppressed": true,
        "createdAt": "2026-08-27T17:30:14.596Z",
        "apifyMarginPercentage": 0.2
      },
      {
        "pricingModel": "PAY_PER_EVENT",
        "startedAt": "2026-08-27T19:45:00.000Z",
        "reasonForChange": "Updated private launch pricing with tier discounts for PDF processing.",
        "pricingPerEvent": {
          "actorChargeEvents": {
            "start": {
              "eventTitle": "Actor start",
              "eventDescription": "Charged once when a valid extraction run starts.",
              "eventPriceUsd": 0.005,
              "isOneTimeEvent": true
            },
            "pdf-processed": {
              "eventTitle": "PDF processed",
              "eventDescription": "Charged for each PDF with usable extracted text. Failed downloads and unreadable documents are not charged.",
              "eventTieredPricingUsd": {
                "FREE": {
                  "tieredEventPriceUsd": 0.00575
                },
                "BRONZE": {
                  "tieredEventPriceUsd": 0.005
                },
                "SILVER": {
                  "tieredEventPriceUsd": 0.0039
                },
                "GOLD": {
                  "tieredEventPriceUsd": 0.003
                },
                "PLATINUM": {
                  "tieredEventPriceUsd": 0.002
                },
                "DIAMOND": {
                  "tieredEventPriceUsd": 0.0014
                }
              },
              "isPrimaryEvent": true
            }
          }
        },
        "createdAt": "2026-08-27T19:45:13.924Z",
        "apifyMarginPercentage": 0.2
      },
      {
        "pricingModel": "PAY_PER_EVENT",
        "startedAt": "2026-08-27T20:00:00.000Z",
        "reasonForChange": "Updated private launch pricing with tier discounts for PDF processing.",
        "pricingPerEvent": {
          "actorChargeEvents": {
            "start": {
              "eventTitle": "Actor start",
              "eventDescription": "Charged once when a valid extraction run starts.",
              "eventPriceUsd": 0.005,
              "isOneTimeEvent": true
            },
            "pdf-processed": {
              "eventTitle": "PDF processed",
              "eventDescription": "Charged for each PDF with usable extracted text. Failed downloads and unreadable documents are not charged.",
              "eventTieredPricingUsd": {
                "FREE": {
                  "tieredEventPriceUsd": 0.0046
                },
                "BRONZE": {
                  "tieredEventPriceUsd": 0.004
                },
                "SILVER": {
                  "tieredEventPriceUsd": 0.0031201
                },
                "GOLD": {
                  "tieredEventPriceUsd": 0.0024
                },
                "PLATINUM": {
                  "tieredEventPriceUsd": 0.0016
                },
                "DIAMOND": {
                  "tieredEventPriceUsd": 0.0011201
                }
              },
              "isPrimaryEvent": true
            }
          }
        },
        "isPriceChangeNotificationSuppressed": true,
        "createdAt": "2026-08-27T19:56:43.511Z",
        "apifyMarginPercentage": 0.2
      }
    ],
    "notice": "NONE",
    "isCritical": false,
    "isGeneric": false,
    "hasNoDataset": false,
    "isSourceCodeHidden": true,
    "standbyUrl": null,
    "actorPermissionLevel": "LIMITED_PERMISSIONS",
    "readmeSummary": "## PDF Text and Markdown Extractor\n\nConverts public PDF links into clean extracted text and pragmatic Markdown suitable for retrieval-augmented generation (RAG), search, indexing, research, and document automation. The Actor extracts native text layers, generates an LLM-ready Markdown rendition (a semantic, chunkable text representation rather than a visual reconstruction), and produces page-level segments that include OCR state and extraction warnings. It can run OCR on pages that lack a usable text layer with configurable language support, harvest embedded PDF metadata (title, author, subject, creator, producer, dates), and produce character and word counts and explicit per-document processing status and error information for batch workflows.\n\n## Use cases\n\n- Prepare PDF text and Markdown for RAG and retrieval pipelines for AI agents and search.\n- Batch-extract research papers, reports, and archives for analysis and indexing.\n- Convert manuals, invoices, policy documents, and compliance documents into structured text for operations and process automation.\n- Produce LLM-ready Markdown and page-level chunks for summarization, chunking, and downstream NLP workflows.\n- Recover readable content from scanned PDFs using OCR with language selection.\n- Integrate PDF-to-Markdown extraction into programmatic workflows and automated document-processing pipelines.",
    "deploymentKey": "ssh-rsa AAAAB3NzaC1yc2EAAAADAQABAAABAQCod6sEaHjBcyH3zpF6QbUvfHZcQcPL1mHqHl3natvhnWEPLfKj8NrOR4jWTBzbIq0hazNhhvuvgNYH6XuuIXqhfHN0+yFdHXOPBv7NraspdKQtZfbb18/euMWqKk0cggjgO/n+HbwuQVAkBrKuSFR9SMN4EdjI2cWMLS6B6/bREBBDfM6NsFyqevLVZbXkN6kvHSpi3ydTALD8PQhz/9IR5THWULg27DFVlv1S8VOY2GbZh2gSpNTbJlR9JFGSDR83gJLElvdf/BFYYRcw21XNFwp63TTNF1es5ktBSh1G0x+Um0uYp64dmhFHaQPAaLMj328XTu7DbDIF+Zb7wpgx \n"
  }
}