{
  "name": "pdf-to-text",
  "summary": "Extracts the text layer of a PDF, page by page, in reading order (no OCR).",
  "rationale": "Pulling a PDF text layer needs a real PDF engine and a hardened fetch: an agent that fetches the URL itself walks past every SSRF control, and carrying pdfjs per call is not something an agent does inline. Offline, best-effort text-layer extraction; no OCR. No external CMap or standard-font data is configured or fetched. PDFs with self-contained character mappings can yield text, including CJK; documents requiring unavailable predefined CMaps can return empty or partial text without an error. Embedded ToUnicode alone does not guarantee extraction when the encoding still needs an external CMap. Missing standard-font data does not necessarily prevent extraction. A successful response does not certify complete text; truncated reports only the max_pages limit.",
  "family": "documents",
  "price_usd": "0.005",
  "free": false,
  "timeout_ms": 40000,
  "max_timeout_seconds": 61,
  "input_schema": {
    "$schema": "https://json-schema.org/draft/2020-12/schema",
    "type": "object",
    "properties": {
      "url": {
        "type": "string",
        "maxLength": 2048,
        "format": "uri",
        "description": "Absolute http(s) URL of the PDF, up to 2048 characters; redirects are followed and the body must start with the %PDF- signature."
      },
      "max_pages": {
        "default": 100,
        "description": "How many leading pages to extract, 1 to 1000; a shorter document is not an error, and a longer one is cut and reported as truncated.",
        "type": "integer",
        "minimum": 1,
        "maximum": 1000
      }
    },
    "required": [
      "url"
    ],
    "additionalProperties": false,
    "description": "Offline, best-effort text-layer extraction; no OCR. No external CMap or standard-font data is configured or fetched. PDFs with self-contained character mappings can yield text, including CJK; documents requiring unavailable predefined CMaps can return empty or partial text without an error. Embedded ToUnicode alone does not guarantee extraction when the encoding still needs an external CMap. Missing standard-font data does not necessarily prevent extraction. A successful response does not certify complete text; truncated reports only the max_pages limit. PDF text extraction accepts at most 1000000 raw and separately 1000000 output UTF-16 code units, 10000 items per page and 100000 items across selected pages. Exceeding a limit rejects the call with 413 too_large. truncated refers only to max_pages."
  },
  "input_example": {
    "url": "https://grist.tools/samples/documents/report.pdf",
    "max_pages": 100
  },
  "output_example": {
    "source_url": "https://grist.tools/samples/documents/report.pdf",
    "total_pages": 2,
    "extracted_pages": 2,
    "truncated": false,
    "chars": 122,
    "text": "Sample Report\nThis is a public test document.\nIt has two pages of plain text.\n\nPage Two\nThe second page closes the report."
  },
  "errors": [
    "invalid_input",
    "blocked_target",
    "unreachable_target",
    "upstream_timeout",
    "unsupported_content_type",
    "too_large",
    "unprocessable",
    "internal"
  ]
}