{
  "name": "pdf-to-markdown",
  "summary": "Converts a PDF to Markdown, inferring headings from font size (best-effort, not layout-perfect; no OCR).",
  "rationale": "A PDF has no heading structure, only sized glyphs; recovering Markdown needs a real PDF engine plus a hardened fetch, which an agent cannot do inline without walking past the SSRF controls. Offline, best-effort text-layer extraction; no OCR. No external CMap or standard-font data is configured or fetched. PDFs with self-contained character mappings can yield text, including CJK; documents requiring unavailable predefined CMaps can return empty or partial text without an error. Embedded ToUnicode alone does not guarantee extraction when the encoding still needs an external CMap. Missing standard-font data does not necessarily prevent extraction. A successful response does not certify complete text; truncated reports only the max_pages limit.",
  "family": "documents",
  "price_usd": "0.005",
  "free": false,
  "timeout_ms": 40000,
  "max_timeout_seconds": 61,
  "input_schema": {
    "$schema": "https://json-schema.org/draft/2020-12/schema",
    "type": "object",
    "properties": {
      "url": {
        "type": "string",
        "maxLength": 2048,
        "format": "uri",
        "description": "Absolute http(s) URL of the PDF, up to 2048 characters; redirects are followed and the body must start with the %PDF- signature."
      },
      "max_pages": {
        "default": 100,
        "description": "How many leading pages to convert, 1 to 1000; a shorter document is not an error, and a longer one is cut and reported as truncated.",
        "type": "integer",
        "minimum": 1,
        "maximum": 1000
      }
    },
    "required": [
      "url"
    ],
    "additionalProperties": false,
    "description": "Offline, best-effort text-layer extraction; no OCR. No external CMap or standard-font data is configured or fetched. PDFs with self-contained character mappings can yield text, including CJK; documents requiring unavailable predefined CMaps can return empty or partial text without an error. Embedded ToUnicode alone does not guarantee extraction when the encoding still needs an external CMap. Missing standard-font data does not necessarily prevent extraction. A successful response does not certify complete text; truncated reports only the max_pages limit. PDF text extraction accepts at most 1000000 raw and separately 1000000 output UTF-16 code units, 10000 items per page and 100000 items across selected pages. Exceeding a limit rejects the call with 413 too_large. truncated refers only to max_pages."
  },
  "input_example": {
    "url": "https://grist.tools/samples/documents/markdown-guide.pdf",
    "max_pages": 100
  },
  "output_example": {
    "source_url": "https://grist.tools/samples/documents/markdown-guide.pdf",
    "total_pages": 1,
    "extracted_pages": 1,
    "truncated": false,
    "chars": 249,
    "markdown": "# Field Notes\n\n## A short guide in three parts\n\nThis sample shows how headings are inferred.\n\nLarger text becomes a heading, body text stays plain.\n\n### Second section\n\nEach line of the page becomes its own block.\n\nThe document ends after this line."
  },
  "errors": [
    "invalid_input",
    "blocked_target",
    "unreachable_target",
    "upstream_timeout",
    "unsupported_content_type",
    "too_large",
    "unprocessable",
    "internal"
  ]
}