{
  "name": "ocr-image-to-text",
  "summary": "Reads the text out of an image (scan, screenshot, photo) with Tesseract, returned with a mean confidence.",
  "rationale": "OCR needs a real engine and its language data AND a hardened fetch: an agent that pulls the image URL itself walks past every SSRF control, and carrying a WASM Tesseract plus a language model is not something an agent does per call.",
  "family": "documents",
  "price_usd": "0.005",
  "free": false,
  "timeout_ms": 45000,
  "max_timeout_seconds": 66,
  "input_schema": {
    "$schema": "https://json-schema.org/draft/2020-12/schema",
    "type": "object",
    "properties": {
      "url": {
        "type": "string",
        "maxLength": 2048,
        "format": "uri",
        "description": "Absolute http(s) URL of the image, up to 2048 characters and 10 MB; redirects are followed, the bytes must be PNG, JPEG, TIFF, WebP or GIF by signature, and at most 40 megapixels."
      },
      "lang": {
        "default": "eng",
        "description": "Tesseract language model to read the text with; only languages whose data ships with the service are accepted, today eng (the default).",
        "type": "string",
        "enum": [
          "eng"
        ]
      }
    },
    "required": [
      "url"
    ],
    "additionalProperties": false
  },
  "input_example": {
    "url": "https://grist.tools/samples/images/ocr-scan.png",
    "lang": "eng"
  },
  "output_example": {
    "source_url": "https://grist.tools/samples/images/ocr-scan.png",
    "lang": "eng",
    "image_format": "png",
    "text": "SAMPLE SCAN FOR OCR\nSECOND LINE 2026\n",
    "confidence": 86
  },
  "errors": [
    "invalid_input",
    "blocked_target",
    "unreachable_target",
    "upstream_timeout",
    "unsupported_content_type",
    "too_large",
    "unprocessable",
    "internal"
  ]
}