[
  {
    "id": "mistral-document-ai-docs-1",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Use the Document AI OCR processor to extract text and structured content from PDF documents and images.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-2",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Table formatting supports null, markdown, and html values through the table_format parameter.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-3",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Header and footer extraction uses the extract_header and extract_footer parameters.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-4",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Block extraction uses the include_blocks parameter. When enabled, each page contains a blocks array with paragraph-level bounding boxes, structural block labels, and extracted content in reading order.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-5",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Confidence scores are available for extracted content at page, block, or word granularity through the confidence_scores_granularity parameter.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-6",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Multilingual OCR performs strongly across more than 40 languages.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-7",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "bbox_annotation: gives you the annotation of the bboxes extracted by the OCR model (charts/ figures etc) based on user requirement and provided bbox/image annotation format.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-8",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "document_annotation: returns the annotation of the entire document based on the provided document annotation format.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-9",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "Extraction of key information like vendor details and amounts from invoices for automated accounting.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-10",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "Capture of receipt data, including merchant names and transaction amounts, for expense management.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-11",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "Extraction of key clauses and terms from contracts for easier review and management",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-12",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/document_qna",
    "excerpt": "The Document QnA capability combines OCR with large language model capabilities to enable natural language interaction with document content.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-13",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/document_qna",
    "excerpt": "Multi-document queries and comparisons",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-14",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "OCR with image: even from low-quality or handwritten sources.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-15",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "For PDFs, pass a publicly available URL, pass a Base64-encoded PDF, or upload a PDF file.",
    "fetchedAt": "2026-09-10T18:46:58.781Z"
  },
  {
    "id": "mistral-document-ai-docs-16",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Table formatting supports `null`, `markdown`, and `html` values through the `table_format` parameter.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-17",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Header and footer extraction uses the `extract_header` and `extract_footer` parameters. When you use them, the response includes header and footer content in the `header` and `footer` fields.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-18",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Block extraction uses the `include_blocks` parameter. When enabled, each page contains a `blocks` array with paragraph-level bounding boxes, structural block labels, and extracted content in reading order.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-19",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Confidence scores are available for extracted content at page, block, or word granularity through the `confidence_scores_granularity` parameter.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-20",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "`bbox_annotation`: gives you the annotation of the bboxes extracted by the OCR model (charts/ figures etc) based on user requirement and provided bbox/image annotation format.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-21",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "`document_annotation`: returns the annotation of the entire document based on the provided document annotation format.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-22",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/document_qna",
    "excerpt": "Building document Q&A applications",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-docs-23",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Document formats include: `image_url`: PNG, JPEG/JPG, AVIF, and other image formats. `document_url`: PDF, PPTX, DOCX, and other document formats.",
    "fetchedAt": "2026-09-10T18:47:51.196Z"
  },
  {
    "id": "mistral-document-ai-supp-1",
    "tier": "claimed-docs",
    "url": "https://mistral.ai/solutions/document-ai/",
    "excerpt": "Document AI solution page: positioned for \"Compliance-first organizations requiring secure on-premises deployment\" with \"Secure deployments\" — Mistral sells self-hosted/on-prem enterprise deployments of its models alongside La Plateforme, and hosts a public Trust Center at trust.mistral.ai.",
    "fetchedAt": "2026-09-10T19:16:37.000Z"
  },
  {
    "id": "mistral-document-ai-docs-24",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "When enabled, each page contains a blocks array with paragraph-level bounding boxes, structural block labels, and extracted content in reading order.",
    "fetchedAt": "2026-09-16T21:30:38.952Z"
  },
  {
    "id": "mistral-document-ai-docs-25",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/annotations",
    "excerpt": "bbox_annotation: gives you the annotation of the bboxes extracted by the OCR model (charts/ figures etc) based on user requirement and provided bbox/image annotation format. The user may ask to describe/caption the figure for instance.",
    "fetchedAt": "2026-09-16T21:30:38.952Z"
  },
  {
    "id": "mistral-document-ai-docs-26",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/document_qna",
    "excerpt": "This allows you to extract information and insights from documents by asking questions in natural language.",
    "fetchedAt": "2026-09-16T21:30:38.952Z"
  },
  {
    "id": "mistral-document-ai-docs-27",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/document_qna",
    "excerpt": "Analyzing research papers and technical documents\n*   Extracting information from business documents\n*   Processing legal documents and contracts\n*   Building document Q&A applications",
    "fetchedAt": "2026-09-16T21:30:38.952Z"
  },
  {
    "id": "mistral-document-ai-docs-28",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/resources/sdks",
    "excerpt": "We provide official SDK clients in both Python and Typescript",
    "fetchedAt": "2026-09-16T21:30:38.952Z"
  },
  {
    "id": "mistral-document-ai-docs-29",
    "tier": "claimed-docs",
    "url": "https://docs.mistral.ai/studio/document-processing/basic_ocr",
    "excerpt": "Header and footer extraction uses the extract_header and extract_footer parameters. When you use them, the response includes header and footer content in the header and footer fields.",
    "fetchedAt": "2026-09-16T21:36:37.264Z"
  },
  {
    "id": "mistral-document-ai-comm-1",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/43282905",
    "excerpt": "Dang. Super fast and significantly more accurate than google, Claude and others. Pricing: $1/1000 pages... this looks great at pdf to markdown.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-2",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/43282905",
    "excerpt": "It outperforms the competition significantly AND can extract embedded images from the text. I really like LLMs for OCR more and more.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-3",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/43282905",
    "excerpt": "From my testing... it decided that the entire page is an image and returned ![img-0.jpeg] with coordinates for the entire page. Our tool, doctly.ai is much slower and async, but much more accurate.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-4",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/43282905",
    "excerpt": "This worked indeed. Although I had to cut my document into smaller chunks. 900 pages at once ended with a timeout.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-5",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/43282905",
    "excerpt": "Just tested with a multilingual (bidi) English/Hebrew document. The Hebrew output had no correspondence to the text whatsoever... Their benchmark results are impressive, but I'm a little disappointed.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-6",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/48645152",
    "excerpt": "I used Abbyy Finereader for several years. I loved it... Modern VLMs put classic FineReader to shame for processing low-resolution/degraded/non-standard text. If you have an OCR problem, Mistral OCR 4 is probably great.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-7",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/48645152",
    "excerpt": "I was processing 55 year old paper files, most of them severely degraded, with its predecessor model. I was very impressed! I also tried Abbyy Finereader but it didn't even come close in my experience.",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-comm-8",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/48645152",
    "excerpt": "Yes, we have successfully used Mistral OCR for digitizing handwritten forms. You always have a low percentage that need human review, but overall Mistral has been highly accurate (their price is amazing, too).",
    "fetchedAt": "2026-09-10T18:51:50.202Z"
  },
  {
    "id": "mistral-document-ai-probe-1",
    "tier": "probe",
    "url": "https://docs.mistral.ai/llms.txt",
    "excerpt": "PROBE llms.txt: HTTP 200 at https://docs.mistral.ai/llms.txt # MistralAI\n\n## Docs\n\n[Agents & Conversations](https://docs.mistral.ai/docs/agents/agents_and_conversations.md): Agents",
    "fetchedAt": "2026-09-16T21:36:38.804Z"
  },
  {
    "id": "mistral-document-ai-probe-2",
    "tier": "probe",
    "url": "https://docs.mistral.ai/studio/document-processing/overview.md",
    "excerpt": "PROBE docs-md: HTTP 404 at https://docs.mistral.ai/studio/document-processing/overview.md",
    "fetchedAt": "2026-09-16T21:36:38.804Z"
  },
  {
    "id": "mistral-document-ai-probe-3",
    "tier": "probe",
    "url": "https://docs.mistral.ai/openapi.json",
    "excerpt": "PROBE openapi: all candidate paths 404 (https://docs.mistral.ai/openapi.json, https://docs.mistral.ai/swagger.json, https://docs.mistral.ai/api/openapi.json, https://docs.mistral.ai/.well-known/openapi.json)",
    "fetchedAt": "2026-09-16T21:36:38.804Z"
  }
]
