[
  {
    "id": "baseten-docs-1",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Use Model APIs to call supported language models without deploying them.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-2",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Deploy an open-source, fine-tuned, or custom model on dedicated GPUs.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-3",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Fine-tune with Loops or run your own training code with Training Jobs.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-4",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "To point a coding agent at Model APIs, see [Coding agents](/inference/model-apis/coding-agents).",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-5",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/agent-setup.md",
    "excerpt": "Install the Baseten skill and MCP servers so your coding agent can manage your Baseten workspace and search these docs.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-6",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/structured-outputs.md",
    "excerpt": "Structured outputs let you generate text that conforms to specific JSON schemas, providing reliable data extraction and controlled text generation.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-7",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/function-calling.md",
    "excerpt": "Function calling* (also called *tool calling*) lets a model choose a tool and produce its arguments from a user request.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-8",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/streaming.md",
    "excerpt": "Return model output token by token as it is generated.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-9",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/overview.md",
    "excerpt": "Async inference returns a request ID quickly and completes later through webhook or polling, which suits batch work, long documents",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-10",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "Model APIs bill by token and enforce request and token rate limits. You can also set a workspace budget and query usage by API key or model.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-11",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "Cached input tokens are prompt tokens served from the KV cache at a discounted rate. Caching is automatic and requires no request flags.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-gh-1",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "Write once, run anywhere: Package model code, weights, and dependencies with a model server that behaves the same in development and production.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-gh-2",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "You write a `config.yaml` that specifies the model, the hardware, and the engine, then `uvx truss push` builds a TensorRT-optimized container and deploys it. No Python code, no Dockerfile, no container management.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-12",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/deprecation.md",
    "excerpt": "Migrate to a dedicated deployment with the deprecated model weights. Contact us for assistance.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-13",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/training/index.md",
    "excerpt": "Baseten trains models on managed GPUs and deploys the resulting checkpoints to production inference on the same platform.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-14",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/concepts/howbasetenworks.md",
    "excerpt": "Baseten provisions GPUs through MCM, runs your training container, and syncs checkpoints to storage as the job progresses.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-15",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/concepts/howbasetenworks.md",
    "excerpt": "Deployments run active-active across clusters and clouds. If a region or provider loses capacity, MCM reroutes and reprovisions workloads.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-16",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/coding-agents.md",
    "excerpt": "Switch can also install Pi's direct Baseten provider, compare Baseten spend with estimated costs from Anthropic or OpenAI, and route requests back to those providers.",
    "fetchedAt": "2026-09-04T21:02:29.428Z"
  },
  {
    "id": "baseten-docs-17",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Deploy an open-source, fine-tuned, or custom model on dedicated infrastructure.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-18",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "a `config.yaml` can define the model, hardware, and inference engine without custom serving code",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-19",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Use a Python model class or a custom Docker server when you need custom preprocessing, postprocessing, dependencies, or server behavior.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-20",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "They also support stable environments for development, staging, and production.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-21",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "They support the OpenAI Chat Completions API and the Anthropic Messages API in beta, so you can use familiar client SDKs.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-22",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/coding-agents.md",
    "excerpt": "Connect Claude Code, Codex CLI, or Pi with Baseten Switch.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-23",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "To monitor token and request consumption by API key or model, see Usage.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-24",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "To raise a Basic account's limits, request email verification. You can also use that form to move to Pro or Enterprise.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-25",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/function-calling.md",
    "excerpt": "Function calling (also called tool calling) lets a model choose a tool and produce its arguments from a user request.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-26",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/overview.md",
    "excerpt": "Streaming sends tokens as they are generated over server-sent events, which suits long generations and UIs where partial output beats a blank wait.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-27",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/overview.md",
    "excerpt": "Async inference returns a request ID quickly and completes later through webhook or polling, which suits batch work, long documents, or any case where the caller should not hold a connection open for minutes.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-28",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/training/index.md",
    "excerpt": "Your Axolotl config, TRL script, or custom loop runs unchanged in a container. Baseten provisions the GPUs, syncs checkpoints as your job saves them, and deploys any checkpoint as a production endpoint.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-29",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/training/index.md",
    "excerpt": "Deploy any synced checkpoint with one CLI command",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-gh-3",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "Fast developer loop: Iterate with live reload, skip Docker and Kubernetes configuration, and use a batteries-included serving environment.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-gh-4",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "Support for all Python frameworks: From `transformers` and `diffusers` to PyTorch and TensorFlow to vLLM, SGLang, and TensorRT-LLM, Truss supports models created and served with any framework.",
    "fetchedAt": "2026-09-04T21:04:40.434Z"
  },
  {
    "id": "baseten-docs-30",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/overview",
    "excerpt": "Call hosted models through an OpenAI-compatible API, deploy your own models on dedicated infrastructure",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-docs-31",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "To monitor token and request consumption by API key or model, see [Usage](#usage).",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-docs-32",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/training/index.md",
    "excerpt": "Fine-tuning or running RL on a supported base model: Loops provisions a dedicated trainer and paired sampler, and each training step is an API call from a Python loop you write.",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-docs-33",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/structured-outputs.md",
    "excerpt": "Because Baseten exposes an OpenAI-compatible endpoint, you can use LangChain's `ChatOpenAI` with `with_structured_output` by pointing `base_url` at Baseten",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-docs-34",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/coding-agents.md",
    "excerpt": "Use [Baseten Switch](/reference/cli/baseten-switch) to route requests from Claude Code or Codex CLI to Model APIs.",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-gh-5",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "Truss lets you serve models with the Baseten Inference Stack as well as deploy models from any open-source framework: vLLM, SGLang, TensorRT-LLM, `transformers`, `diffusers`, PyTorch, TensorFlow, and more.",
    "fetchedAt": "2026-09-04T21:06:48.214Z"
  },
  {
    "id": "baseten-docs-35",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/overview.md",
    "excerpt": "Async inference returns a request ID quickly and completes later through webhook or polling",
    "fetchedAt": "2026-09-04T21:08:35.001Z"
  },
  {
    "id": "baseten-docs-36",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/training/index.md",
    "excerpt": "Loops provisions a dedicated trainer and paired sampler, and each training step is an API call from a Python loop you write.",
    "fetchedAt": "2026-09-04T21:08:35.001Z"
  },
  {
    "id": "baseten-docs-37",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "x-ratelimit-remaining-requests: Reports how many requests remain before you reach the request rate limit.",
    "fetchedAt": "2026-09-04T21:08:35.001Z"
  },
  {
    "id": "baseten-gh-6",
    "tier": "github",
    "url": "https://github.com/basetenlabs/truss",
    "excerpt": "Deploying a model to Baseten via Truss turns a Hugging Face model into a production-ready API endpoint. You write a `config.yaml` that specifies the model, the hardware, and the engine, then `uvx truss push` builds a TensorRT-optimized container and deploys it.",
    "fetchedAt": "2026-09-04T21:08:35.001Z"
  },
  {
    "id": "baseten-docs-38",
    "tier": "claimed-docs",
    "url": "https://docs.baseten.co/inference/model-apis/pricing-and-limits.md",
    "excerpt": "You can also set a workspace budget and query usage by API key or model.",
    "fetchedAt": "2026-09-04T21:08:35.001Z"
  },
  {
    "id": "baseten-comm-1",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/40813005",
    "excerpt": "User building on Baseten Chains: 'there are real challenges in traditional setups going from prompt in -> prediction out to building the actual backends that blend business logic and inference for multiple models, large inputs, etc.' after working with it for a couple of weeks.",
    "fetchedAt": "2026-09-04T21:09:58.882Z"
  },
  {
    "id": "baseten-comm-2",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/40813005",
    "excerpt": "User feedback on Baseten: 'The new docs are great. Awesome work there.'",
    "fetchedAt": "2026-09-04T21:09:58.882Z"
  },
  {
    "id": "baseten-comm-3",
    "tier": "community",
    "url": "https://hn.algolia.com/api/v1/items/44270214",
    "excerpt": "Developer notes that with Baseten.co embedding workloads, the client (not server) becomes the bottleneck due to Python's GIL, prompting them to build a Rust-based client that releases the GIL during requests to improve throughput when querying vector DBs.",
    "fetchedAt": "2026-09-04T21:09:58.882Z"
  },
  {
    "id": "baseten-probe-1",
    "tier": "probe",
    "url": "https://docs.baseten.co/llms.txt",
    "excerpt": "PROBE llms.txt: HTTP 200 at https://docs.baseten.co/llms.txt # Baseten\n\n- [Baseten overview](https://docs.baseten.co/overview.md): Run hosted models, deploy custom models, and train",
    "fetchedAt": "2026-09-04T21:10:22.527Z"
  },
  {
    "id": "baseten-probe-2",
    "tier": "probe",
    "url": "https://docs.baseten.co/overview.md",
    "excerpt": "PROBE docs-md: HTTP 200 at https://docs.baseten.co/overview.md > ## Documentation Index\n> Fetch the complete documentation index at: https://docs.baseten.co/llms.txt\n> Use this file t",
    "fetchedAt": "2026-09-04T21:10:22.527Z"
  },
  {
    "id": "baseten-probe-3",
    "tier": "probe",
    "url": "https://docs.baseten.co/openapi.json",
    "excerpt": "PROBE openapi: all candidate paths 404 (https://docs.baseten.co/openapi.json, https://docs.baseten.co/swagger.json, https://docs.baseten.co/api/openapi.json, https://docs.baseten.co/.well-known/openapi.json)",
    "fetchedAt": "2026-09-04T21:10:22.527Z"
  },
  {
    "id": "baseten-probe-4",
    "tier": "probe",
    "url": "https://docs.baseten.co/agent-setup",
    "excerpt": "official MCP server documented at https://docs.baseten.co/agent-setup",
    "fetchedAt": "2026-09-04T21:10:22.527Z"
  },
  {
    "id": "baseten-probe-rt-1",
    "tier": "probe",
    "url": "https://inference.baseten.co/v1/models",
    "excerpt": "PROBE models-endpoint (2026-09-04): GET https://inference.baseten.co/v1/models without a key returned HTTP 401 (No Authorization header provided) \u2014 the OpenAI-style models endpoint is live and speaks JSON, but enumerating the catalog requires an API key.",
    "fetchedAt": "2026-09-04T21:11:21.000Z"
  },
  {
    "id": "baseten-probe-rt-2",
    "tier": "probe",
    "url": "https://status.baseten.co",
    "excerpt": "PROBE status-page (2026-09-04): https://status.baseten.co returns HTTP 200 and renders a public service-status page (page body includes \"operational\").",
    "fetchedAt": "2026-09-04T21:11:21.000Z"
  },
  {
    "id": "baseten-probe-rt-3",
    "tier": "probe",
    "url": "https://docs.baseten.co/mcp",
    "excerpt": "PROBE mcp-endpoint (2026-09-04): POST initialize to https://docs.baseten.co/mcp answered HTTP 200 with a JSON-RPC/MCP response (event: message data: {\"result\":{\"protocolVersion\":\"2025-06-18\",\"capabilities\":{\"tools\":{\"listChanged\":true},\"resources\":{\"listChanged\":true}) \u2014 a live, publicly reachable MCP server.",
    "fetchedAt": "2026-09-04T21:11:21.000Z"
  }
]
