{"domain":"cerebras-inference.main-kill-isr.mintlify.me","count":1,"changes":[{"captured_at":"2026-07-20T18:48:33","card_hash":"3525f5597875f4d863b73eea5d7928c5cc3872d950a96638714bcc40169449d5","previous_card_hash":null,"diff":{"skills_added":[{"id":"metrics","name":"cerebras-metrics","description":"Set up Prometheus scraping and Grafana dashboards for Cerebras dedicated inference endpoints. Use when configuring observability for dedicated endpoints, building monitoring dashboards, or integrating with Prometheus/Grafana Cloud/Datadog.","tags":[],"inputModes":null,"outputModes":null},{"id":"models","name":"cerebras-models","description":"Discover models available on Cerebras Inference and migrate workloads between them. Use when listing available models, checking rate limits per tier, or converting an existing workload to a different model.","tags":[],"inputModes":null,"outputModes":null},{"id":"openai-compatibility","name":"cerebras-openai-compatibility","description":"Use the OpenAI SDK or OpenAI-compatible clients with Cerebras by swapping the base URL. Use when migrating from OpenAI, or using Cerebras with OpenAI-compatible tooling.","tags":[],"inputModes":null,"outputModes":null},{"id":"output-control","name":"cerebras-output-control","description":"Control Cerebras Chat Completions output using stop sequences, frequency/presence penalties, temperature, and sampling parameters. Use when filtering phrases, tuning determinism, or adjusting response creativity.","tags":[],"inputModes":null,"outputModes":null},{"id":"payload-optimization","name":"cerebras-payload-optimization","description":"Reduce TTFT on the Cerebras API by compressing request payloads with gzip or msgpack. Use when benchmarking compression strategies, measuring prompt size in tokens, or optimizing large chat payloads.","tags":[],"inputModes":null,"outputModes":null},{"id":"prompt-caching","name":"cerebras-prompt-caching","description":"Measure and optimize Cerebras automatic prompt caching. Use when benchmarking cache hit rate, understanding prompt_cache_key scoping, or analyzing TTFT reduction from warm caches.","tags":[],"inputModes":null,"outputModes":null},{"id":"rate-limits","name":"cerebras-rate-limits","description":"Use Cerebras rate limit response headers to maximize throughput and avoid 429 errors. Use when building clients that resume as soon as limits reset, or debugging unexpected rate limiting behavior.","tags":[],"inputModes":null,"outputModes":null},{"id":"reasoning","name":"cerebras-reasoning","description":"Configure and benchmark reasoning on Cerebras models (gpt-oss-120b, zai-glm-4.7). Use when testing reasoning formats, measuring performance across effort levels, or debugging multi-turn reasoning retention.","tags":[],"inputModes":null,"outputModes":null},{"id":"structured-outputs","name":"cerebras-structured-outputs","description":"Enforce JSON schema compliance on Cerebras model responses using strict mode. Use when debugging schema validation errors, checking strict=true compatibility, or migrating from JSON mode to structured outputs.","tags":[],"inputModes":null,"outputModes":null},{"id":"tool-use","name":"cerebras-tool-use","description":"Implement tool calling with Cerebras models, including parallel tool calls and end-to-end latency measurement. Use when building agentic workflows, benchmarking parallel vs. sequential tool calls, or debugging tool call schemas.","tags":[],"inputModes":null,"outputModes":null}],"skills_removed":[],"skills_changed":[],"fields_changed":[{"field":"name","before":null,"after":"Cerebras Inference"},{"field":"version","before":null,"after":"1.0.0"},{"field":"protocolVersion","before":null,"after":"0.3"},{"field":"url","before":null,"after":"https://inference-docs.cerebras.ai/"},{"field":"documentationUrl","before":null,"after":"https://inference-docs.cerebras.ai/"},{"field":"preferredTransport","before":null,"after":"HTTP+JSON"}],"other_changed":true,"is_empty":false,"human_summary":"added 10 skills · name ∅ → Cerebras Inference · version ∅ → 1.0.0 · protocolVersion ∅ → 0.3 · url ∅ → https://inference-docs.cerebras.ai/ · documentationUrl ∅ → https://inference-docs.cerebras.ai/ · preferredTransport ∅ → HTTP+JSON"}}]}