{"name":"Cogito","description":"Open-source LLM inference on AWS Trainium and NVIDIA Blackwell. 14× faster than typical providers.","url":"https://docs.cogito.decart.ai/","version":"1.0.0","protocolVersion":"0.3","preferredTransport":"HTTP+JSON","supportedInterfaces":[{"url":"https://docs.cogito.decart.ai/","protocolBinding":"HTTP+JSON","protocolVersion":"0.3"}],"provider":{"url":"https://docs.cogito.decart.ai/","organization":"Cogito"},"documentationUrl":"https://docs.cogito.decart.ai/","capabilities":{"streaming":false,"pushNotifications":false},"defaultInputModes":["text/plain"],"defaultOutputModes":["text/plain"],"skills":[{"id":"cogito","name":"cogito","description":"Use when building LLM applications that need fast inference on open-source models, integrating chat completions, function calling, streaming responses, or structured JSON outputs. Reach for this skill when optimizing for throughput, cost, or when you need OpenAI-compatible APIs with models like Kimi K2.6, GLM-5.2, GPT-OSS, or Qwen.","tags":[],"url":"https://docs.cogito.decart.ai/.well-known/agent-skills/cogito/skill.md"}]}