{"name":"ZeroGPU","description":"ZeroGPU is the compute efficiency layer for AI inference. It runs repeatable, high-volume tasks - classification, extraction, PII redaction, moderation, summarization, routing - on specialized small and nano language models across an edge-powered network, faster and cheaper than centralized GPUs, through one OpenAI-compatible API.","url":"https://docs.zerogpu.ai/","version":"1.0.0","protocolVersion":"0.3","preferredTransport":"HTTP+JSON","supportedInterfaces":[{"url":"https://docs.zerogpu.ai/","protocolBinding":"HTTP+JSON","protocolVersion":"0.3"}],"provider":{"url":"https://docs.zerogpu.ai/","organization":"ZeroGPU"},"documentationUrl":"https://docs.zerogpu.ai/","capabilities":{"streaming":false,"pushNotifications":false},"defaultInputModes":["text/plain"],"defaultOutputModes":["text/plain"],"skills":[{"id":"zerogpu","name":"zerogpu","description":"Use when building AI applications that need fast, cost-efficient inference for high-volume tasks like text classification, data extraction, content moderation, summarization, and routing. Reach for ZeroGPU when you need to integrate specialized small/nano models via OpenAI-compatible APIs, process large batches of requests asynchronously, or optimize inference costs for production workloads.","tags":[],"url":"https://docs.zerogpu.ai/.well-known/agent-skills/zerogpu/skill.md"}]}