[{"chat":true,"chatConfig":{"attributes":["smart"],"contextWindowTokens":524288,"descriptionShort":"Fast 1M-context DeepSeek model with vision","intelligence":{"high":39,"low":31,"medium":36,"off":22},"reasoningConfig":{"defaultEnabled":true,"effortMap":{"high":"xhigh","low":"low","medium":"high"},"params":{"/v1/chat/completions":{"disable":{"chat_template_kwargs":{"thinking":false}},"enable":{"chat_template_kwargs":{"reasoning_effort":"$EFFORT","thinking":true}}},"/v1/responses":{"disable":{"chat_template_kwargs":{"thinking":false}},"enable":{"chat_template_kwargs":{"reasoning_effort":"$EFFORT","thinking":true}}}},"reasoningHistoryPolicy":"tool-call-only","supportsEffort":true,"supportsToggle":true}},"contextWindowTokens":1048576,"createdAt":1789084800,"description":"DeepSeek's efficient MoE model with a 1M-token context window, image input, Engram memory, built-in speculative decoding, and tool calling","details":"DeepSeek V4.1 Flash with 552 billion total parameters (6 of 384 routed experts plus one shared expert active per token; about 8B active for prefill and 16B for decode), an Engram n-gram memory, and a native 1M-token context window. Compressed-KV attention keeps very long contexts affordable, built-in DSpark speculative decoding accelerates generation, and it accepts image input.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"experimental":true,"image":"deepseek.png","modelName":"deepseek-v4-1-flash","multimodal":true,"name":"DeepSeek V4.1 Flash","nameShort":"DeepSeek V4.1 Flash","paid":true,"parameters":"552 billion (MoE)","pricing":{"cachedInputTokenPricePer1M":0.13,"inputTokenPricePer1M":0.65,"outputTokenPricePer1M":1.45,"requestPrice":0},"recommendedUse":"Ideal for long-document and image-grounded analysis, large-codebase understanding, and high-volume agentic workloads over very long inputs.","repo":"tinfoilsh/confidential-deepseek-v4-1-flash","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-13"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["smart"],"contextWindowTokens":524288,"descriptionShort":"Best for complex coding and agentic tasks","intelligence":{"high":45,"low":39,"medium":42},"reasoningConfig":{"defaultEnabled":true,"effortMap":{"high":"max","low":"low","medium":"high"},"params":{"/v1/chat/completions":{"enable":{"chat_template_kwargs":{"reasoning_effort":"$EFFORT"}}},"/v1/responses":{"enable":{"chat_template_kwargs":{"reasoning_effort":"$EFFORT"}}}},"reasoningHistoryPolicy":"all","supportsEffort":true,"supportsToggle":false}},"contextWindowTokens":1048576,"createdAt":1788391929,"description":"Z.ai's latest flagship for agentic coding and long-horizon tasks, with a 1M-token context window","details":"GLM-5.3 shares GLM-5.2's architecture and improves on it through post-training: a 50% gain over GLM-5.2 on Z.ai's internal code bench, open-weights state of the art on Terminal Bench 3.0 and Agents' Last Exam, and much stronger long-horizon tool use. Thinking is always on, with low/high/max reasoning effort selectable per request.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"zai.png","modelName":"glm-5-3","multimodal":false,"name":"GLM-5.3","nameShort":"GLM-5.3","paid":true,"parameters":"743 billion (39B active)","pricing":{"cachedInputTokenPricePer1M":0.45,"inputTokenPricePer1M":1.8,"outputTokenPricePer1M":5.75,"requestPrice":0},"recommendedUse":"Agentic engineering, repo-scale code generation and refactoring, long-running tool-use sessions, and any workload that needs sustained reasoning over very long context.","repo":"tinfoilsh/confidential-glm5-3-nvfp4","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-09"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["fast"],"contextWindowTokens":524288,"descriptionShort":"Fast model with vision","intelligence":{"high":42,"low":36,"medium":39},"reasoningConfig":{"defaultEnabled":true,"effortMap":{"high":"max","low":"low","medium":"high"},"params":{"/v1/chat/completions":{"enable":{"chat_template_kwargs":{"clear_thinking":false,"reasoning_effort":"$EFFORT"}}},"/v1/responses":{"enable":{"chat_template_kwargs":{"clear_thinking":false,"reasoning_effort":"$EFFORT"}}}},"reasoningHistoryPolicy":"all","supportsEffort":true,"supportsToggle":false}},"contextWindowTokens":1048576,"createdAt":1788307200,"deprecated":true,"deprecationDate":"2026-10-09","description":"Z.AI's fast multimodal MoE model with a 1M-token context window, image input, reasoning, and tool calling","details":"GLM-5.3 Flash is a 320B-parameter mixture-of-experts model with 18B active parameters per token, hybrid linear and sparse attention, and a native 1M-token context window. Supports text and image inputs, always-on reasoning with selectable effort, tool calling, and structured outputs, with built-in speculative decoding for fast generation.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"experimental":true,"image":"zai.png","modelName":"glm-5-3-flash","multimodal":true,"name":"GLM-5.3 Flash","nameShort":"GLM-5.3 Flash","paid":true,"parameters":"320 billion (18B active)","pricing":{"cachedInputTokenPricePer1M":0.1,"inputTokenPricePer1M":0.4,"outputTokenPricePer1M":1.25,"requestPrice":0},"recommendedUse":"Ideal for long-document and multimodal tasks, agentic tool use, and high-volume workloads where speed and cost matter.","repo":"tinfoilsh/confidential-glm5-3-flash","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-09"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["smart"],"contextWindowTokens":262144,"descriptionShort":"Most capable Kimi model","intelligence":{"on":45},"reasoningConfig":{"reasoningHistoryPolicy":"all"}},"contextWindowTokens":262144,"createdAt":1786147200,"description":"Moonshot's flagship multimodal MoE model with hybrid attention for long-context reasoning, coding, and agentic workflows","details":"Moonshot's Kimi K3 with 2.8 trillion total parameters (16 of 896 experts active per token), built on Kimi Delta Attention with native vision input, tool calling, and reasoning modes.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"moonshot.png","modelName":"kimi-k3","multimodal":true,"name":"Kimi K3","nameShort":"Kimi K3","paid":true,"parameters":"2.8 trillion (MoE)","pricing":{"cachedInputTokenPricePer1M":0.8,"inputTokenPricePer1M":4,"outputTokenPricePer1M":20,"requestPrice":0},"recommendedUse":"Ideal for complex reasoning, long-horizon coding, multimodal understanding, and agentic workflows with tool calling.","repo":"tinfoilsh/confidential-kimi-k3","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-02"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["fast"],"contextWindowTokens":131072,"descriptionShort":"Best for fast multilingual chat","intelligence":{"off":8}},"contextWindowTokens":131072,"createdAt":1721764788,"description":"High-performance multilingual language model optimized for dialogue","details":"A powerful multilingual language model optimized for dialogue use cases.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions"],"image":"llama.png","modelName":"llama3-3-70b","multimodal":false,"name":"Llama 3.3 70B","nameShort":"Llama 3.3","paid":true,"parameters":"70 billion","pricing":{"inputTokenPricePer1M":1.75,"outputTokenPricePer1M":2.75,"requestPrice":0},"recommendedUse":"Ideal for production workloads requiring fast response times and high throughput.","repo":"tinfoilsh/confidential-llama3-3-70b","safety":{"categoryBreakdown":{"childEndangerment":30,"massViolence":60,"selfHarm":10},"metadata":{"date":"2026-09-02"},"score":97},"supportedLanguages":"English, German, French, Italian, Portuguese, Hindi, Spanish, and Thai","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["fast"],"contextWindowTokens":262144,"descriptionShort":"Best for everyday tasks with images","intelligence":{"off":14,"on":15},"reasoningConfig":{"defaultEnabled":true,"params":{"/v1/chat/completions":{"disable":{"chat_template_kwargs":{"enable_thinking":false}},"enable":{"chat_template_kwargs":{"enable_thinking":true}}},"/v1/responses":{"disable":{"chat_template_kwargs":{"enable_thinking":false}},"enable":{"chat_template_kwargs":{"enable_thinking":true}}}},"reasoningHistoryPolicy":"tool-call-only","supportsToggle":true}},"contextWindowTokens":262144,"createdAt":1775088000,"description":"Google's most capable open model with image understanding and long-context support","details":"A 31B model from Google DeepMind with strong performance across reasoning, coding, and vision tasks. Supports text and image inputs with a 256K context window. Features a built-in thinking mode for step-by-step reasoning, function calling for agentic workflows, and support for 35+ languages.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"gemma.png","modelName":"gemma4-31b","multimodal":true,"name":"Gemma 4 31B","nameShort":"Gemma 4","paid":true,"parameters":"31 billion","pricing":{"inputTokenPricePer1M":0.4,"outputTokenPricePer1M":1,"requestPrice":0},"recommendedUse":"Ideal for multimodal applications requiring image understanding, long-context reasoning, coding tasks, and agentic workflows with tool calling.","repo":"tinfoilsh/confidential-gemma4-31b","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-02"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":true,"chatConfig":{"attributes":["fast"],"contextWindowTokens":131072,"descriptionShort":"Best for quick reasoning tasks","intelligence":{"high":12,"low":10,"medium":11},"reasoningConfig":{"params":{"/v1/chat/completions":{"enable":{"reasoning_effort":"$EFFORT"}},"/v1/responses":{"enable":{"reasoning":{"effort":"$EFFORT"}}}},"reasoningHistoryPolicy":"tool-call-only","supportsEffort":true}},"contextWindowTokens":131072,"createdAt":1722888693,"description":"Open-weight model designed for powerful reasoning, agentic tasks, and versatile developer use cases","details":"An open-weight model with 117B parameters and 5.1B active parameters. Features adjustable reasoning effort, full access to the model's chain of thought, fine-tuning support, and built-in agentic abilities including function calling, web browsing, and code execution.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"openai.png","modelName":"gpt-oss-120b","multimodal":false,"name":"GPT-OSS 120B","nameShort":"GPT-OSS","paid":true,"parameters":"117 billion (5.1B active)","pricing":{"inputTokenPricePer1M":0.15,"outputTokenPricePer1M":0.6,"requestPrice":0},"recommendedUse":"Ideal for production use cases requiring high reasoning capabilities, agentic operations, and specialized applications.","repo":"tinfoilsh/confidential-gpt-oss-120b","safety":{"categoryBreakdown":{"childEndangerment":0,"massViolence":0,"selfHarm":0},"metadata":{"date":"2026-09-02"},"score":100},"supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"chat","usageMetric":"tokens"},{"chat":false,"contextWindowTokens":131072,"createdAt":1737676800,"description":"Safety reasoning model for content classification and trust \u0026 safety applications","details":"A safety-focused reasoning model built on GPT-OSS with 117B parameters and 5.1B active parameters. Classifies text based on custom safety policies you provide, enabling content filtering, labeling, and moderation workflows. Bring your own policy, inspect the model's reasoning for debugging, and adjust reasoning effort (low, medium, high). Apache 2.0 licensed.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"openai.png","modelName":"gpt-oss-safeguard-120b","multimodal":false,"name":"GPT-OSS Safeguard 120B","nameShort":"GPT-OSS Safeguard","paid":true,"parameters":"117 billion (5.1B active)","pricing":{"inputTokenPricePer1M":0.15,"outputTokenPricePer1M":0.6,"requestPrice":0},"recommendedUse":"Ideal for safety use cases including content moderation, policy enforcement, LLM guardrails, and Trust \u0026 Safety labeling workflows.","repo":"tinfoilsh/confidential-gpt-oss-safeguard-120b","supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":true,"type":"safety","usageMetric":"tokens"},{"chat":false,"contextWindowTokens":8192,"createdAt":1756512000,"description":"PII span detection and redaction inside a secure enclave.","details":"OpenAI's Privacy Filter is a 1.5B parameter token-classification model (50M active), served without a system prompt. POST /redact on pii-filter.tinfoil.sh accepts text and returns the upstream model's detected spans and redacted text. It is a custom API, not Chat Completions or Responses. Each successful API inference costs $0.005 whether or not PII is detected. Web search also calls this endpoint when pii_check_options is enabled and applies its own removal policy to the returned spans; the filter fee is independent of the search outcome.","endpoint":"/redact","endpoints":["/redact"],"image":"openai.png","modelName":"pii-filter","multimodal":false,"name":"Privacy Filter","nameShort":"Privacy Filter","paid":true,"parameters":"1.5 billion (50M active)","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.005},"recommendedUse":"Detect and redact PII through the attested /redact endpoint, or enable pii_check_options on web search requests.","repo":"tinfoilsh/confidential-pii-cpu","supportedLanguages":"Primarily English; performance varies on other languages","toolCalling":false,"type":"safety","usageMetric":"requests"},{"contextWindowTokens":8192,"createdAt":1713509298,"description":"Open-source text embedding model that outperforms OpenAI models on key benchmarks","details":"A long-context text encoder that outperforms OpenAI's embedding models on key short- and long-context benchmarks. Specialized for generating text embeddings, semantic search, and RAG applications.","endpoint":"/v1/embeddings","endpoints":["/v1/embeddings"],"image":"nomic.png","modelName":"nomic-embed-text","multimodal":false,"name":"Nomic Embed Text","nameShort":"Nomic Embed","paid":true,"parameters":"137 million","pricing":{"inputTokenPricePer1M":0.05,"outputTokenPricePer1M":0,"requestPrice":0},"recommendedUse":"Ideal for retrieval-augmented generation (RAG), semantic search, clustering, and document similarity tasks.","repo":"tinfoilsh/confidential-nomic-embed-text","supportedLanguages":"Multilingual","toolCalling":false,"type":"embedding","usageMetric":"tokens"},{"createdAt":1716076551,"description":"Private document processing and conversion service.","details":"Document processing endpoint that converts PDFs, Word documents, and other formats into structured data. Supports document parsing, text extraction, and format conversion with high accuracy.","endpoint":"/v1/convert/file","endpoints":["/v1/convert/file"],"image":"doc-upload.png","modelName":"doc-upload","multimodal":true,"name":"Private Document Processing","nameShort":"Doc Upload","paid":true,"parameters":"Document processing service","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.05},"recommendedUse":"Ideal for document upload, processing, conversion, and text extraction workflows.","repo":"tinfoilsh/confidential-doc-upload","supportedLanguages":"Multilingual document processing","toolCalling":false,"type":"document","usageMetric":"requests"},{"chat":false,"contextWindowTokens":32768,"createdAt":1737590400,"description":"Audio-capable model built on Mistral Small 3.1 for transcription, translation, and spoken queries","details":"Combines speech processing with strong text capabilities from its Mistral Small 3.1 foundation. Transcribes audio up to 30 minutes with automatic language detection, understands spoken input up to 40 minutes, answers questions from audio, and can trigger functions directly from voice commands.","endpoint":"/v1/audio/transcriptions","endpoints":["/v1/audio/transcriptions","/v1/audio/translations","/v1/chat/completions"],"image":"mistral.png","modelName":"voxtral-small-24b","multimodal":false,"name":"Voxtral Small 24B","nameShort":"Voxtral","paid":true,"parameters":"24 billion","pricing":{"inputTokenPricePer1M":0.2,"outputTokenPricePer1M":0.6,"requestPrice":0},"recommendedUse":"Ideal for speech transcription, audio Q\u0026A, summarization, translation, and voice-triggered function calling.","repo":"tinfoilsh/confidential-voxtral-small-24b","supportedLanguages":"English, Spanish, French, Portuguese, Hindi, German, Dutch, Italian","toolCalling":false,"type":"audio","usageMetric":"tokens"},{"chat":false,"createdAt":1738454400,"description":"Web search tool to augment chat models.","details":"Web search tool with optional automatic PII blocking. Can be used alongside chat models to provide internet sources related to the conversation.","endpoint":"/v1/chat/completions","endpoints":["/v1/chat/completions","/v1/responses"],"image":"websearch.png","modelName":"websearch","multimodal":false,"name":"Private Web Search","nameShort":"Web Search","paid":true,"parameters":"N/A","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.05},"recommendedUse":"Use as a tool alongside chat models to provide web information.","repo":"tinfoilsh/confidential-websearch","supportedLanguages":"Multilingual with strong performance across major languages","toolCalling":false,"type":"tool","usageMetric":"requests"},{"createdAt":1716699056,"description":"High-performance speech recognition and transcription model","details":"Advanced speech recognition model with great accuracy, speed, and multilingual support. Optimized for real-time transcription with enhanced performance on diverse audio conditions, accents, and terminology.","endpoint":"/v1/audio/transcriptions","endpoints":["/v1/audio/transcriptions"],"image":"openai.png","modelName":"whisper-large-v3-turbo","multimodal":false,"name":"Whisper Large V3 Turbo","nameShort":"Whisper Turbo","paid":true,"parameters":"809 million","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.01},"recommendedUse":"Ideal for transcription services, captioning, voice interfaces, and multilingual audio processing applications.","repo":"tinfoilsh/confidential-whisper-large-v3","supportedLanguages":"Supports 90+ languages with high accuracy across diverse accents and dialects","toolCalling":false,"type":"audio","usageMetric":"requests"},{"chat":false,"createdAt":1740700800,"description":"Expressive text-to-speech model with instruction-controlled style and multilingual support","details":"A 1.7B-parameter text-to-speech model supporting 10 languages (Chinese, English, Japanese, Korean, German, French, Russian, Portuguese, Spanish, and Italian). Features 9 premium voice timbres with instruction-controlled emotion and prosody, and low-latency streaming generation.","endpoint":"/v1/audio/speech","endpoints":["/v1/audio/speech"],"image":"qwen.png","modelName":"qwen3-tts","multimodal":false,"name":"Qwen3-TTS 1.7B","nameShort":"Qwen3-TTS","paid":true,"parameters":"1.7 billion","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.01},"recommendedUse":"Ideal for speech synthesis, voice cloning, custom voice generation, and multilingual TTS applications.","repo":"tinfoilsh/confidential-qwen3-tts","supportedLanguages":"Chinese, English, Japanese, Korean, German, French, Russian, Portuguese, Spanish, Italian","toolCalling":false,"type":"tts","usageMetric":"requests"},{"chat":false,"createdAt":1781900473,"description":"Text-to-speech model from Mistral's Voxtral family.","details":"A 4B-parameter text-to-speech model that generates expressive, natural speech in nine languages. Supports preset voices and low-latency audio generation with 24 kHz output.","endpoint":"/v1/audio/speech","endpoints":["/v1/audio/speech"],"image":"mistral.png","modelName":"voxtral-tts","multimodal":false,"name":"Voxtral TTS","nameShort":"Voxtral TTS","paid":true,"parameters":"4 billion","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.002},"recommendedUse":"Ideal for multilingual speech synthesis, narration, and spoken responses for voice assistants and customer support applications.","repo":"tinfoilsh/confidential-realtime-models","supportedLanguages":"English, French, Spanish, German, Italian, Portuguese, Dutch, Arabic, Hindi","toolCalling":false,"type":"tts","usageMetric":"requests"},{"chat":false,"createdAt":1781900473,"description":"Streaming realtime speech-to-text model from Mistral's Voxtral family.","details":"A 4B-parameter streaming speech-to-text model with a causal audio encoder. Transcribes incoming audio in 13 languages through the realtime WebSocket endpoint for low-latency voice applications.","endpoint":"/v1/realtime","endpoints":["/v1/realtime"],"image":"mistral.png","modelName":"voxtral-mini-4b-realtime","multimodal":false,"name":"Voxtral Mini 4B Realtime","nameShort":"Voxtral Realtime","paid":true,"parameters":"4 billion","pricing":{"inputTokenPricePer1M":0,"outputTokenPricePer1M":0,"requestPrice":0.002},"recommendedUse":"Ideal for live transcription, subtitles, dictation, and speech input for voice assistants.","repo":"tinfoilsh/confidential-realtime-models","supportedLanguages":"English, Chinese, Hindi, Spanish, Arabic, French, Portuguese, Russian, German, Japanese, Korean, Italian, Dutch","toolCalling":false,"type":"audio","usageMetric":"requests"}]