{"object":"list","data":[{"id":"GLM-5.2","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":1048576,"zdr_supported":true,"wafer":{"display_name":"GLM-5.2","description":"General Language Model 5.2 — high-quality bilingual (EN/ZH) generation with strong coding and reasoning capabilities.","tier":"serverless_only","context_length":1048576,"max_output_tokens":null,"capabilities":{"vision":false,"tools":true,"reasoning":true,"zdr":{"same_capabilities":true,"supported":true},"messages":{"tools":true,"vision":false,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":true,"regex":true,"tools":true,"grammar":true,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"native_effort","efforts":["none","high","max"],"default":"max","description":"GLM exposes a practical off/high/max control; low and medium do not create separate tiers on this backend."}},"pricing":{"currency":"usd","input_cents_per_million":126,"output_cents_per_million":396,"cache_read_cents_per_million":23}}},{"id":"GLM-5.3","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":1048576,"zdr_supported":true,"wafer":{"display_name":"GLM-5.3","description":"GLM-5.3 — Z.ai's flagship GLM-5.3 MoE (391B, NVFP4), self-hosted on the Wafer fleet with a 1M-token context window and frontier coding/agentic performance. Available serverless.","tier":"serverless_only","context_length":1048576,"max_output_tokens":null,"capabilities":{"vision":false,"tools":true,"reasoning":true,"zdr":{"same_capabilities":true,"supported":true},"messages":{"tools":true,"vision":false,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":true,"regex":false,"tools":true,"grammar":false,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"system_prompt","efforts":["none","low","high","max"],"default":"low","description":"GLM-5.3 Wafer system-prompt conditioning, all three tiers prompted; ladder verified on OR's own probe question + a hard puzzle at temp 0 (2026-09-01): low ~50-70 (trained 'Reasoning Effort: Low' suppressor label), high ~530-610 (label-free, 300-word budget), max ~940-1270 ('Maximum' label, two-method verification, 2000-word budget) reasoning tokens; worst-case total ~2.4k ≈ 60-80s, inside OR's ~145s probe budget. ⚠ 'Reasoning Effort: High' is a TRAINED SUPPRESSOR phrase (caps ~100 tok, overrides body) — never label the high prompt. Promptless tiers free-run 1.9-4.5k tok on proof questions — never ship a promptless tier. default_reasoning_effort=low ⇒ unparameterized traffic gets brief reasoning by design."}},"pricing":{"currency":"usd","input_cents_per_million":119,"output_cents_per_million":440,"cache_read_cents_per_million":26}}},{"id":"Kimi-K3","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":1048576,"zdr_supported":true,"supports_vision":true,"wafer":{"display_name":"Kimi-K3","description":"Kimi K3 sparse MoE model, self-hosted on the Wafer fleet. Available serverless with a 1M-token context window and strong coding/agentic performance.","tier":"serverless_only","context_length":1048576,"max_output_tokens":null,"capabilities":{"vision":true,"tools":true,"reasoning":true,"zdr":{"same_capabilities":true,"supported":true},"messages":{"tools":true,"vision":true,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":true,"regex":true,"tools":true,"grammar":true,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"native_effort","efforts":["none","high","max"],"default":"none","description":"Kimi-K3 reasoning is exposed as off/high/max; low and medium collapse to the same enabled tier."}},"pricing":{"currency":"usd","input_cents_per_million":300,"output_cents_per_million":1275,"cache_read_cents_per_million":30}}},{"id":"Kimi-K2.6","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":262144,"zdr_supported":false,"supports_vision":true,"wafer":{"display_name":"Kimi-K2.6","description":"Kimi K2.6 sparse MoE model with a 262K context window. Available serverless and not included in standard Wafer Pass. Non-ZDR only: requests with `Wafer-ZDR: required` are rejected.","tier":"serverless_only","context_length":262144,"max_output_tokens":null,"capabilities":{"vision":true,"tools":true,"reasoning":true,"zdr":{"disabled_reason":"self_hosted_backend_decommissioned","supported":false},"messages":{"tools":true,"vision":true,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":false,"regex":"partitioned","tools":true,"grammar":false,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"native_effort","efforts":["none","low","medium","high"],"default":"none","description":"Moonshot Kimi accepts none, low, medium, and high. Max is not a supported Kimi-K2.6 effort."}},"pricing":{"currency":"usd","input_cents_per_million":114,"output_cents_per_million":480,"cache_read_cents_per_million":19}}},{"id":"GLM-5.3-Flash","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":1048576,"zdr_supported":true,"supports_vision":true,"wafer":{"display_name":"GLM-5.3-Flash","description":"GLM-5.3-Flash — Z.ai's fast GLM-5.3 MoE variant, self-hosted on the Wafer fleet with a 1M-token context window and strong coding/agentic performance. Available serverless to gateway partners.","tier":"serverless_only","context_length":1048576,"max_output_tokens":null,"capabilities":{"vision":true,"tools":true,"reasoning":true,"zdr":{"same_capabilities":true,"supported":true},"messages":{"tools":true,"vision":true,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":true,"regex":false,"tools":true,"grammar":false,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"system_prompt","efforts":["none","low","high","max"],"default":"low","description":"GLM-5.3 Flash uses Wafer system-prompt conditioning for low, high, and max; medium is not a distinct tier."}},"pricing":{"currency":"usd","input_cents_per_million":10,"output_cents_per_million":35,"cache_read_cents_per_million":2}}},{"id":"Qwen3.5-397B-A17B","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":262144,"zdr_supported":false,"supports_vision":true},{"id":"DeepSeek-V4-Flash-0731-Fast","object":"model","created":1788537871,"owned_by":"wafer","max_model_len":1048576,"zdr_supported":true,"wafer":{"display_name":"DeepSeek-V4-Flash-0731-Fast","description":"The same model served for high TPS.","tier":"serverless_only","context_length":1048576,"max_output_tokens":null,"capabilities":{"vision":false,"tools":true,"reasoning":true,"zdr":{"same_capabilities":true,"supported":true},"messages":{"tools":true,"vision":false,"reasoning":true,"streaming":true,"supported":true,"tool_streaming":true},"responses":{"tools":true,"streaming":true,"supported":true,"text_format":["text","json_object","json_schema"],"raw_json_schema_text":true},"chat_completions":{"n":true,"regex":false,"tools":true,"grammar":false,"streaming":true,"supported":true,"json_object":true,"json_schema":true,"tool_streaming":true,"json_schema_refs":true,"tools_with_response_format":true},"reasoning_effort":{"control":"system_prompt","efforts":["none","low","high","max"],"default":"low","description":"DeepSeek-V4 Flash uses Wafer system-prompt conditioning for low, high, and max; medium is not a distinct tier."}},"pricing":{"currency":"usd","input_cents_per_million":10,"output_cents_per_million":25,"cache_read_cents_per_million":5}}}]}