{"data":[{"id":"openai/gpt-oss-20b","hugging_face_id":"openai/gpt-oss-20b","name":"gpt-oss-20b","created":1783359731,"input_modalities":["text"],"output_modalities":["text"],"quantization":"fp4","context_length":131072,"max_output_length":16384,"pricing":{"prompt":"0","completion":"0","request":"0","image":"0","web_search":"0","internal_reasoning":"0","input_cache_read":"0","input_cache_write":"0"},"supported_sampling_parameters":["temperature","top_p","top_k","min_p","max_tokens","stop","seed","frequency_penalty","presence_penalty"],"supported_parameters":["temperature","top_p","top_k","min_p","max_tokens","stop","seed","frequency_penalty","presence_penalty","stream","stream_options","response_format","reasoning","reasoning_effort"],"supported_features":["reasoning"],"reasoning":{"parser":"gpt-oss","efforts":["minimal","low","medium","high"],"default_effort":"medium","effort_control":"best_effort"},"description":"gpt-oss-20b is an open-weight 21B parameter model released by OpenAI under the Apache 2.0 license. It uses a Mixture-of-Experts (MoE) architecture with 3.6B active parameters per forward pass, optimized for lower-latency inference and deployability on consumer or single-GPU hardware. The model is trained in OpenAI’s Harmony response format and supports reasoning level configuration, fine-tuning, and agentic capabilities including function calling, tool use, and structured outputs.","deprecation_date":null,"is_ready":false,"is_free":true,"discount_to_user":0,"capacity_tpm":30000,"datacenters":[{"country_code":"US"}],"openrouter":{"slug":"openai/gpt-oss-20b"}}]}