{"success":true,"models":[{"modelId":"deepseek-v4-flash-0731","modelType":"text-generation","modelVersion":"v1","taskDisplayName":"Text Generation","quantized":false,"quantization":null,"maxTokens":1048576,"modelDisplayName":"deepseek-v4-flash-0731","parameters":"284B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731/blob/main/LICENSE","use_cases":["Chat"],"description":"DeepSeek-V4-Flash-0731 is a version-pinned, open-weight mixture-of-experts model optimized for coding, reasoning, and agentic workloads. It supports a 1M-token context window, making it well suited for large codebases, long documents, extended conversations, and multi-step tasks. The model supports function calling and both fast non-reasoning and higher-effort reasoning modes, providing strong performance with efficient inference.","favicon_url":"https://models-favicon.zerogpu.ai/deepseek-ai/deepseek-color.png","model_doc_url":"https://docs.zerogpu.ai/docs/model-catalog","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.16,"output_per_1m_tokens":0.38,"sample_responses_body":null,"sample_output_response":null,"sample_chat_completions_body":{"model":"deepseek-v4-flash-0731","messages":[{"role":"system","content":"You are a support triage analyst. Be concise."},{"role":"user","content":"Classify this ticket by urgency (low/medium/high) and category, then summarize it in one line: 'Our checkout page has been throwing intermittent 502 errors since this morning's deploy. Roughly 1 in 5 customers can't complete payment, and abandoned carts are spiking. Rolling back didn't help."}],"max_tokens":800},"sample_output_chat_completions":{"id":"gen-1788988936-4yDVDgvhgNEnzAJs7S5B","model":"deepseek-v4-flash-0731","usage":{"cost":0.00003168,"is_byok":false,"cost_details":{"upstream_inference_cost":0.00003168,"upstream_inference_prompt_cost":0.000010692,"upstream_inference_completions_cost":0.000020988},"total_tokens":134,"prompt_tokens":81,"completion_tokens":53,"prompt_tokens_details":{"audio_tokens":0,"video_tokens":0,"cached_tokens":0,"cache_write_tokens":0,"cache_creation_tokens":0},"completion_tokens_details":{"audio_tokens":0,"image_tokens":0,"reasoning_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"Urgency: **High**  \nCategory: **Checkout / Payment Failure**  \n\nSummary: Intermittent 502 errors on checkout since this morning's deploy, affecting ~20% of customers with spiking abandoned carts; rollback didn’t resolve.","provider_specific_fields":{"refusal":null,"reasoning":null}},"finish_reason":"stop","provider_specific_fields":{"native_finish_reason":"stop"}}],"created":1788988936,"provider":"StreamLake"}},"displayPriority":21,"createdAt":"2026-09-11T17:14:26.1571+00:00"},{"modelId":"zlm-v1-moderation-edge","modelType":"moderation-ensemble","modelVersion":"v1","taskDisplayName":"Text Moderation","quantized":false,"quantization":null,"maxTokens":800,"modelDisplayName":"zlm-v1-moderation-edge","parameters":"86M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Text Moderation","Brand Safety","Ad Tech"],"description":"zlm-v1-moderation-edge is a specialized nano model fine-tuned for moderating chat conversations. Similar to OpenAI’s moderation models, it detects unsafe, harmful, and policy-sensitive content and returns structured safety signals that applications can use to block, flag, filter, or route messages. Its compact architecture is optimized for low-latency, high-volume inference, making it well suited as a lightweight safety layer for chat applications, agents, and AI pipelines.","favicon_url":"https://models-favicon.zerogpu.ai/zlm-v1-iab-classify-edge/logo_dark.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/moderations","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":[{"text":"I am so angry at this person that I want to hurt them. They are worthless and should be scared of what I might do next.","type":"text"}],"model":"zlm-v1-moderation-edge"},"sample_output_response":null,"sample_chat_completions_body":{"input":[{"text":"I am so angry at this person that I want to hurt them. They are worthless and should be scared of what I might do next.","type":"text"}],"model":"zlm-v1-moderation-edge"},"sample_output_chat_completions":null},"displayPriority":20,"createdAt":"2026-08-06T14:57:46+00:00"},{"modelId":"glm-5.2","modelType":"text-generation","modelVersion":"v1","taskDisplayName":"Text Generation","quantized":false,"quantization":null,"maxTokens":262144,"modelDisplayName":"glm-5.2","parameters":"753B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Coding","Chat","Function Calling","Reasoning","RAG","Summarization"],"description":"glm-5.2 is Zhipu AI’s flagship open-weight model for advanced reasoning, coding, research, and agentic applications. Its 262K context window allows it to work across large codebases, long documents, extended conversations, and multi-step tasks while retaining relevant context. It supports function calling, tool use, and adjustable reasoning effort, making it well suited for software engineering, deep research, document analysis, planning, and complex automated workflows that require strong reasoning over extended tasks.","favicon_url":"https://models-favicon.zerogpu.ai/z.ai/z.jpeg","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/glm-5.2","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":1.1,"output_per_1m_tokens":3.5,"sample_responses_body":null,"sample_output_response":null,"sample_chat_completions_body":{"model":"glm-5.2","messages":[{"role":"system","content":"You are a meticulous code reviewer. Be brief."},{"role":"user","content":"Find the bug in this debounce implementation and show the fix: function\n  debounce(fn, ms) { let t; return (...args) => { clearTimeout(t); t = setTimeout(() => fn(args), ms); }; }"}],"max_tokens":800},"sample_output_chat_completions":{"id":"id-1788988948678","model":"glm-5.2","usage":{"neurons":103.85455322265625,"total_tokens":306,"prompt_tokens":68,"completion_tokens":238,"prompt_tokens_details":{"cached_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"**Bug:** `fn` is called with the `args` array as a single argument instead of spreading the arguments.\n\n**Fix:**\n\n```javascript\nfunction debounce(fn, ms) {\n  let t;\n  return (...args) => {\n    clearTimeout(t);\n    t = setTimeout(() => fn(...args), ms);\n  };\n}\n```","reasoning_content":"The user wants to find a bug in a debounce implementation and provide a fix.\nCode provided:\n```javascript\nfunction debounce(fn, ms) {\n  let t;\n  return (...args) => {\n    clearTimeout(t);\n    t = setTimeout(() => fn(args), ms);\n  };\n}\n```\n\nThe bug is in the line `t = setTimeout(() => fn(args), ms);`.\nHere, `args` is an array of arguments. The function `fn` is called with a single argument, which is the array of arguments. It should be called with the arguments spread out: `fn(...args)`.\n\nFix:\n```javascript\nfunction debounce(fn, ms) {\n  let t;\n  return (...args) => {\n    clearTimeout(t);\n    t = setTimeout(() => fn(...args), ms);\n  };\n}\n```"},"logprobs":null,"finish_reason":"stop"}],"created":1788988948}},"displayPriority":19,"createdAt":"2026-07-23T03:14:46.602995+00:00"},{"modelId":"gpt-oss-120b","modelType":"text-generation","modelVersion":"1","taskDisplayName":"Text Generation","quantized":false,"quantization":null,"maxTokens":131072,"modelDisplayName":"gpt-oss-120b","parameters":"120B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://github.com/openai/gpt-oss/blob/main/LICENSE","use_cases":["Coding","Chat","Function Calling","Reasoning","RAG","Summarization","Translation"],"description":"gpt-oss-120b is OpenAI’s open-weight reasoning model built for advanced text generation, coding, research, and agentic workflows. It supports function calling, structured outputs, streaming, and multilingual use, making it a strong fit for applications that need deeper reasoning and reliable multi-step execution. On ZeroGPU, gpt-oss-120b runs across the hybrid inference cloud, using optimized cloud and edge capacity to deliver strong performance while balancing speed and cost for production workloads.","favicon_url":"https://models-favicon.zerogpu.ai/gpt-oss-120b/gpt-oss-120b.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/gpt-oss-120b","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.15,"output_per_1m_tokens":0.6,"sample_responses_body":{"input":"Name my WiFi network something tha will make my neighbors laugh. Give me your top 3 with a one-line reason each.","model":"gpt-oss-120b","max_tokens":800,"instructions":"You are a witty naming consultant. Be brief."},"sample_output_response":{"id":"id-1788988956049","text":null,"user":null,"model":"gpt-oss-120b","tools":[],"top_p":1,"usage":{"neurons":0,"input_tokens":0,"total_tokens":0,"output_tokens":0},"object":"response","output":[{"id":"rs_b64ffe5e1ab3c1af","type":"reasoning","status":null,"content":[{"text":"We need to produce top 3 witty WiFi network names that are funny to neighbors, accompanied by a one-line reason each. Must be brief. Should be creative. Provide 3 names plus reason. Likely in bullet points. Provide some humor. Let's produce.","type":"reasoning_text"}],"summary":[],"encrypted_content":null},{"id":"msg_b94e8eddb59fc638","role":"assistant","type":"message","phase":null,"status":"completed","content":[{"text":"**1. “Free Wi‑Fi – No, Really!”** – Because watching anyone scramble for the password never gets old.  \n\n**2. “Drop It Like It’s Hotspot”** – A cheeky nod to the beats that keep the whole block grooving.  \n\n**3. “Tell My Wife I’m Working”** – Guarantees a chuckle (and maybe a few sympathetic sighs) from anyone passing by.","type":"output_text","logprobs":null,"annotations":[]}]}],"prompt":null,"status":"completed","metadata":null,"reasoning":null,"background":false,"created_at":1788988956,"truncation":"disabled","temperature":1,"tool_choice":"none","instructions":"You are a witty naming consultant. Be brief.","service_tier":"auto","top_logprobs":null,"input_messages":null,"max_tool_calls":null,"output_messages":null,"presence_penalty":0,"frequency_penalty":0,"max_output_tokens":130969,"ec_transfer_params":null,"incomplete_details":null,"kv_transfer_params":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"gpt-oss-120b","messages":[{"role":"system","content":"You are a witty naming consultant. Be brief."},{"role":"user","content":"Name my WiFi network something that will make my neighbors laugh. Give me your top 3 with a one-line reason each."}],"max_tokens":800},"sample_output_chat_completions":{"id":"id-1788988954278","model":"gpt-oss-120b","usage":{"neurons":23.45906639099121,"total_tokens":399,"prompt_tokens":103,"completion_tokens":296,"prompt_tokens_details":{"cached_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","audio":null,"content":"**1. FBI Surveillance Van #42** – Neighbors will grin at the classic “big brother” prank that makes every scan feel like a covert operation.  \n\n**2. Pretty Fly for a Wi‑Fi** – A cheeky nod to the ’90s hit that turns a mundane network into a sing‑along joke.  \n\n**3. Drop It Like It’s Hotspot** – A playful mash‑up of a pop lyric and tech lingo that’s instantly meme‑worthy.  ","refusal":null,"reasoning":"We need to propose top 3 WiFi network names that will make neighbors laugh, with a one-line reason each. Be witty. Provide brief. Probably include humor, puns, references. Ensure it's appropriate. Provide top 3 list.\n\nWe should consider that WiFi names are visible to anyone scanning, so they should be funny but not offensive. Provide reasons.\n\nLet's craft: 1) \"FBI Surveillance Van #42\" - classic joke about suspicious. 2) \"Pretty Fly for a Wi-Fi\" - pun on song. 3) \"Drop It Like It's Hotspot\" - pun.\n\nAdd reasons: 1) Plays on conspiracy paranoia. 2) Cheeky song reference. 3) Combines phrase with tech.\n\nMake sure each one-liner reason is brief. Provide as bullet list.\n\nLet's produce.","annotations":null,"function_call":null,"reasoning_content":"We need to propose top 3 WiFi network names that will make neighbors laugh, with a one-line reason each. Be witty. Provide brief. Probably include humor, puns, references. Ensure it's appropriate. Provide top 3 list.\n\nWe should consider that WiFi names are visible to anyone scanning, so they should be funny but not offensive. Provide reasons.\n\nLet's craft: 1) \"FBI Surveillance Van #42\" - classic joke about suspicious. 2) \"Pretty Fly for a Wi-Fi\" - pun on song. 3) \"Drop It Like It's Hotspot\" - pun.\n\nAdd reasons: 1) Plays on conspiracy paranoia. 2) Cheeky song reference. 3) Combines phrase with tech.\n\nMake sure each one-liner reason is brief. Provide as bullet list.\n\nLet's produce."},"logprobs":null,"token_ids":null,"stop_reason":null,"finish_reason":"stop","routed_experts":null}],"created":1788988954}},"displayPriority":18,"createdAt":"2026-07-17T08:01:26.891155+00:00"},{"modelId":"qwen3-30b-a3b-fp8","modelType":"text-generation","modelVersion":"1","taskDisplayName":"Text Generation","quantized":true,"quantization":"fp8","maxTokens":32768,"modelDisplayName":"qwen3-30b-a3b-fp8","parameters":"30B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://huggingface.co/Qwen/Qwen3-30B-A3B-FP8/blob/main/LICENSE","use_cases":["Coding","Chat","Function Calling","Reasoning","RAG","Summarization","Translation"],"description":"qwen3-30b-a3b-fp8 is Qwen’s open-weight mixture-of-experts model designed for efficient reasoning, coding, multilingual text generation, and agentic workflows. It activates only 3B of its 30B parameters per token, delivering strong model quality with lower latency and inference cost. It supports function calling, streaming, and multilingual workloads, making it well suited for agents, code assistance, automation, structured data tasks, and high-volume production applications. On ZeroGPU, it runs across the hybrid inference cloud to optimize speed and cost for production workloads.","favicon_url":"https://models-favicon.zerogpu.ai/qwen3-30b-a3b-fp8/qwen3-30b-a3b-fp8.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/qwen3-30b-a3b-fp8","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.05,"output_per_1m_tokens":0.3,"sample_responses_body":null,"sample_output_response":null,"sample_chat_completions_body":{"model":"qwen3-30b-a3b-fp8","messages":[{"role":"system","content":"You are a pragmatic senior engineer. Be brief."},{"role":"user","content":"My API is getting rate-limited by a third party. Give me 3\n  battle-tested strategies to handle it, one line each."}],"max_tokens":800},"sample_output_chat_completions":{"id":"id-1788989033049","model":"qwen3-30b-a3b-fp8","usage":{"neurons":14.411247253417969,"total_tokens":517,"prompt_tokens":52,"completion_tokens":465,"prompt_tokens_details":{"cached_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","audio":null,"content":"\n\n1. Implement caching to reduce redundant requests.  \n2. Use a queue with rate-limiting logic to throttle requests.  \n3. Contact the third-party provider to negotiate a higher rate limit.","refusal":null,"reasoning":"\nOkay, the user's API is getting rate-limited by a third party, and they need three battle-tested strategies. Let me think about the common solutions for rate limiting.\n\nFirst, maybe they can implement caching. If the third party allows it, caching responses can reduce the number of requests. But I need to make sure that's a valid approach here. Some APIs might not allow caching, but it's a common strategy.\n\nSecond, using a queue system. If the API calls are asynchronous, queuing them up and processing them at a controlled rate could help. That way, you don't hit the rate limit all at once. But how to implement that? Maybe with a message queue like RabbitMQ or Kafka, or even a simple in-memory queue.\n\nThird, maybe they can contact the third-party provider to increase the rate limit. If the current limit is too low, negotiating a higher limit might be necessary. But that's more of a direct approach. However, sometimes it's not possible, so maybe the user needs alternatives.\n\nWait, the user asked for three strategies. Let me check if these are all valid. Caching, queuing, and contacting the provider. But maybe there's another approach. Oh, like retrying with exponential backoff. But the user wants three, so maybe that's the third. But the user said \"battle-tested,\" so which ones are more proven?\n\nAlternatively, maybe using a proxy or a load balancer to distribute the requests. But that might be more complex. Or maybe using a circuit breaker pattern to handle failures gracefully. But the user is specifically about rate limiting, so the main strategies would be caching, queuing, and contacting the provider. Let me confirm each of these is a standard approach.\n\nYes, those three are commonly recommended. So the answer would be:\n\n1. Implement caching to reduce redundant requests.\n2. Use a queue with rate-limiting logic to throttle requests.\n3. Contact the third-party provider to request a rate limit increase.\n\nThat should cover the main strategies. Each is a single line, as requested.\n","annotations":null,"function_call":null,"reasoning_content":"\nOkay, the user's API is getting rate-limited by a third party, and they need three battle-tested strategies. Let me think about the common solutions for rate limiting.\n\nFirst, maybe they can implement caching. If the third party allows it, caching responses can reduce the number of requests. But I need to make sure that's a valid approach here. Some APIs might not allow caching, but it's a common strategy.\n\nSecond, using a queue system. If the API calls are asynchronous, queuing them up and processing them at a controlled rate could help. That way, you don't hit the rate limit all at once. But how to implement that? Maybe with a message queue like RabbitMQ or Kafka, or even a simple in-memory queue.\n\nThird, maybe they can contact the third-party provider to increase the rate limit. If the current limit is too low, negotiating a higher limit might be necessary. But that's more of a direct approach. However, sometimes it's not possible, so maybe the user needs alternatives.\n\nWait, the user asked for three strategies. Let me check if these are all valid. Caching, queuing, and contacting the provider. But maybe there's another approach. Oh, like retrying with exponential backoff. But the user wants three, so maybe that's the third. But the user said \"battle-tested,\" so which ones are more proven?\n\nAlternatively, maybe using a proxy or a load balancer to distribute the requests. But that might be more complex. Or maybe using a circuit breaker pattern to handle failures gracefully. But the user is specifically about rate limiting, so the main strategies would be caching, queuing, and contacting the provider. Let me confirm each of these is a standard approach.\n\nYes, those three are commonly recommended. So the answer would be:\n\n1. Implement caching to reduce redundant requests.\n2. Use a queue with rate-limiting logic to throttle requests.\n3. Contact the third-party provider to request a rate limit increase.\n\nThat should cover the main strategies. Each is a single line, as requested.\n"},"logprobs":null,"token_ids":null,"stop_reason":null,"finish_reason":"stop","routed_experts":null}],"created":1788989033}},"displayPriority":17,"createdAt":"2026-07-17T21:55:48.407315+00:00"},{"modelId":"llama-3.1-8b-instruct-fast","modelType":"text-generation","modelVersion":"1","taskDisplayName":"Summarization","quantized":false,"quantization":"Q4_K_L","maxTokens":131072,"modelDisplayName":"llama-3.1-8b-instruct-fast","parameters":"8B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Summarization","Chat","RAG","Translation","Intent Detection"],"description":"llama-3.1-8b-instruct-fast is Meta’s Llama 3.1 8B Instruct model optimized by ZeroGPU for fast, cost-efficient summarization and text processing at scale. Its 128K context window makes it well suited for long documents, transcripts, articles, email threads, and conversations that need to be processed in a single pass. The model supports multilingual workloads and is a strong fit for summarization pipelines, content processing, and agent workflows where low latency and predictable inference cost matter. On ZeroGPU, it runs across the hybrid inference cloud to optimize speed and cost for high-volume production workloads.","favicon_url":"https://models-favicon.zerogpu.ai/llama-3.2-3b-instruct/meta.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/llama-3-1-8b-instruct-fast","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":null,"sample_output_response":null,"sample_chat_completions_body":{"model":"llama-3.1-8b-instruct-fast","messages":[{"role":"user","content":"The global semiconductor industry is undergoing one of its most significant structural shifts in decades, driven by a combination of geopolitical tensions, surging demand from artificial intelligence workloads, and a wave of government-backed industrial policy across the United States, Europe, and Asia. At the center of this transformation is a race to onshore chip manufacturing capacity that had, for thirty years, been quietly concentrated in Taiwan and South Korea. The United States CHIPS and Science Act, signed into law in 2022, allocated over $52 billion in subsidies to incentivize domestic semiconductor fabrication. Since then, companies including TSMC, Intel, Samsung, and Micron have announced or broken ground on new fabs in Arizona, Ohio, Texas, and Idaho. However, construction timelines have slipped, costs have ballooned, and a shortage of skilled workers has prompted some manufacturers to bring in engineers from overseas — a move that has drawn political scrutiny even as it addresses a genuine talent gap. Meanwhile, the explosion of AI model training and inference has fundamentally altered the demand profile for chips. Graphics processing units originally designed for gaming, particularly those made by NVIDIA, have become the primary compute substrate for large language models. NVIDIA's H100 and successor Blackwell-series GPUs now trade at significant premiums on secondary markets, with some hyperscalers reporting lead times of over a year for large cluster orders. This bottleneck has accelerated investment in custom silicon: Google's Tensor Processing Units, Amazon's Trainium and Inferentia chips, and Meta's MTIA accelerators are all attempts to reduce dependence on third-party GPU supply chains. On the geopolitical front, the United States has progressively tightened export controls on advanced chips and chip-making equipment destined for China, with the most recent rules targeting gate-all-around transistor technology and high-bandwidth memory. China, in response, has dramatically increased state investment in its domestic semiconductor ecosystem, with companies like SMIC and Huawei making incremental but closely watched progress at advanced nodes. Analysts remain divided on how quickly China can close the gap, with estimates ranging from five to fifteen years depending on assumptions about technology transfer, equipment access, and the pace of indigenous innovation. The equipment layer — often overlooked in public discourse — may prove to be the most consequential chokepoint. ASML, the Dutch company that holds a near-monopoly on extreme ultraviolet lithography machines, has become a focal point of both trade diplomacy and industrial strategy. Without EUV tools, fabricating chips below 7 nanometers at scale is effectively impossible, giving ASML and its supply chain extraordinary leverage in a world where nations are suddenly treating semiconductor access as a matter of national security."}]},"sample_output_chat_completions":{"id":"id-1788989029808","model":"llama-3.1-8b-instruct-fast","usage":{"neurons":11.240795135498047,"total_tokens":818,"prompt_tokens":562,"completion_tokens":256,"prompt_tokens_details":{"cached_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","audio":null,"content":"The global semiconductor industry is undergoing a significant transformation due to a combination of factors, including geopolitical tensions, rising demand for artificial intelligence workloads, and government-backed industrial policies in the US, Europe, and Asia. This transformation is centered around the onshoring of chip manufacturing capacity, which had previously been concentrated in Taiwan and South Korea.\n\nThe US CHIPS and Science Act, signed into law in 2022, has allocated over $52 billion in subsidies to incentivize domestic semiconductor fabrication. As a result, companies such as TSMC, Intel, Samsung, and Micron have announced or broken ground on new fabs in various US states, including Arizona, Ohio, Texas, and Idaho. However, construction timelines have slipped, costs have increased, and a shortage of skilled workers has prompted some manufacturers to bring in engineers from overseas.\n\nThe explosion of AI model training and inference has fundamentally altered the demand profile for chips. Graphics processing units (GPUs) originally designed for gaming have become the primary compute substrate for large language models. NVIDIA's H100 and successor Blackwell-series GPUs are now trading at significant premiums on secondary markets, with some hyperscalers reporting lead times of over a year for large cluster orders.\n\nThis bottleneck has accelerated investment in custom silicon, with companies such as","refusal":null,"reasoning":null,"annotations":null,"function_call":null},"logprobs":null,"token_ids":null,"stop_reason":null,"finish_reason":"length","routed_experts":null}],"created":1788989029}},"displayPriority":16,"createdAt":"2026-05-26T16:34:16.809145+00:00"},{"modelId":"deberta-v3-small","modelType":"nli-deberta-v3","modelVersion":"1","taskDisplayName":"Text Classification","quantized":true,"quantization":"onnx-int-q8","maxTokens":400,"modelDisplayName":"deberta-v3-small","parameters":"142M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://github.com/microsoft/DeBERTa/blob/master/SECURITY.md","use_cases":["Email Filtering","Intent Detection"],"description":"deberta-v3-small is Microsoft’s lightweight text classification model optimized for fast, low-cost zero-shot classification. Provide text together with your own candidate labels, and the model returns confidence scores for each category without requiring a fixed taxonomy or task-specific training. It is well suited for intent detection, content tagging, routing, filtering, prioritization, and other high-volume classification workflows where speed and cost matter. Its compact architecture makes it a strong alternative to using a general-purpose LLM for straightforward classification tasks.","favicon_url":"https://models-favicon.zerogpu.ai/nli-deberta-v3-small/icons8-microsoft-96.png","model_doc_url":"https://github.com/microsoft/DeBERTa/tree/master/docs","privacy_service":"https://github.com/microsoft/DeBERTa/blob/master/LICENSE","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"Apple is expected to unveil its next-generation M5 chip at WWDC this June, promising a 40% boost in GPU performance and a new dedicated AI core for on-device machine learning tasks.","model":"deberta-v3-small","instructions":"[sports, finance, politics]"},"sample_output_response":{"id":"00116326-faa6-4b1b-87e7-1997d720660a","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"deberta-v3-small","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":46,"total_tokens":69,"output_tokens":23,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"00116326-faa6-4b1b-87e7-1997d720660a","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"politics\":0.5794752172968877,\"finance\":0.2777695568376636,\"sports\":0.14275522586544867}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788988934,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788988934,"instructions":"[sports, finance, politics]","max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"deberta-v3-small","messages":[{"role":"system","content":"[Sports,Politics,Technology,Finance,Entertainment]"},{"role":"user","content":"Apple is expected to unveil its next-generation M5 chip at WWDC this June, promising a 40% boost in GPU performance and a new dedicated AI core for on-device machine learning tasks."}]},"sample_output_chat_completions":{"id":"348f5b95-ea82-4aec-a98a-1b7d6594ca4b","model":"deberta-v3-small","usage":{"total_tokens":86,"prompt_tokens":46,"completion_tokens":40,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"Technology\":0.6170251412569493,\"Politics\":0.13927849027811745,\"Entertainment\":0.1242877091571418,\"Sports\":0.07005207600975918,\"Finance\":0.04935658329803223}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788988932,"service_tier":"default"}},"displayPriority":15,"createdAt":"2026-03-09T12:09:06.022639+00:00"},{"modelId":"zlm-v2-iab-classify-edge-enriched","modelType":"iab-enrichment-classify","modelVersion":"v2","taskDisplayName":"Text Classification","quantized":false,"quantization":null,"maxTokens":800,"modelDisplayName":"zlm-v2-iab-classify-edge-enriched","parameters":"170M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Ad Tech"],"description":"zlm-v2-iab-classify-edge-enriched is a multilingual content classification model built for contextual intelligence, advertising, and agent workflows. It analyzes text across 50+ languages and returns IAB content categories across the 1.0 and 2.2 taxonomies, along with topics, keywords, audience and interest segments, user intent, and confidence scores. It is optimized for low-latency edge inference and helps ad systems, agents, publishers, and recommendation platforms power contextual targeting, content categorization, brand-safety workflows, and real-time signal enrichment.","favicon_url":"https://models-favicon.zerogpu.ai/zlm-v1-iab-classify-edge-enriched/logo_dark.png","model_doc_url":"https://docs.zerogpu.ai/","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"Technology has quietly reshaped the rhythm of everyday life, weaving itself into routines so seamlessly that it often goes unnoticed. From the moment a smartphone alarm wakes someone in the morning to the final glance at a glowing screen before sleep, digital systems guide communication, navigation, work, and entertainment. This transformation did not happen overnight. It emerged through decades of incremental innovation, each new tool building upon the last, until the modern world became deeply interconnected.One of the most significant changes has been the speed at which information travels.","model":"zlm-v2-iab-classify-edge-enriched"},"sample_output_response":{"id":"5afbbe2f-ef39-4833-8024-f8afb1729c99","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"zlm-v2-iab-classify-edge-enriched","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":150,"total_tokens":884,"output_tokens":734,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"5afbbe2f-ef39-4833-8024-f8afb1729c99","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"audience\":[{\"id\":687,\"parent_id\":206,\"name\":\"Technology & Computing\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Technology & Computing\",\"score\":0.7817539670915485},{\"id\":703,\"parent_id\":687,\"name\":\"Consumer Electronics\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Technology & Computing\",\"tier3_name\":\"Consumer Electronics\",\"score\":0.6822751829244846},{\"id\":690,\"parent_id\":687,\"name\":\"Computing\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Technology & Computing\",\"tier3_name\":\"Computing\",\"score\":0.5447119734202016}],\"content\":{\"iab_1_0\":[{\"code\":\"IAB19\",\"name\":\"Technology & Computing\",\"tier\":1,\"parent_code\":null,\"score\":0.8513270663261827},{\"code\":\"IAB19-6\",\"name\":\"Cell Phones\",\"tier\":2,\"parent_code\":\"IAB19\",\"score\":0.6809580262884584},{\"code\":\"IAB9-30\",\"name\":\"Video & Computer Games\",\"tier\":2,\"parent_code\":\"IAB9\",\"score\":0.5204820652021064}],\"iab_2_2\":[{\"id\":596,\"parent_id\":0,\"name\":\"Technology & Computing\",\"tier1_name\":\"Technology & Computing\",\"score\":0.8513270663261827},{\"id\":635,\"parent_id\":632,\"name\":\"Smartphones\",\"tier1_name\":\"Technology & Computing\",\"tier2_name\":\"Consumer Electronics\",\"tier3_name\":\"Smartphones\",\"score\":0.6809580262884584},{\"id\":637,\"parent_id\":632,\"name\":\"Wearable Technology\",\"tier1_name\":\"Technology & Computing\",\"tier2_name\":\"Consumer Electronics\",\"tier3_name\":\"Wearable Technology\",\"score\":0.6415416535034761},{\"id\":632,\"parent_id\":596,\"name\":\"Consumer Electronics\",\"tier1_name\":\"Technology & Computing\",\"tier2_name\":\"Consumer Electronics\",\"score\":0.595014844803955},{\"id\":597,\"parent_id\":596,\"name\":\"Artificial Intelligence\",\"tier1_name\":\"Technology & Computing\",\"tier2_name\":\"Artificial Intelligence\",\"score\":0.5541231139744616},{\"id\":683,\"parent_id\":680,\"name\":\"Mobile Games\",\"tier1_name\":\"Video Gaming\",\"tier2_name\":\"Mobile Games\",\"score\":0.5204820652021064}]},\"topics\":[{\"name\":\"technology & computing\",\"score\":0.8513270663261827},{\"name\":\"smartphones\",\"score\":0.6809580262884584},{\"name\":\"wearable technology\",\"score\":0.6415416535034761},{\"name\":\"consumer electronics\",\"score\":0.595014844803955},{\"name\":\"artificial intelligence\",\"score\":0.5541231139744616},{\"name\":\"mobile games\",\"score\":0.5204820652021064}],\"keywords\":[\"technology\",\"everyday life\",\"routines\",\"smartphone alarm\",\"glowing screen\",\"navigation\",\"work\",\"entertainment\"],\"user_intent\":{\"name\":\"Identify key topics and concepts in the text\",\"category\":\"informational\",\"score\":0.7307}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788989058,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788989058,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"zlm-v2-iab-classify-edge-enriched","messages":[{"role":"user","content":"What are the basics of exercise and physical fitness?Exercise is anything that gets your body moving. Regular exercise is one of the best things you can do for your health. It has many benefits, including improving your overall health and fitness, and reducing your risk for many chronic (long-term) diseases.Every physical fitness routine is built on a few simple ideas. These include:Make exercise a habit, as your body adapts to the type of activity you do most often. Regular practice will help you improve.Build up your activity level slowly to help you continue to get stronger, faster, or more flexible without pushing too hard all at once.Challenge yourself by lifting slightly heavier weights, adding a few more minutes to your walk, or increasing your pace."}]},"sample_output_chat_completions":{"id":"60d3fa34-cd2c-4354-ac2a-86c312fca6d8","model":"zlm-v2-iab-classify-edge-enriched","usage":{"total_tokens":983,"prompt_tokens":192,"completion_tokens":791,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"audience\":[{\"id\":406,\"parent_id\":206,\"name\":\"Healthy Living\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Healthy Living\",\"score\":0.8267880846346385},{\"id\":415,\"parent_id\":406,\"name\":\"Wellness\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Healthy Living\",\"tier3_name\":\"Wellness\",\"score\":0.7113321459055835},{\"id\":408,\"parent_id\":406,\"name\":\"Fitness and Exercise\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Healthy Living\",\"tier3_name\":\"Fitness and Exercise\",\"score\":0.6374571135852},{\"id\":413,\"parent_id\":406,\"name\":\"Senior Health\",\"tier1_name\":\"Interest\",\"tier2_name\":\"Healthy Living\",\"tier3_name\":\"Senior Health\",\"score\":0.5551626996774902}],\"content\":{\"iab_1_0\":[{\"code\":\"IAB7-1\",\"name\":\"Exercise\",\"tier\":2,\"parent_code\":\"IAB7\",\"score\":0.9646817625854934},{\"code\":\"IAB7\",\"name\":\"Health & Fitness\",\"tier\":1,\"parent_code\":null,\"score\":0.9354958077737687},{\"code\":\"IAB7-44\",\"name\":\"Weight Loss\",\"tier\":2,\"parent_code\":\"IAB7\",\"score\":0.6993215402057686},{\"code\":\"IAB9-30\",\"name\":\"Video & Computer Games\",\"tier\":2,\"parent_code\":\"IAB9\",\"score\":0.648178913437892},{\"code\":\"IAB7-36\",\"name\":\"Physical Therapy\",\"tier\":2,\"parent_code\":\"IAB7\",\"score\":0.6470848317721539}],\"iab_2_2\":[{\"id\":225,\"parent_id\":223,\"name\":\"Fitness and Exercise\",\"tier1_name\":\"Healthy Living\",\"tier2_name\":\"Fitness and Exercise\",\"score\":0.9646817625854934},{\"id\":223,\"parent_id\":0,\"name\":\"Healthy Living\",\"tier1_name\":\"Healthy Living\",\"score\":0.9354958077737687},{\"id\":232,\"parent_id\":223,\"name\":\"Wellness\",\"tier1_name\":\"Healthy Living\",\"tier2_name\":\"Wellness\",\"score\":0.8064933325286461},{\"id\":231,\"parent_id\":223,\"name\":\"Weight Loss\",\"tier1_name\":\"Healthy Living\",\"tier2_name\":\"Weight Loss\",\"score\":0.6993215402057686},{\"id\":695,\"parent_id\":685,\"name\":\"Exercise and Fitness Video Games\",\"tier1_name\":\"Video Gaming\",\"tier2_name\":\"Video Game Genres\",\"tier3_name\":\"Exercise and Fitness Video Games\",\"score\":0.648178913437892},{\"id\":236,\"parent_id\":232,\"name\":\"Physical Therapy\",\"tier1_name\":\"Healthy Living\",\"tier2_name\":\"Wellness\",\"tier3_name\":\"Physical Therapy\",\"score\":0.6470848317721539}]},\"topics\":[{\"name\":\"fitness and exercise\",\"score\":0.9646817625854934},{\"name\":\"healthy living\",\"score\":0.9354958077737687},{\"name\":\"wellness\",\"score\":0.8064933325286461},{\"name\":\"weight loss\",\"score\":0.6993215402057686},{\"name\":\"exercise and fitness video games\",\"score\":0.648178913437892},{\"name\":\"physical therapy\",\"score\":0.6470848317721539}],\"keywords\":[\"exercise\",\"physical fitness\",\"health benefits\",\"long-term risks\",\"exercise habit\",\"exercise routine\"],\"user_intent\":{\"name\":\"Identify basic fitness and fitness ideas\",\"category\":\"transactional\",\"score\":0.5131}}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788989057,"service_tier":"default"}},"displayPriority":15,"createdAt":"2026-07-16T04:12:25.299529+00:00"},{"modelId":"gliner2-base-v1","modelType":"gliner2","modelVersion":"2","taskDisplayName":"PII","quantized":false,"quantization":null,"maxTokens":800,"modelDisplayName":"gliner2-base-v1","parameters":"205M","modelUsecases":{"ner":{"url":null,"description":"Named entity recognition with descriptions and spans","usecase_display_name":"Entity Extraction","sample_responses_body":{"input":"The application is built with Python 3.11 and uses PostgreSQL 15 for storage. It runs on Kubernetes with Docker containers and communicates via gRPC.","model":"gliner2-base-v1","metadata":{"labels":["programming language","database","technology","protocol"],"usecase":"ner","threshold":0.3}},"sample_chat_completions_body":{"model":"gliner2-base-v1","messages":[{"role":"user","content":"The application is built with Python 3.11 and uses PostgreSQL 15 for storage. It runs on Kubernetes with Docker containers and communicates via gRPC."}],"metadata":{"labels":["programming language","database","technology","protocol"],"usecase":"ner","threshold":0.3}}},"json":{"url":null,"description":"Parse complex JSON structures from text","usecase_display_name":"Structured Data Extraction","sample_responses_body":{"input":"Best regards, John Smith, Senior Software Engineer at Acme Corp. Phone: (555) 123-4567, Email: john.smith@acme.com, Office: 123 Main Street, Suite 400, San Francisco, CA 94105","model":"gliner2-base-v1","metadata":{"schema":{"contact":["name::str::Full name","title::str::Job title","company::str::Company name","phone::str::Phone number","email::str::Email address","address::str::Office address"]},"usecase":"json"}},"sample_chat_completions_body":{"model":"gliner2-base-v1","messages":[{"role":"user","content":"Best regards, John Smith, Senior Software Engineer at Acme Corp. Phone: (555) 123-4567, Email: john.smith@acme.com, Office: 123 Main Street, Suite 400, San Francisco, CA 94105"}],"metadata":{"schema":{"contact":["name::str::Full name","title::str::Job title","company::str::Company name","phone::str::Phone number","email::str::Email address","address::str::Office address"]},"usecase":"json"}}},"classification":{"url":null,"description":"Single and multi-label classification with confidence scores","usecase_display_name":"Text Classification","sample_responses_body":{"input":"I absolutely love this product! The quality is outstanding and the customer service was incredibly helpful.","model":"gliner2-base-v1","metadata":{"schema":{"sentiment":["positive","negative","neutral"]},"usecase":"classification"}},"sample_chat_completions_body":{"model":"gliner2-base-v1","messages":[{"role":"user","content":"I absolutely love this product! The quality is outstanding and the customer service was incredibly helpful."}],"metadata":{"schema":{"sentiment":["positive","negative","neutral"]},"usecase":"classification"}}}},"pricing":{"currency":"USD","terms_url":"https://github.com/fastino-ai/GLiNER2/blob/main/LICENSE","use_cases":["PII"],"description":"gliner2-base-v1 is a versatile extraction-and-classification model for the structured tasks that fill most production pipelines. Point it at any text and, with a single API call, pull named entities by your own labels, populate a typed JSON schema straight from messy input, or classify by sentiment, intent, or topic. No fine-tuning and no prompt engineering, just a label set or schema at inference time. Because it's purpose-built and CPU-optimized, it runs faster and cheaper than routing this work to a general-purpose frontier model. Reach for gliner-multi-pii-v1 when the job is dedicated PII redaction. When you need clean structure out of raw text, this is the model.","favicon_url":"https://models-favicon.zerogpu.ai/gliner2-base-v1/fastino.png","model_doc_url":"https://github.com/fastino-ai/GLiNER2/blob/main/README.md","privacy_service":"https://github.com/fastino-ai/GLiNER2/blob/main/LICENSE","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"The application is built with Python 3.11 and uses PostgreSQL 15 for storage. It runs on Kubernetes with Docker containers and communicates via gRPC.","model":"gliner2-base-v1","metadata":{"labels":["programming language","database","technology","protocol"],"usecase":"ner","threshold":0.3}},"sample_output_response":{"id":"87cb749d-f829-4d2d-a82b-0b0634c1e399","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"gliner2-base-v1","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":38,"total_tokens":75,"output_tokens":37,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"87cb749d-f829-4d2d-a82b-0b0634c1e399","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"entities\":{\"programming language\":[\"Python 3.11\"],\"database\":[\"PostgreSQL 15\"],\"technology\":[\"Kubernetes\",\"Docker\",\"gRPC\"],\"protocol\":[\"gRPC\"]}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{"usecase":"ner"},"reasoning":{"effort":null,"summary":null},"created_at":1788988945,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788988945,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"gliner2-base-v1","messages":[{"role":"user","content":"The application is built with Python 3.11 and uses PostgreSQL 15 for storage. It runs on Kubernetes with Docker containers and communicates via gRPC."}],"metadata":{"labels":["programming language","database","technology","protocol"],"usecase":"ner","threshold":0.3}},"sample_output_chat_completions":{"id":"840c44c3-769b-4116-847c-bc632e1326b3","model":"gliner2-base-v1","usage":{"total_tokens":75,"prompt_tokens":38,"completion_tokens":37,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"entities\":{\"programming language\":[\"Python 3.11\"],\"database\":[\"PostgreSQL 15\"],\"technology\":[\"Kubernetes\",\"Docker\",\"gRPC\"],\"protocol\":[\"gRPC\"]}}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788988944,"service_tier":"default"}},"displayPriority":12,"createdAt":"2026-04-03T17:18:47.56526+00:00"},{"modelId":"zlm-v1-iab-domain-classifier","modelType":"iab-classify","modelVersion":"1","taskDisplayName":"Text Classification","quantized":false,"quantization":null,"maxTokens":100,"modelDisplayName":"zlm-v1-iab-domain-classifier","parameters":"149M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Ad Tech","Chat"],"description":"zlm-v1-iab-domain-classifier is a low-latency domain classification model built for advertising, contextual intelligence, and large-scale enrichment workflows. It takes a raw domain name as input and returns structured IAB content categories, topics, keywords, and user-intent signals without requiring page content. This makes it well suited for bidstream enrichment, contextual targeting, domain intelligence, publisher classification, and other high-volume adtech workflows where fast, lightweight classification is needed.","favicon_url":"https://models-favicon.zerogpu.ai/zlm-v1-iab-domain-classifier/logo_dark.png","model_doc_url":"https://docs.zerogpu.ai/","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"coursera.com","model":"zlm-v1-iab-domain-classifier"},"sample_output_response":{"id":"76ec1df7-cb86-41f1-b341-2bafea5a88cf","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"zlm-v1-iab-domain-classifier","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":9,"total_tokens":350,"output_tokens":341,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"76ec1df7-cb86-41f1-b341-2bafea5a88cf","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"audience\":[{\"id\":23,\"parent_id\":20,\"name\":\"Undergraduate Education\",\"tier1_name\":\"Demographic\",\"tier2_name\":\"Education & Occupation\",\"tier3_name\":\"Education (Highest Level)\",\"score\":0.86137356782421},{\"id\":20,\"parent_id\":17,\"name\":\"College Education\",\"tier1_name\":\"Demographic\",\"tier2_name\":\"Education & Occupation\",\"tier3_name\":\"Education (Highest Level)\",\"score\":0.8185467622936815}],\"content\":{\"iab_1_0\":[{\"code\":\"IAB5\",\"name\":\"Education\",\"tier\":1,\"parent_code\":null,\"score\":0.9975345244047042},{\"code\":\"IAB5-6\",\"name\":\"Distance Learning\",\"tier\":2,\"parent_code\":\"IAB5\",\"score\":0.9304775059727456}],\"iab_2_2\":[{\"id\":132,\"parent_id\":0,\"name\":\"Education\",\"tier1_name\":\"Education\",\"score\":0.9975345244047042},{\"id\":148,\"parent_id\":132,\"name\":\"Online Education\",\"tier1_name\":\"Education\",\"tier2_name\":\"Online Education\",\"score\":0.9304775059727456}]}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788989046,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788989046,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"zlm-v1-iab-domain-classifier","messages":[{"role":"user","content":"coursera.com"}]},"sample_output_chat_completions":{"id":"46aab0f1-2a11-4008-a7f5-7cd827b7fb8c","model":"zlm-v1-iab-domain-classifier","usage":{"total_tokens":350,"prompt_tokens":9,"completion_tokens":341,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"audience\":[{\"id\":23,\"parent_id\":20,\"name\":\"Undergraduate Education\",\"tier1_name\":\"Demographic\",\"tier2_name\":\"Education & Occupation\",\"tier3_name\":\"Education (Highest Level)\",\"score\":0.86137356782421},{\"id\":20,\"parent_id\":17,\"name\":\"College Education\",\"tier1_name\":\"Demographic\",\"tier2_name\":\"Education & Occupation\",\"tier3_name\":\"Education (Highest Level)\",\"score\":0.8185467622936815}],\"content\":{\"iab_1_0\":[{\"code\":\"IAB5\",\"name\":\"Education\",\"tier\":1,\"parent_code\":null,\"score\":0.9975345244047042},{\"code\":\"IAB5-6\",\"name\":\"Distance Learning\",\"tier\":2,\"parent_code\":\"IAB5\",\"score\":0.9304775059727456}],\"iab_2_2\":[{\"id\":132,\"parent_id\":0,\"name\":\"Education\",\"tier1_name\":\"Education\",\"score\":0.9975345244047042},{\"id\":148,\"parent_id\":132,\"name\":\"Online Education\",\"tier1_name\":\"Education\",\"tier2_name\":\"Online Education\",\"score\":0.9304775059727456}]}}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788989045,"service_tier":"default"}},"displayPriority":8,"createdAt":"2026-06-25T14:25:23+00:00"},{"modelId":"gliner-multi-pii-v1","modelType":"gliner","modelVersion":"1","taskDisplayName":"PII","quantized":false,"quantization":null,"maxTokens":800,"modelDisplayName":"gliner-multi-pii-v1","parameters":"300M","modelUsecases":{"redact":{"url":null,"description":"Detect PII in text and return a redacted version with [LABEL] masks or per-character replacement","usecase_display_name":"PII Redaction","sample_responses_body":{"input":"Hello Jane Doe, this is John Doe reaching out regarding my recent order. If you need any additional details, feel free to call me at 415-555-0134 during business hours. You can also email me at hi@example.com, and I’ll respond as soon as possible. Looking forward to your update on the issue.","model":"gliner-multi-pii-v1","metadata":{"mask":"label","usecase":"redact"}},"sample_chat_completions_body":{"model":"gliner-multi-pii-v1","messages":[{"role":"user","content":"Hello Jane Doe, this is John Doe reaching out regarding my recent order. If you need any additional details, feel free to call me at 415-555-0134 during business hours. You can also email me at hi@example.com, and I’ll respond as soon as possible. Looking forward to your update on the issue."}],"metadata":{"mask":"label","usecase":"redact"}}},"extract-pii":{"url":null,"description":"Multilingual PII detection across 40+ entity types (identity, contact, government_id, financial, medical, etc.)","usecase_display_name":"PII Extraction","sample_responses_body":{"input":"Contact John Doe at john@example.com or +1-415-555-0134.","model":"gliner-multi-pii-v1","metadata":{"usecase":"extract-pii","threshold":0.5,"categories":["identity","contact"]}},"sample_chat_completions_body":{"model":"gliner-multi-pii-v1","messages":[{"role":"user","content":"Contact John Doe at john@example.com or +1-415-555-0134."}],"metadata":{"usecase":"extract-pii","threshold":0.5,"categories":["identity","contact"]}}}},"pricing":{"currency":"USD","terms_url":"https://github.com/fastino-ai/GLiNER2/blob/main/LICENSE","use_cases":["PII"],"description":"GLiNER Multi PII is a multilingual PII detection and redaction model that supports on-prem deployments as well. It identifies 40+ personally identifiable entity types — identity, contact, government IDs, financial, medical and more. It works natively across six languages: English, French, German, Spanish, Italian, and Portuguese, so a single model covers multi-market and cross-border data without separate per-language pipelines. Built for zero-shot label, it accepts custom label sets at inference time and supports curated PII catalogues, redaction with label or character masks, and generic NER with user-supplied labels.","favicon_url":"https://models-favicon.zerogpu.ai/gliner-multi-pii-v1/fastino.png","model_doc_url":"https://github.com/fastino-ai/GLiNER2/blob/main/README.md","privacy_service":"https://github.com/fastino-ai/GLiNER2/blob/main/LICENSE","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"Hello Jane Doe, this is John Doe reaching out regarding my recent order. If you need any additional details, feel free to call me at 415-555-0134 during business hours. You can also email me at hi@example.com, and I’ll respond as soon as possible. Looking forward to your update on the issue.","model":"gliner-multi-pii-v1","metadata":{"mask":"label","usecase":"redact"}},"sample_output_response":{"id":"77b58986-a1a9-4908-888a-e8a071361faa","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"gliner-multi-pii-v1","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":73,"total_tokens":258,"output_tokens":185,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"77b58986-a1a9-4908-888a-e8a071361faa","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"redacted_text\":\"Hello [PERSON], this is [PERSON] reaching out regarding my recent order. If you need any additional details, feel free to call me at [PHONE_NUMBER] during business hours. You can also email me at [EMAIL], and I’ll respond as soon as possible. Looking forward to your update on the issue.\",\"entities\":[{\"text\":\"Jane Doe\",\"label\":\"person\",\"start\":6,\"end\":14,\"score\":0.9982},{\"text\":\"John Doe\",\"label\":\"person\",\"start\":24,\"end\":32,\"score\":0.9981},{\"text\":\"415-555-0134\",\"label\":\"phone number\",\"start\":133,\"end\":145,\"score\":0.9731},{\"text\":\"hi@example.com\",\"label\":\"email\",\"start\":194,\"end\":208,\"score\":0.978}],\"entities_by_label\":{\"person\":[\"Jane Doe\",\"John Doe\"],\"phone number\":[\"415-555-0134\"],\"email\":[\"hi@example.com\"]}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{"mask":"label","usecase":"redact"},"reasoning":{"effort":null,"summary":null},"created_at":1788988943,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788988943,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"gliner-multi-pii-v1","messages":[{"role":"user","content":"Hello Jane Doe, this is John Doe reaching out regarding my recent order. If you need any additional details, feel free to call me at 415-555-0134 during business hours. You can also email me at hi@example.com, and I’ll respond as soon as possible. Looking forward to your update on the issue."}],"metadata":{"mask":"label","usecase":"redact"}},"sample_output_chat_completions":{"id":"33474320-af81-486f-a511-1ecec6cecd24","model":"gliner-multi-pii-v1","usage":{"total_tokens":258,"prompt_tokens":73,"completion_tokens":185,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"redacted_text\":\"Hello [PERSON], this is [PERSON] reaching out regarding my recent order. If you need any additional details, feel free to call me at [PHONE_NUMBER] during business hours. You can also email me at [EMAIL], and I’ll respond as soon as possible. Looking forward to your update on the issue.\",\"entities\":[{\"text\":\"Jane Doe\",\"label\":\"person\",\"start\":6,\"end\":14,\"score\":0.9982},{\"text\":\"John Doe\",\"label\":\"person\",\"start\":24,\"end\":32,\"score\":0.9981},{\"text\":\"415-555-0134\",\"label\":\"phone number\",\"start\":133,\"end\":145,\"score\":0.9731},{\"text\":\"hi@example.com\",\"label\":\"email\",\"start\":194,\"end\":208,\"score\":0.978}],\"entities_by_label\":{\"person\":[\"Jane Doe\",\"John Doe\"],\"phone number\":[\"415-555-0134\"],\"email\":[\"hi@example.com\"]}}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788988941,"service_tier":"default"}},"displayPriority":7,"createdAt":"2026-04-30T01:35:24.747505+00:00"},{"modelId":"zlm-v1-iab-classify-edge","modelType":"iab-classify","modelVersion":"1","taskDisplayName":"Text Classification","quantized":false,"quantization":null,"maxTokens":400,"modelDisplayName":"zlm-v1-iab-classify-edge","parameters":"90M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Ad Tech"],"description":"ZeroGPU's IAB classifier maps any text to the industry-standard IAB Content Taxonomy in a single, fast inference call. Each call returns categories across both the 1.0 and 2.2 taxonomies plus matched audience segments, every result scored for confidence. At 90M parameters on ONNX, it's built for the high-volume, sub-millisecond classification that ad tech and content platforms demand, right at the edge. When you need clean category and audience signals on every request, this is the model — and when you need the full profile, reach for the enriched variant.","favicon_url":"https://models-favicon.zerogpu.ai/zlm-v1-iab-classify-edge/logo_dark.png","model_doc_url":"https://docs.zerogpu.ai/","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"Technology has quietly reshaped the rhythm of everyday life, weaving itself into routines so seamlessly that it often goes unnoticed. From the moment a smartphone alarm wakes someone in the morning to the final glance at a glowing screen before sleep, digital systems guide communication, navigation, work, and entertainment. This transformation did not happen overnight. It emerged through decades of incremental innovation, each new tool building upon the last, until the modern world became deeply interconnected.One of the most significant changes has been the speed at which information travels.","model":"zlm-v1-iab-classify-edge"},"sample_output_response":{"id":"e41006a7-93a4-476d-baac-7dd7a4e3d2c3","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"zlm-v1-iab-classify-edge","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":150,"total_tokens":477,"output_tokens":327,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"e41006a7-93a4-476d-baac-7dd7a4e3d2c3","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"audience\":[{\"name\":\"Technology & Computing\",\"score\":0.7817539670915485},{\"name\":\"Consumer Electronics\",\"score\":0.6822751829244846},{\"name\":\"Artificial Intelligence\",\"score\":0.5530661812982451},{\"name\":\"Computing\",\"score\":0.5447119734202016}],\"content\":{\"iab_1_0\":[{\"name\":\"Technology & Computing\",\"score\":0.8513270663261827},{\"name\":\"Cell Phones\",\"score\":0.6809580262884584},{\"name\":\"Video & Computer Games\",\"score\":0.5204820652021064}],\"iab_2_2\":[{\"name\":\"Technology & Computing\",\"score\":0.8513270663261827},{\"name\":\"Smartphones\",\"score\":0.6809580262884584},{\"name\":\"Wearable Technology\",\"score\":0.6415416535034761},{\"name\":\"Consumer Electronics\",\"score\":0.595014844803955},{\"name\":\"Artificial Intelligence\",\"score\":0.5541231139744616},{\"name\":\"Mobile Games\",\"score\":0.5204820652021064}]}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788989042,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788989042,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"zlm-v1-iab-classify-edge","messages":[{"role":"user","content":"What are the basics of exercise and physical fitness?Exercise is anything that gets your body moving. Regular exercise is one of the best things you can do for your health. It has many benefits, including improving your overall health and fitness, and reducing your risk for many chronic (long-term) diseases.Every physical fitness routine is built on a few simple ideas. These include:Make exercise a habit, as your body adapts to the type of activity you do most often. Regular practice will help you improve.Build up your activity level slowly to help you continue to get stronger, faster, or more flexible without pushing too hard all at once.Challenge yourself by lifting slightly heavier weights, adding a few more minutes to your walk, or increasing your pace."}]},"sample_output_chat_completions":{"id":"e32e0fe9-1baa-407a-92fb-7389d4c7ca99","model":"zlm-v1-iab-classify-edge","usage":{"total_tokens":563,"prompt_tokens":192,"completion_tokens":371,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"{\"audience\":[{\"name\":\"Fitness and Exercise\",\"score\":0.8420880787857076},{\"name\":\"Healthy Living\",\"score\":0.8267880846346385},{\"name\":\"Wellness\",\"score\":0.7113321459055835},{\"name\":\"Weight Loss\",\"score\":0.6642570579317274},{\"name\":\"Fitness and Exercise\",\"score\":0.6374571135852},{\"name\":\"Senior Health\",\"score\":0.5551626996774902}],\"content\":{\"iab_1_0\":[{\"name\":\"Exercise\",\"score\":0.9646817625854934},{\"name\":\"Health & Fitness\",\"score\":0.9354958077737687},{\"name\":\"Weight Loss\",\"score\":0.6993215402057686},{\"name\":\"Video & Computer Games\",\"score\":0.648178913437892},{\"name\":\"Physical Therapy\",\"score\":0.6470848317721539}],\"iab_2_2\":[{\"name\":\"Fitness and Exercise\",\"score\":0.9646817625854934},{\"name\":\"Healthy Living\",\"score\":0.9354958077737687},{\"name\":\"Wellness\",\"score\":0.8064933325286461},{\"name\":\"Weight Loss\",\"score\":0.6993215402057686},{\"name\":\"Exercise and Fitness Video Games\",\"score\":0.648178913437892},{\"name\":\"Physical Therapy\",\"score\":0.6470848317721539}]}}","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788989041,"service_tier":"default"}},"displayPriority":4,"createdAt":"2026-03-04T16:10:05.376044+00:00"},{"modelId":"LFM2.5-1.2B-Thinking","modelType":"lfm2.5","modelVersion":"2.5","taskDisplayName":"Text Generation","quantized":true,"quantization":"Q4_K_M","maxTokens":32768,"modelDisplayName":"LFM2.5-1.2B-Thinking","parameters":"1.2B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://www.liquid.ai/terms-conditions","use_cases":["Intent Detection","Translation","Summarization","Reasoning","Function Calling"],"description":"LFM2.5-1.2B-Thinking is Liquid AI’s compact 1.2B-parameter reasoning model designed for multi-step problem solving, planning, data extraction, and agentic workflows. Its small footprint and 32K context window make it well suited for low-latency inference at the edge, especially for high-volume tasks that need more reasoning than a simple classifier. It is a strong fit for agent planning, routing, tool orchestration, structured extraction, and lightweight reasoning workloads where speed and cost matter. For knowledge-heavy or complex coding tasks, a larger general-purpose model may be a better fit.","favicon_url":"https://models-favicon.zerogpu.ai/LFM2.5-1.2B-Thinking/LFM2.5-1.2B-Thinking_liquid_ai_logo.png","model_doc_url":"https://docs.liquid.ai/deployment/on-device/android/ai-agent-usage-guide#text-models","privacy_service":"https://www.liquid.ai/lfm-license","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"Our whole team got locked out of the dashboard this morning after the payment failed. We have a product demo with a customer in 2 hours and cannot log in. Card on file should be valid.","model":"LFM2.5-1.2B-Thinking","instructions":"You are a support triage assistant. Read the ticket, briefly reason about the signals, then assign one team (Billing, Technical, Account, General) and one priority (Low, Medium, High). Decide once and commit. End with a single line: TEAM | PRIORITY."},"sample_output_response":{"id":"a6c2f274-d98f-4847-9fb2-c0db4544d5d0","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"LFM2.5-1.2B-Thinking","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":112,"total_tokens":603,"output_tokens":491,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"a6c2f274-d98f-4847-9fb2-c0db4544d5d0","role":"assistant","type":"message","status":"completed","content":[{"text":"Okay, let's tackle this. The user says their team is locked out of the dashboard because payments failed, and they can't log in now. They need to handle a product demo in two hours with a valid card on file. So first, I need to figure out what's going on here.\n\nThe task is to assign a team (Billing, Technical, Account, General) and a priority (Low, Medium, High). Then decide which team handles it and set the priority accordingly. The response should end with TEAM | PRIORITY on one line.\n\nFirst, the problem involves technical issues related to login. Since they mentioned payment failure and need to log in for a demo, the immediate issue is likely Technical support because they might need help fixing the login problem. The team handling technical issues would be Technical. \n\nNow, priority: since the customer has a product demo in two hours, it's urgent. But since the payment failure might be a billing issue (Billing team), but the primary problem here is the login blockage. The urgency is high because time is critical for the demo. However, the card is valid, so maybe Billing could resolve it if they can check payment details. Alternatively, Technical might handle the login issue. Since the user mentioned \"assign one team and priority\", I need to pick one team that can address both the technical problem and the urgency. Since the demo is time-sensitive, Technical might be better because they can fix the login issue. The priority: High because it's critical for the demo happening soon. \n\nWait, but Billing team might handle payment issues. But if the problem is login, maybe Technical is more appropriate. Priority is High due to imminent demo deadline. So Team: Technical, Priority: High. Then the final answer should be TEAM | PRIORITY: TEAM Technical | PRIORITY High. Let me confirm: the system says to end with a single line TEAM | PRIORITY. So if I pick Technical and High, then \"Technical | High\".\n\nTechnical | High  \nTEAM | PRIORITY.","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788990220,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788990220,"instructions":"You are a support triage assistant. Read the ticket, briefly reason about the signals, then assign one team (Billing, Technical, Account, General) and one priority (Low, Medium, High). Decide once and commit. End with a single line: TEAM | PRIORITY.","max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"LFM2.5-1.2B-Thinking","messages":[{"role":"system","content":"You are a support triage assistant. Read the ticket, briefly reason about the signals, then assign one team (Billing, Technical, Account, General) and one priority (Low, Medium, High). Decide once and commit. End with a single line: TEAM | PRIORITY."},{"role":"user","content":"Our whole team got locked out of the dashboard this morning after the payment failed. We have a product demo with a customer in 2 hours and cannot log in. Card on file should be valid."}]},"sample_output_chat_completions":{"id":"b5ed46d1-c66e-482c-a637-1e9776ddfae0","model":"LFM2.5-1.2B-Thinking","usage":{"total_tokens":672,"prompt_tokens":112,"completion_tokens":560,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"Okay, let's tackle this. The user says their whole team is locked out of the dashboard because of a payment failure, and they can't log in. They need to handle a product demo with a customer in 2 hours, but their card isn't valid anymore. The task is to assign a team and priority for this issue.\n\nFirst, the signals here are: lockouts due to payment failure, inability to log in, urgency because of the demo in 2 hours. Since they can't use the card, maybe the problem is related to payment issues or account access. The possible teams are Billing, Technical, Account, General. Priorities are Low, Medium, High.\n\nThe main issue here seems like a technical problem affecting access. Since the payment failed, maybe the billing team would handle resolving payment issues. Alternatively, if the card is expired or invalid, the Technical team might fix it. But since they can't log in, maybe Technical is involved. However, the user mentions their own team is locked out, but maybe the support team needs to coordinate between teams. Wait, the question says to assign one team and one priority. Since the user is part of a team that's locked out, perhaps the main issue is technical (since payment failed), so Technical team would handle fixing that. The priority is High because the demo is imminent and they can't proceed without access. Alternatively, maybe Billing since payment issues are related to billing. But the user says their card is valid on file, but maybe the problem is with the payment gateway or account status. Since the customer is coming soon, High priority makes sense. So Team: Technical, Priority: High. Then the final answer should be TEAM | PRIORITY: TEAM=Technical, PRIORITY=High → Technical | High. Wait but maybe Account? Wait, the user says \"card on file should be valid\" so maybe Billing is handling card details. Alternatively, maybe since it's about payment failure, Technical team would resolve that. I think Technical and High priority. Let me confirm: The problem is locked out due to payment failure. Since they can't log in, likely a technical issue. Assign Team: Technical, Priority: High. So the answer should be TEAM: Technical, PRIORITY: High → Technical | High.\n\nTechnical | High  \nTEAM | PRIORITY.","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788990212,"service_tier":"default"}},"displayPriority":3,"createdAt":"2026-03-23T17:52:50+00:00"},{"modelId":"LFM2.5-1.2B-Instruct","modelType":"lfm2.5","modelVersion":"2.5","taskDisplayName":"Text Generation","quantized":true,"quantization":"Q4_K_M","maxTokens":32768,"modelDisplayName":"LFM2.5-1.2B-Instruct","parameters":"1.2B","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://www.liquid.ai/terms-conditions","use_cases":["Intent Detection","Translation","Summarization","Chat","Function Calling"],"description":"LFM2.5-1.2B-Instruct is Liquid AI’s compact 1.2B-parameter instruction model built for efficient text generation, conversational AI, and lightweight agent workflows. It supports native tool calling and multilingual use, while its small footprint makes it well suited for low-latency inference on edge devices and resource-constrained environments. It is a strong fit for chat, simple assistants, instruction following, routing, and high-volume agent tasks where speed, efficiency, and cost matter more than deep reasoning.","favicon_url":"https://models-favicon.zerogpu.ai/LFM2.5-1.2B-Instruct/liquid_ai_logo.png","model_doc_url":"https://docs.liquid.ai/deployment/on-device/android/ai-agent-usage-guide#text-models","privacy_service":"https://www.liquid.ai/lfm-license","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"I just completed a 5K run in 28 minutes. Give me a short motivational follow-up message.","model":"LFM2.5-1.2B-Instruct","instructions":"You are a friendly in-app assistant for a fitness app."},"sample_output_response":{"id":"8d090cbe-01b6-4beb-adf3-074d799f6e6b","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"LFM2.5-1.2B-Instruct","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":40,"total_tokens":87,"output_tokens":47,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"8d090cbe-01b6-4beb-adf3-074d799f6e6b","role":"assistant","type":"message","status":"completed","content":[{"text":"That's an amazing accomplishment! Keep up the great work—each step brings you closer to your goals. You're stronger than you think! 💪🏃‍♂️\n\nWould you like tips to help you keep improving?","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788988965,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788988965,"instructions":"You are a friendly in-app assistant for a fitness app.","max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"model":"LFM2.5-1.2B-Instruct","messages":[{"role":"system","content":"You are a friendly in-app assistant for a fitness app."},{"role":"user","content":"I just completed a 5K run in 28 minutes. Give me a short motivational follow-up message."}]},"sample_output_chat_completions":{"id":"39af127f-2036-4a9e-bdc9-c98a31d6044a","model":"LFM2.5-1.2B-Instruct","usage":{"total_tokens":87,"prompt_tokens":40,"completion_tokens":47,"prompt_tokens_details":{"audio_tokens":0,"cached_tokens":0},"completion_tokens_details":{"audio_tokens":0,"reasoning_tokens":0,"accepted_prediction_tokens":0,"rejected_prediction_tokens":0}},"object":"chat.completion","choices":[{"index":0,"message":{"role":"assistant","content":"That's an amazing accomplishment! Keep up the great work—each step brings you closer to your goals. You're stronger than you think! 💪🏃‍♂️\n\nWould you like tips to help you keep improving?","refusal":null,"annotations":[]},"logprobs":null,"finish_reason":"stop"}],"created":1788988961,"service_tier":"default"}},"displayPriority":2,"createdAt":"2026-03-26T17:18:47.56526+00:00"},{"modelId":"zlm-v1-signal-extract","modelType":"enrichment-seq2seq","modelVersion":"v1","taskDisplayName":"Text Classification","quantized":false,"quantization":null,"maxTokens":400,"modelDisplayName":"zlm-v1-signal-extract","parameters":"80M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://zerogpu.ai/terms","use_cases":["Ad Tech","Chat"],"description":"zlm-v1-signal-extract is a specialized ZeroGPU model for turning unstructured text into structured signals that downstream systems can act on. It extracts topics, keywords, intent, and other contextual attributes in a single inference call, making it well suited for content enrichment, contextual intelligence, ad targeting, agent routing, recommendation systems, and analytics pipelines. The model returns structured output and is designed for fast, high-volume workloads where lightweight signal extraction is more efficient than using a general-purpose LLM.","favicon_url":"https://models-favicon.zerogpu.ai/zlm-v1-iab-classify-edge/logo_dark.png","model_doc_url":"https://docs.zerogpu.ai/","privacy_service":"https://zerogpu.ai/privacy-policy","input_per_1m_tokens":0.02,"output_per_1m_tokens":0.05,"sample_responses_body":{"input":"How can I pay using my credit card?","model":"zlm-v1-signal-extract","categories":["signals"]},"sample_output_response":{"id":"ecdcf63a-97bb-4cc1-9519-9cddc149e005","text":{"format":{"type":"text"}},"user":null,"error":null,"model":"zlm-v1-signal-extract","store":true,"tools":[],"top_p":1,"usage":{"input_tokens":9,"total_tokens":49,"output_tokens":40,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"object":"response","output":[{"id":"ecdcf63a-97bb-4cc1-9519-9cddc149e005","role":"assistant","type":"message","status":"completed","content":[{"text":"{\"keywords\":[\"credit card\",\"login\",\"official website\",\"account access\"],\"user_intent\":{\"name\":\"pay using credit card\",\"category\":\"navigational\",\"score\":0.95}}","type":"output_text","annotations":[]}]}],"status":"completed","metadata":{},"reasoning":{"effort":null,"summary":null},"created_at":1788989056,"truncation":"auto","temperature":1,"tool_choice":"auto","completed_at":1788989056,"instructions":null,"max_output_tokens":null,"incomplete_details":null,"parallel_tool_calls":true,"previous_response_id":null},"sample_chat_completions_body":{"messages":[{"role":"user","content":"How can I pay using my credit card?"}],"categories":["signals"],"max_keywords":6},"sample_output_chat_completions":null},"displayPriority":0,"createdAt":"2026-07-16T04:13:43.125566+00:00"},{"modelId":"bge-small-en-v1.5","modelType":"BERT","modelVersion":"v1","taskDisplayName":"Text Embedding","quantized":false,"quantization":null,"maxTokens":512,"modelDisplayName":"bge-small-en-v1.5","parameters":"33M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://huggingface.co/BAAI/bge-small-en-v1.5","use_cases":["Semantic Search","RAG","Similarity Matching","Document Retrieval"],"description":"bge-small-en-v1.5 is a lightweight English embedding model from BAAI designed for semantic search, retrieval, similarity scoring, ranking, and RAG workflows. It converts text into 384-dimensional dense embeddings that capture semantic meaning, making it well suited for finding related documents, matching queries to content, clustering text, and powering vector search. Its compact size makes it a strong choice for high-volume embedding workloads where latency and cost matter.","favicon_url":"https://cdn-avatars.huggingface.co/v1/production/uploads/1664511063789-632c234f42c386ebd2710434.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/bge-small-en-v1-5","privacy_service":"https://huggingface.co/BAAI/bge-small-en-v1.5","input_per_1m_tokens":0.004,"output_per_1m_tokens":0,"sample_responses_body":{"input":"Apple announced the iPhone 18 Pro and iPhone 18 Pro Max in September 2026, introducing its latest generation of premium iPhones alongside new Apple Intelligence features and updated hardware designed for AI-powered experiences.","model":"bge-small-en-v1.5"},"sample_output_response":{"data":[{"index":0,"object":"embedding","embedding":[-0.09202273190021515,-0.012649217620491982,0.013731790706515312,-0.023960713297128677,-0.02368033491075039,0.023265227675437927,0.00935368612408638,0.007233922835439444,0.008743375539779663,0.048679422587156296,0.04626619815826416,0.062070637941360474,-0.03487664833664894,-0.001030408195219934,0.03456650301814079,0.04675877466797829,0.07235351949930191,-0.16296768188476562,-0.0675748810172081,-0.017709992825984955,0.05496418848633766,-0.002614137250930071,-0.043803438544273376,-0.050362925976514816,-0.015353885479271412,0.04105903208255768,-0.036215391010046005,0.015455962158739567,-0.01940186880528927,-0.08967839926481247,0.0411781370639801,-0.0010917322942987084,0.06808426976203918,-0.01861838810145855,-0.055680885910987854,-0.03827083483338356,-0.10508666932582855,0.01741459034383297,0.01431107148528099,0.02881617657840252,-0.016794756054878235,0.00982289481908083,-0.042352840304374695,0.019441476091742516,0.025103731080889702,0.049459222704172134,-0.013105456717312336,-0.047856368124485016,-0.03968410938978195,-0.05028378590941429,0.07150757312774658,-0.003980314824730158,0.029480135068297386,-0.021641289815306664,-0.009826264344155788,0.015647368505597115,-0.033841222524642944,-0.02441381849348545,0.04310641065239906,0.07761572301387787,0.04379939287900925,-0.056369196623563766,-0.184054434299469,0.09509719163179398,0.004553019534796476,0.029505416750907898,0.012350276112556458,-0.06336450576782227,0.008856141939759254,-0.026057522743940353,0.015193683095276356,-0.004560056608170271,0.04642387852072716,-0.03520816192030907,-0.004917451646178961,0.06363947689533234,0.04967692866921425,-0.037382155656814575,0.04398861527442932,-0.02398291602730751,0.046163469552993774,-0.04309014603495598,-0.05326875299215317,-0.004907146096229553,0.027286015450954437,-0.0343150720000267,-0.014373340643942356,-0.015506497584283352,0.010586194694042206,-0.04573124274611473,-0.0646257996559143,0.01950766146183014,0.030869511887431145,0.05022627115249634,-0.040752436965703964,0.01584898866713047,0.04297974333167076,-0.0393073596060276,-0.03330981358885765,0.32746556401252747,-0.03245248645544052,0.015413442626595497,0.053003180772066116,-0.009076392278075218,0.023769693449139595,-0.03640449419617653,-0.07464694231748581,0.026110852137207985,-0.0513390377163887,-0.010461081750690937,-0.006727155297994614,0.013664394617080688,0.056697867810726166,0.047958895564079285,0.011109077371656895,0.004776224959641695,-0.003224665066227317,0.0787012055516243,0.10145211964845657,-0.022672103717923164,0.0034687600564211607,0.026544826105237007,0.00403239531442523,0.0038258149288594723,-0.07090514153242111,0.013365622609853745,0.010365044698119164,0.10613814741373062,-0.08300755172967911,-0.008398408070206642,0.05363151803612709,0.0207392405718565,-0.10141114145517349,0.0437997505068779,0.09539458900690079,-0.037923846393823624,-0.05206451192498207,-0.04426981136202812,0.01392200868576765,-0.05512819439172745,-0.04608217626810074,-0.02542390115559101,0.08440026640892029,-0.08424761891365051,-0.01892993226647377,0.02876853011548519,-0.049458108842372894,0.019983641803264618,0.03894282504916191,-0.0036818478256464005,-0.049906253814697266,0.039521750062704086,0.042540501803159714,-0.041131600737571716,0.02084154635667801,0.04900052770972252,-0.0013438756577670574,0.0010749484645202756,-0.06644319742918015,0.06454232335090637,-0.025757335126399994,0.028544820845127106,-0.036802589893341064,-0.008225667290389538,0.033540599048137665,-0.18110983073711395,0.029439818114042282,0.02727825753390789,0.014191587455570698,0.013254844583570957,0.0015214699087664485,0.034632809460163116,0.04482649639248848,0.043910227715969086,0.04233675077557564,0.011504917405545712,-0.025921551510691643,0.018012873828411102,-0.03311470150947571,-0.015394533984363079,-0.011205842718482018,-0.04706235229969025,-0.030018897727131844,-0.022833019495010376,0.023752059787511826,-0.032737065106630325,0.0647469237446785,-0.05378261208534241,0.11291617900133133,0.0037638768553733826,-0.00016582146054133773,0.04657110944390297,-0.06050657108426094,0.07347036898136139,0.0038799908943474293,-0.02043825387954712,-0.04233165457844734,0.057764239609241486,0.06465432792901993,-0.07570301741361618,-0.014139702543616295,-0.050386421382427216,0.015668602660298347,0.02641191892325878,0.021640079095959663,-0.052134234458208084,-0.03293068706989288,0.0057283127680420876,0.06373613327741623,0.03994883969426155,0.008153853937983513,0.002542013768106699,0.1314178705215454,-0.04486621916294098,-0.01842437870800495,-0.02041652612388134,-0.07134736329317093,0.02241600677371025,-0.025542449206113815,0.008587931282818317,-0.022990712895989418,0.004920446779578924,0.03067035786807537,-0.22002841532230377,0.008502228185534477,0.0017084450228139758,-0.030809829011559486,-0.025517096742987633,-0.055248718708753586,0.024368073791265488,-0.05111595243215561,0.0190411489456892,0.061271633952856064,0.0358588732779026,0.09288686513900757,0.01918652094900608,0.0626506507396698,0.048279497772455215,0.033364638686180115,0.05903071537613869,0.0018558190204203129,-0.013393406756222248,0.08251422643661499,-0.030209766700863838,0.007458274718374014,-0.05464792251586914,0.02951160818338394,-0.005680758040398359,0.007991868071258068,0.10789529979228973,0.007701714988797903,-0.024606332182884216,0.02578401006758213,0.005628433544188738,-0.013781523331999779,0.013708755373954773,-0.10304085165262222,0.012530005536973476,-0.017510458827018738,-0.008718983270227909,-0.04164305701851845,-0.05468854308128357,-0.050293929874897,-0.0068552615121006966,0.06989758461713791,-0.0006853280938230455,-0.04239792376756668,0.0004709762579295784,-0.04940994456410408,0.0065794410184025764,0.08534801006317139,-0.0017745036166161299,-0.008273163810372353,0.0018915970576927066,0.05265960097312927,0.024524638429284096,-0.019238943234086037,0.04358161985874176,-0.03722507134079933,-0.09034059196710587,0.07876874506473541,0.024358907714486122,0.005521045532077551,0.008685626089572906,-0.0014445006381720304,-0.07503936439752579,-0.03528553247451782,0.002905204426497221,-0.028832949697971344,-0.004105026833713055,0.006664895452558994,-0.06258200109004974,-0.0800192654132843,0.032445214688777924,0.020780237391591072,0.0016029768157750368,-0.030201291665434837,0.0433381162583828,-0.05447155237197876,0.0812678337097168,-0.00507687870413065,0.08374607563018799,-0.034162845462560654,0.01677907444536686,0.02851618081331253,0.05001477152109146,0.01770070567727089,0.015191083773970604,0.03375381976366043,0.03687154874205589,0.0026218420825898647,-0.018604610115289688,-0.060792431235313416,-0.005496606230735779,-0.05983292683959007,-0.00788507517427206,-0.0410369448363781,0.06303630024194717,-0.03920983150601387,-0.20314037799835205,0.04612966254353523,0.07547056674957275,0.005850460845977068,-0.03685472905635834,0.010104878805577755,0.004716459661722183,0.012789126485586166,-0.002317410195246339,0.030365532264113426,-0.011908427812159061,-0.04320037364959717,0.0346238799393177,0.042172592133283615,-0.00856857467442751,-0.028731169179081917,0.08308260142803192,-0.07771097123622894,0.00634220615029335,0.02975575439631939,-0.004854144528508186,-0.016627248376607895,0.11983750760555267,-0.05196226015686989,-0.0024268885608762503,-0.03465183451771736,-0.04450055584311485,0.032234322279691696,-0.007985720410943031,-0.025159195065498352,0.02177439257502556,0.016506727784872055,0.026294389739632607,-0.02757219411432743,0.0073161558248102665,0.02297736518085003,-0.0017698041629046202,0.0011741507332772017,-0.07718688249588013,0.06731422245502472,-0.08027389645576477,-0.05014760047197342,-0.06470681726932526,0.062204357236623764,0.09175944328308105,0.003945464733988047,-0.07492146641016006,0.04742419347167015,-0.013012362644076347,-0.02769402042031288,0.01935245282948017,-0.044279273599386215,-0.01632615551352501,-0.027672983705997467,0.005936210975050926,-0.010767778381705284,0.04858498275279999,-0.004354491364210844,-0.02543764002621174,-0.00046580188791267574,0.06889168918132782,0.047706518322229385,-0.07291319966316223,0.04089903086423874,0.00653028953820467]}],"model":"bge-small-en-v1.5","usage":{"total_tokens":41,"prompt_tokens":41},"object":"list"},"sample_chat_completions_body":null,"sample_output_chat_completions":null},"displayPriority":0,"createdAt":"2026-08-20T16:18:32.672409+00:00"},{"modelId":"all-minilm-l6-v2","modelType":"BERT","modelVersion":"v1","taskDisplayName":"Text Embedding","quantized":false,"quantization":null,"maxTokens":512,"modelDisplayName":"all-minilm-l6-v2","parameters":"22.7M","modelUsecases":null,"pricing":{"currency":"USD","terms_url":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","use_cases":["Semantic Search","Similarity Matching","Document Retrieval","RAG"],"description":"all-MiniLM-L6-v2 is a lightweight sentence embedding model that converts sentences and short paragraphs into 384-dimensional dense vectors. These embeddings capture semantic meaning, making the model well suited for semantic search, retrieval, similarity scoring, clustering, deduplication, recommendation, and text ranking. Its compact architecture is optimized for fast, low-cost inference, making it a strong choice for high-volume embedding workloads and retrieval pipelines.","favicon_url":"https://models-favicon.zerogpu.ai/all-minilm-l6-v2/all-minilm-l6-v2.png","model_doc_url":"https://docs.zerogpu.ai/api-reference/models/all-minilm-l6-v2","privacy_service":"https://huggingface.co/datasets/choosealicense/licenses/blob/main/markdown/apache-2.0.md","input_per_1m_tokens":0.004,"output_per_1m_tokens":0,"sample_responses_body":{"input":"Apple announced the iPhone 18 Pro and iPhone 18 Pro Max in September 2026, introducing its latest generation of premium iPhones alongside new Apple Intelligence features and updated hardware designed for AI-powered experiences.","model":"all-minilm-l6-v2"},"sample_output_response":null,"sample_chat_completions_body":{"input":"Apple is expected to unveil its next-generation M5 chip at WWDC this June, promising a 40% boost in GPU performance and a new dedicated AI core for on-device machine learning tasks.","model":"all-minilm-l6-v2"},"sample_output_chat_completions":null},"displayPriority":0,"createdAt":"2026-08-11T14:55:41.912882+00:00"}]}