From 7f578cb980c85a840347130eccdc5475fd29cbf8 Mon Sep 17 00:00:00 2001 From: shroominic Date: Thu, 29 May 2025 12:42:59 +0000 Subject: [PATCH 1/2] update models --- models.json | 588 +++++++++++++++++++++++++--------------------------- 1 file changed, 287 insertions(+), 301 deletions(-) diff --git a/models.json b/models.json index acca2523..6f05d95b 100644 --- a/models.json +++ b/models.json @@ -1,5 +1,103 @@ { "models": [ + { + "id": "google/gemma-2b-it", + "hugging_face_id": "google/gemma-2b-it", + "name": "Google: Gemma 2 2B", + "created": 1748460815, + "description": "Gemma 2 2B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini).\n\nGemma models are well-suited for a variety of text generation tasks, including question answering, summarization, and reasoning.\n\nSee the [launch announcement](https://blog.google/technology/developers/google-gemma-2/) for more details. Usage of Gemma is subject to Google's [Gemma Terms of Use](https://ai.google.dev/gemma/terms).", + "context_length": 8192, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Gemini", + "instruct_type": "gemma" + }, + "pricing": { + "prompt": "0.0000001", + "completion": "0.0000001", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 8192, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "stop", + "frequency_penalty", + "presence_penalty", + "top_k", + "repetition_penalty", + "logit_bias", + "min_p", + "response_format" + ] + }, + { + "id": "deepseek/deepseek-r1-0528", + "hugging_face_id": "deepseek-ai/DeepSeek-R1-0528", + "name": "DeepSeek: R1 0528", + "created": 1748455170, + "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass.\n\nFully open-source model.", + "context_length": 163840, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "DeepSeek", + "instruct_type": "deepseek-r1" + }, + "pricing": { + "prompt": "0.0000005", + "completion": "0.00000218", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 163840, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "reasoning", + "include_reasoning", + "presence_penalty", + "frequency_penalty", + "repetition_penalty", + "top_k", + "stop", + "seed", + "min_p", + "logit_bias", + "top_logprobs", + "logprobs", + "response_format", + "structured_outputs" + ] + }, { "id": "sarvamai/sarvam-m", "hugging_face_id": "sarvamai/sarvam-m", @@ -165,7 +263,7 @@ "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, - "is_moderated": true + "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ @@ -1069,7 +1167,7 @@ }, "top_provider": { "context_length": 128000, - "max_completion_tokens": null, + "max_completion_tokens": 20000, "is_moderated": false }, "per_request_limits": null, @@ -1175,20 +1273,20 @@ }, "per_request_limits": null, "supported_parameters": [ - "tools", - "tool_choice", "max_tokens", "temperature", "top_p", "reasoning", "include_reasoning", + "seed", + "tools", + "tool_choice", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "response_format", "top_k", - "seed", "min_p", "structured_outputs", "logprobs", @@ -1234,20 +1332,20 @@ "top_p", "reasoning", "include_reasoning", - "presence_penalty", + "stop", "frequency_penalty", - "repetition_penalty", + "presence_penalty", "top_k", + "repetition_penalty", + "logit_bias", + "min_p", + "response_format", + "seed", "tools", "tool_choice", - "stop", - "response_format", "structured_outputs", - "logit_bias", "logprobs", - "top_logprobs", - "seed", - "min_p" + "top_logprobs" ] }, { @@ -1278,7 +1376,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1326,7 +1424,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1374,7 +1472,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1568,10 +1666,18 @@ "supported_parameters": [ "tools", "tool_choice", - "seed", "max_tokens", + "reasoning", + "include_reasoning", + "structured_outputs", "response_format", - "structured_outputs" + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "logit_bias", + "logprobs", + "top_logprobs" ] }, { @@ -1617,52 +1723,6 @@ "structured_outputs" ] }, - { - "id": "qwen/qwen2.5-coder-7b-instruct", - "hugging_face_id": "Qwen/Qwen2.5-Coder-7B-Instruct", - "name": "Qwen: Qwen2.5 Coder 7B Instruct", - "created": 1744734887, - "description": "Qwen2.5-Coder-7B-Instruct is a 7B parameter instruction-tuned language model optimized for code-related tasks such as code generation, reasoning, and bug fixing. Based on the Qwen2.5 architecture, it incorporates enhancements like RoPE, SwiGLU, RMSNorm, and GQA attention with support for up to 128K tokens using YaRN-based extrapolation. It is trained on a large corpus of source code, synthetic data, and text-code grounding, providing robust performance across programming languages and agentic coding workflows.\n\nThis model is part of the Qwen2.5-Coder family and offers strong compatibility with tools like vLLM for efficient deployment. Released under the Apache 2.0 license.", - "context_length": 32768, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Qwen", - "instruct_type": null - }, - "pricing": { - "prompt": "0.00000001", - "completion": "0.00000003", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 32768, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "openai/gpt-4.1", "hugging_face_id": "", @@ -2051,6 +2111,54 @@ "top_logprobs" ] }, + { + "id": "nvidia/llama-3.1-nemotron-ultra-253b-v1", + "hugging_face_id": "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", + "name": "NVIDIA: Llama 3.1 Nemotron Ultra 253B v1", + "created": 1744115059, + "description": "Llama-3.1-Nemotron-Ultra-253B-v1 is a large language model (LLM) optimized for advanced reasoning, human-interactive chat, retrieval-augmented generation (RAG), and tool-calling tasks. Derived from Meta\u2019s Llama-3.1-405B-Instruct, it has been significantly customized using Neural Architecture Search (NAS), resulting in enhanced efficiency, reduced memory usage, and improved inference latency. The model supports a context length of up to 128K tokens and can operate efficiently on an 8x NVIDIA H100 node.\n\nNote: you must include `detailed thinking on` in the system prompt to enable reasoning. Please see [Usage Recommendations](https://huggingface.co/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1#quick-start-and-usage-recommendations) for more.", + "context_length": 131072, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Llama3", + "instruct_type": null + }, + "pricing": { + "prompt": "0.0000006", + "completion": "0.0000018", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 131072, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "reasoning", + "include_reasoning", + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "top_k", + "logit_bias", + "logprobs", + "top_logprobs" + ] + }, { "id": "meta-llama/llama-4-maverick", "hugging_face_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct", @@ -2362,8 +2470,8 @@ "instruct_type": null }, "pricing": { - "prompt": "0.0000008", - "completion": "0.0000008", + "prompt": "0.0000009", + "completion": "0.0000009", "request": "0", "image": "0", "web_search": "0", @@ -2371,7 +2479,7 @@ }, "top_provider": { "context_length": 128000, - "max_completion_tokens": 128000, + "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, @@ -3204,8 +3312,6 @@ "logprobs", "top_logprobs", "seed", - "tools", - "tool_choice", "structured_outputs" ] }, @@ -3344,62 +3450,15 @@ }, "per_request_limits": null, "supported_parameters": [ - "tools", - "tool_choice", "max_tokens", "temperature", - "top_p", + "stop", "reasoning", "include_reasoning", - "top_k", - "stop" - ] - }, - { - "id": "anthropic/claude-3.7-sonnet:thinking", - "hugging_face_id": "", - "name": "Anthropic: Claude 3.7 Sonnet (thinking)", - "created": 1740422110, - "description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and extended, step-by-step processing for complex tasks. The model demonstrates notable improvements in coding, particularly in front-end development and full-stack updates, and excels in agentic workflows, where it can autonomously navigate multi-step processes. \n\nClaude 3.7 Sonnet maintains performance parity with its predecessor in standard mode while offering an extended reasoning mode for enhanced accuracy in math, coding, and instruction-following tasks.\n\nRead more at the [blog post here](https://www.anthropic.com/news/claude-3-7-sonnet)", - "context_length": 200000, - "architecture": { - "modality": "text+image->text", - "input_modalities": [ - "text", - "image" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Claude", - "instruct_type": null - }, - "pricing": { - "prompt": "0.000003", - "completion": "0.000015", - "request": "0", - "image": "0.0048", - "web_search": "0", - "internal_reasoning": "0", - "input_cache_read": "0.0000003", - "input_cache_write": "0.00000375" - }, - "top_provider": { - "context_length": 200000, - "max_completion_tokens": 64000, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ "tools", "tool_choice", - "max_tokens", - "temperature", "top_p", - "reasoning", - "include_reasoning", - "top_k", - "stop" + "top_k" ] }, { @@ -3447,6 +3506,51 @@ "tool_choice" ] }, + { + "id": "anthropic/claude-3.7-sonnet:thinking", + "hugging_face_id": "", + "name": "Anthropic: Claude 3.7 Sonnet (thinking)", + "created": 1740422110, + "description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and extended, step-by-step processing for complex tasks. The model demonstrates notable improvements in coding, particularly in front-end development and full-stack updates, and excels in agentic workflows, where it can autonomously navigate multi-step processes. \n\nClaude 3.7 Sonnet maintains performance parity with its predecessor in standard mode while offering an extended reasoning mode for enhanced accuracy in math, coding, and instruction-following tasks.\n\nRead more at the [blog post here](https://www.anthropic.com/news/claude-3-7-sonnet)", + "context_length": 200000, + "architecture": { + "modality": "text+image->text", + "input_modalities": [ + "text", + "image" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Claude", + "instruct_type": null + }, + "pricing": { + "prompt": "0.000003", + "completion": "0.000015", + "request": "0", + "image": "0.0048", + "web_search": "0", + "internal_reasoning": "0", + "input_cache_read": "0.0000003", + "input_cache_write": "0.00000375" + }, + "top_provider": { + "context_length": 200000, + "max_completion_tokens": 128000, + "is_moderated": true + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop", + "reasoning", + "include_reasoning", + "tools", + "tool_choice" + ] + }, { "id": "perplexity/r1-1776", "hugging_face_id": "perplexity-ai/r1-1776", @@ -4019,12 +4123,12 @@ "stop", "frequency_penalty", "presence_penalty", + "seed", "top_k", + "min_p", "repetition_penalty", "logit_bias", - "min_p", "response_format", - "seed", "logprobs", "top_logprobs" ] @@ -4299,12 +4403,12 @@ "stop", "frequency_penalty", "presence_penalty", - "repetition_penalty", - "response_format", - "top_k", "seed", + "top_k", "min_p", - "logit_bias" + "repetition_penalty", + "logit_bias", + "response_format" ] }, { @@ -4335,7 +4439,7 @@ }, "top_provider": { "context_length": 64000, - "max_completion_tokens": 64000, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -4348,12 +4452,12 @@ "stop", "frequency_penalty", "presence_penalty", + "seed", "top_k", + "min_p", "repetition_penalty", "logit_bias", - "min_p", - "response_format", - "seed" + "response_format" ] }, { @@ -4610,7 +4714,7 @@ "instruct_type": "deepseek-r1" }, "pricing": { - "prompt": "0.0000005", + "prompt": "0.00000045", "completion": "0.00000218", "request": "0", "image": "0", @@ -4634,13 +4738,13 @@ "presence_penalty", "seed", "top_k", + "min_p", "logit_bias", - "logprobs", "top_logprobs", - "repetition_penalty", "response_format", "structured_outputs", - "min_p", + "logprobs", + "repetition_penalty", "tools", "tool_choice" ] @@ -5146,13 +5250,13 @@ "frequency_penalty", "presence_penalty", "seed", + "top_k", + "min_p", + "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", - "top_k", - "min_p", - "repetition_penalty", "structured_outputs" ] }, @@ -5300,8 +5404,8 @@ "instruct_type": "deepseek-r1" }, "pricing": { - "prompt": "0.00000009", - "completion": "0.00000027", + "prompt": "0.0000002", + "completion": "0.0000002", "request": "0", "image": "0", "web_search": "0", @@ -6810,8 +6914,7 @@ "seed", "min_p", "logit_bias", - "top_logprobs", - "logprobs" + "top_logprobs" ] }, { @@ -6899,14 +7002,14 @@ "max_tokens", "temperature", "top_p", - "stop", + "top_k", + "seed", + "repetition_penalty", "frequency_penalty", "presence_penalty", - "seed", - "top_k", - "min_p", - "repetition_penalty", + "stop", "logit_bias", + "min_p", "response_format", "top_logprobs", "tools", @@ -7412,7 +7515,7 @@ "name": "Microsoft: Phi-3.5 Mini 128K Instruct", "created": 1724198400, "description": "Phi-3.5 models are lightweight, state-of-the-art open models. These models were trained with Phi-3 datasets that include both synthetic data and the filtered, publicly available websites data, with a focus on high quality and reasoning-dense properties. Phi-3.5 Mini uses 3.8B parameters, and is a dense decoder-only transformer model using the same tokenizer as [Phi-3 Mini](/models/microsoft/phi-3-mini-128k-instruct).\n\nThe models underwent a rigorous enhancement process, incorporating both supervised fine-tuning, proximal policy optimization, and direct preference optimization to ensure precise instruction adherence and robust safety measures. When assessed against benchmarks that test common sense, language understanding, math, code, long context and logical reasoning, Phi-3.5 models showcased robust and state-of-the-art performance among models with less than 13 billion parameters.", - "context_length": 131072, + "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ @@ -7425,15 +7528,15 @@ "instruct_type": "phi3" }, "pricing": { - "prompt": "0.00000003", - "completion": "0.00000009", + "prompt": "0.0000001", + "completion": "0.0000001", "request": "0", "image": "0", "web_search": "0", "internal_reasoning": "0" }, "top_provider": { - "context_length": 131072, + "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, @@ -7443,15 +7546,7 @@ "tool_choice", "max_tokens", "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" + "top_p" ] }, { @@ -7720,13 +7815,12 @@ "request": "0", "image": "0.003613", "web_search": "0", - "internal_reasoning": "0", - "input_cache_read": "0.00000125" + "internal_reasoning": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, - "is_moderated": true + "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ @@ -8151,7 +8245,7 @@ }, "top_provider": { "context_length": 131072, - "max_completion_tokens": 131072, + "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, @@ -8162,17 +8256,17 @@ "stop", "frequency_penalty", "presence_penalty", - "seed", + "repetition_penalty", + "response_format", "top_k", + "seed", + "min_p", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice", - "response_format", - "structured_outputs", - "repetition_penalty", - "min_p" + "structured_outputs" ] }, { @@ -8300,8 +8394,8 @@ "instruct_type": "gemma" }, "pricing": { - "prompt": "0.0000001", - "completion": "0.0000003", + "prompt": "0.0000008", + "completion": "0.0000008", "request": "0", "image": "0", "web_search": "0", @@ -8309,7 +8403,7 @@ }, "top_provider": { "context_length": 8192, - "max_completion_tokens": null, + "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, @@ -8324,10 +8418,7 @@ "repetition_penalty", "logit_bias", "min_p", - "response_format", - "seed", - "logprobs", - "top_logprobs" + "response_format" ] }, { @@ -8394,8 +8485,8 @@ "instruct_type": "gemma" }, "pricing": { - "prompt": "0.00000002", - "completion": "0.00000006", + "prompt": "0.0000002", + "completion": "0.0000002", "request": "0", "image": "0", "web_search": "0", @@ -8403,7 +8494,7 @@ }, "top_provider": { "context_length": 8192, - "max_completion_tokens": null, + "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, @@ -8414,14 +8505,11 @@ "stop", "frequency_penalty", "presence_penalty", - "seed", - "top_k", - "min_p", - "repetition_penalty", - "logit_bias", "response_format", "top_logprobs", - "logprobs" + "logprobs", + "logit_bias", + "seed" ] }, { @@ -8897,7 +8985,7 @@ "name": "Microsoft: Phi-3 Medium 128K Instruct", "created": 1716508800, "description": "Phi-3 128K Medium is a powerful 14-billion parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. In the MMLU-Pro eval, the model even comes close to a Llama3 70B level of performance.\n\nFor 4k context length, try [Phi-3 Medium 4K](/models/microsoft/phi-3-medium-4k-instruct).", - "context_length": 131072, + "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ @@ -8910,15 +8998,15 @@ "instruct_type": "phi3" }, "pricing": { - "prompt": "0.0000001", - "completion": "0.0000003", + "prompt": "0.000001", + "completion": "0.000001", "request": "0", "image": "0", "web_search": "0", "internal_reasoning": "0" }, "top_provider": { - "context_length": 131072, + "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, @@ -8928,15 +9016,7 @@ "tool_choice", "max_tokens", "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" + "top_p" ] }, { @@ -8984,52 +9064,6 @@ "seed" ] }, - { - "id": "deepseek/deepseek-coder", - "hugging_face_id": "deepseek-ai/DeepSeek-Coder-V2-Instruct", - "name": "DeepSeek-Coder-V2", - "created": 1715644800, - "description": "DeepSeek-Coder-V2, an open-source Mixture-of-Experts (MoE) code language model. It is further pre-trained from an intermediate checkpoint of DeepSeek-V2 with additional 6 trillion tokens.\n\nThe original V1 model was trained from scratch on 2T tokens, with a composition of 87% code and 13% natural language in both English and Chinese. It was pre-trained on project-level code corpus by employing a extra fill-in-the-blank task.", - "context_length": 128000, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Other", - "instruct_type": null - }, - "pricing": { - "prompt": "0.00000004", - "completion": "0.00000012", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 128000, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "google/gemini-flash-1.5", "hugging_face_id": null, @@ -9282,52 +9316,6 @@ "structured_outputs" ] }, - { - "id": "allenai/olmo-7b-instruct", - "hugging_face_id": "allenai/OLMo-7B-Instruct", - "name": "OLMo 7B Instruct", - "created": 1715299200, - "description": "OLMo 7B Instruct by the Allen Institute for AI is a model finetuned for question answering. It demonstrates **notable performance** across multiple benchmarks including TruthfulQA and ToxiGen.\n\n**Open Source**: The model, its code, checkpoints, logs are released under the [Apache 2.0 license](https://choosealicense.com/licenses/apache-2.0).\n\n- [Core repo (training, inference, fine-tuning etc.)](https://github.com/allenai/OLMo)\n- [Evaluation code](https://github.com/allenai/OLMo-Eval)\n- [Further fine-tuning code](https://github.com/allenai/open-instruct)\n- [Paper](https://arxiv.org/abs/2402.00838)\n- [Technical blog post](https://blog.allenai.org/olmo-open-language-model-87ccfc95f580)\n- [W&B Logs](https://wandb.ai/ai2-llm/OLMo-7B/reports/OLMo-7B--Vmlldzo2NzQyMzk5)", - "context_length": 2048, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Other", - "instruct_type": "zephyr" - }, - "pricing": { - "prompt": "0.00000008", - "completion": "0.00000024", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 2048, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "neversleep/llama-3-lumimaid-8b", "hugging_face_id": "NeverSleep/Llama-3-Lumimaid-8B-v0.1", @@ -9542,8 +9530,8 @@ "instruct_type": "mistral" }, "pricing": { - "prompt": "0.0000004", - "completion": "0.0000012", + "prompt": "0.0000009", + "completion": "0.0000009", "request": "0", "image": "0", "web_search": "0", @@ -9610,13 +9598,13 @@ "max_tokens", "temperature", "top_p", - "presence_penalty", - "frequency_penalty", - "repetition_penalty", - "top_k", "stop", + "frequency_penalty", + "presence_penalty", "seed", + "top_k", "min_p", + "repetition_penalty", "logit_bias", "response_format" ] @@ -9837,7 +9825,7 @@ }, "top_provider": { "context_length": 4096, - "max_completion_tokens": null, + "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, @@ -10682,9 +10670,7 @@ "logit_bias", "min_p", "response_format", - "seed", - "logprobs", - "top_logprobs" + "seed" ] }, { @@ -11579,13 +11565,13 @@ "max_tokens", "temperature", "top_p", - "presence_penalty", - "frequency_penalty", - "repetition_penalty", - "top_k", "stop", + "frequency_penalty", + "presence_penalty", "seed", + "top_k", "min_p", + "repetition_penalty", "logit_bias", "response_format", "top_a" From 559fcbf3b9b41a09ea8bd0a4818628ca707140e8 Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Thu, 29 May 2025 19:42:49 +0200 Subject: [PATCH 2/2] fixed non-streaming responses --- .gitignore | 6 +++--- router/proxy.py | 9 ++++++++- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/.gitignore b/.gitignore index 359006b7..51b3e357 100644 --- a/.gitignore +++ b/.gitignore @@ -5,7 +5,7 @@ wallet.sqlite3 # Development .notes -.keys.db -.wallet.sqlite3 -.models.json +.*keys.db +.*wallet.sqlite3 +.*models.json compose.override.yml diff --git a/router/proxy.py b/router/proxy.py index 91f0c0c2..6552bfb3 100644 --- a/router/proxy.py +++ b/router/proxy.py @@ -189,10 +189,17 @@ async def proxy( key, response_json, session ) response_json["cost"] = cost_data + + response_headers = dict(response.headers) + + # Remove Transfer-Encoding header to avoid conflict with Content-Length header in common nginx setups + if "transfer-encoding" in response_headers: + del response_headers["transfer-encoding"] + return Response( content=json.dumps(response_json).encode(), status_code=response.status_code, - headers=dict(response.headers), + headers=response_headers, media_type="application/json", ) except json.JSONDecodeError as e: