diff --git a/app/config/models.json b/app/config/models.json index bb3793cc..7215f69c 100644 --- a/app/config/models.json +++ b/app/config/models.json @@ -1336,6 +1336,890 @@ "cache_read_per_1m": 0.25 } }, + "MiniMaxAI/MiniMax-M2.7": { + "provider": "together", + "model_type": "llm", + "description": "MiniMaxAI MiniMax M2.7 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.3, + "output_per_1m": 1.2, + "cache_read_per_1m": 0.06 + } + }, + "MiniMaxAI/MiniMax-M3": { + "provider": "together", + "model_type": "llm", + "description": "MiniMaxAI MiniMax M3 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.3, + "output_per_1m": 1.2, + "cache_read_per_1m": 0.06 + } + }, + "NousResearch/Nous-Hermes-2-Mixtral-8x7B-DPO": { + "provider": "together", + "model_type": "llm", + "description": "NousResearch Nous Hermes 2 Mixtral 8x7B DPO via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 0.6 + } + }, + "Qwen/QwQ-32B": { + "provider": "together", + "model_type": "llm", + "description": "Qwen QwQ 32B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.2, + "output_per_1m": 1.2 + } + }, + "Qwen/Qwen2-1.5B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2 1.5B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.02, + "output_per_1m": 0.02 + } + }, + "Qwen/Qwen2-72B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2 72B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.9, + "output_per_1m": 0.9 + } + }, + "Qwen/Qwen2-VL-72B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2 VL 72B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.2, + "output_per_1m": 1.2 + } + }, + "Qwen/Qwen2.5-14B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 14B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.8, + "output_per_1m": 0.8 + } + }, + "Qwen/Qwen2.5-72B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 72B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.2, + "output_per_1m": 1.2 + } + }, + "Qwen/Qwen2.5-72B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 72B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.2, + "output_per_1m": 1.2 + } + }, + "Qwen/Qwen2.5-7B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 7B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.3, + "output_per_1m": 0.3 + } + }, + "Qwen/Qwen2.5-Coder-32B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 Coder 32B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.8, + "output_per_1m": 0.8 + } + }, + "Qwen/Qwen2.5-VL-72B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen2.5 VL 72B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.95, + "output_per_1m": 8.0 + } + }, + "Qwen/Qwen3-Coder-480B-A35B-Instruct-FP8": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 Coder 480B A35B Instruct FP8 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 2.0, + "output_per_1m": 2.0 + } + }, + "Qwen/Qwen3-Coder-Next-FP8": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 Coder Next FP8 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.5, + "output_per_1m": 1.2 + } + }, + "Qwen/Qwen3-Next-80B-A3B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 Next 80B A3B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.15, + "output_per_1m": 1.5 + } + }, + "Qwen/Qwen3-Next-80B-A3B-Thinking": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 Next 80B A3B Thinking via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.15, + "output_per_1m": 1.5 + } + }, + "Qwen/Qwen3-VL-32B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 VL 32B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.5, + "output_per_1m": 1.5 + } + }, + "Qwen/Qwen3-VL-8B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3 VL 8B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.18, + "output_per_1m": 0.68 + } + }, + "Qwen/Qwen3.5-397B-A17B": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.5 397B A17B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 3.6, + "cache_read_per_1m": 0.35 + } + }, + "Qwen/Qwen3.5-9B": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.5 9B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.17, + "output_per_1m": 0.25 + } + }, + "Qwen/Qwen3.6-Plus": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.6 Plus via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.5, + "output_per_1m": 3.0 + } + }, + "Qwen/Qwen3.7-Max": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.7 Max via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.5, + "output_per_1m": 4.5, + "cache_read_per_1m": 0.3 + } + }, + "Qwen/Qwen3.7-Plus": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.7 Plus via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.32, + "output_per_1m": 1.28 + } + }, + "Qwen/Qwen3.8-2.4T-A95B": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.8 2.4T A95B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 2.0, + "output_per_1m": 6.0, + "cache_read_per_1m": 0.25 + } + }, + "Qwen/Qwen3.8-Flash": { + "provider": "together", + "model_type": "llm", + "description": "Qwen Qwen3.8 Flash via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.09, + "output_per_1m": 0.282 + } + }, + "arcee-ai/trinity-mini": { + "provider": "together", + "model_type": "llm", + "description": "arcee ai trinity mini via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.045, + "output_per_1m": 0.15 + } + }, + "arize-ai/qwen-2-1.5b-instruct": { + "provider": "together", + "model_type": "llm", + "description": "arize ai qwen 2 1.5b instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.1, + "output_per_1m": 0.1 + } + }, + "deepseek-ai/DeepSeek-R1-0528": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek R1 0528 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 3.0, + "output_per_1m": 7.0 + } + }, + "deepseek-ai/DeepSeek-R1-Distill-Llama-70B": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek R1 Distill Llama 70B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 2.0, + "output_per_1m": 2.0 + } + }, + "deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek R1 Distill Qwen 1.5B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.18, + "output_per_1m": 0.18 + } + }, + "deepseek-ai/DeepSeek-R1-Distill-Qwen-14B": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek R1 Distill Qwen 14B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.6, + "output_per_1m": 1.6 + } + }, + "deepseek-ai/DeepSeek-V3": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek V3 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.25, + "output_per_1m": 1.25 + } + }, + "deepseek-ai/DeepSeek-V3.1": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek V3.1 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 1.7 + } + }, + "deepseek-ai/DeepSeek-V4-Flash-0731": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek V4 Flash 0731 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.14, + "output_per_1m": 0.28, + "cache_read_per_1m": 0.03 + } + }, + "deepseek-ai/DeepSeek-V4-Pro-0813": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek V4 Pro 0813 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.32, + "output_per_1m": 3.96, + "cache_read_per_1m": 0.13 + } + }, + "deepseek-ai/DeepSeek-V4.1-Flash": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai DeepSeek V4.1 Flash via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.3, + "output_per_1m": 1.2, + "cache_read_per_1m": 0.006 + } + }, + "deepseek-ai/deepseek-coder-33b-instruct": { + "provider": "together", + "model_type": "llm", + "description": "deepseek ai deepseek coder 33b instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.8, + "output_per_1m": 0.8 + } + }, + "google/gemma-2-27b-it": { + "provider": "together", + "model_type": "llm", + "description": "google gemma 2 27b it via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.8, + "output_per_1m": 0.8 + } + }, + "google/gemma-4-31B-it": { + "provider": "together", + "model_type": "llm", + "description": "google gemma 4 31B it via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.39, + "output_per_1m": 0.97 + } + }, + "meta-llama/Llama-3-8b-chat-hf": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 3 8b chat hf via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 0.2 + } + }, + "meta-llama/Llama-3.1-405B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 3.1 405B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 3.5, + "output_per_1m": 3.5 + } + }, + "meta-llama/Llama-3.2-1B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 3.2 1B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.06, + "output_per_1m": 0.06 + } + }, + "meta-llama/Llama-3.2-3B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 3.2 3B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.06, + "output_per_1m": 0.06 + } + }, + "meta-llama/Llama-3.3-70B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 3.3 70B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.04, + "output_per_1m": 1.04 + } + }, + "meta-llama/Llama-4-Scout-17B-16E-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Llama 4 Scout 17B 16E Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.18, + "output_per_1m": 0.59 + } + }, + "meta-llama/Meta-Llama-3-70B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Meta Llama 3 70B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.88, + "output_per_1m": 0.88 + } + }, + "meta-llama/Meta-Llama-3-8B-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Meta Llama 3 8B Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 0.2 + } + }, + "meta-llama/Meta-Llama-3.1-70B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Meta Llama 3.1 70B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.88, + "output_per_1m": 0.88 + } + }, + "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + "description": "meta llama Meta Llama 3.1 8B Instruct Turbo via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.18, + "output_per_1m": 0.18 + } + }, + "meta-models/Muse-Glimmer-30B": { + "provider": "together", + "model_type": "llm", + "description": "meta models Muse Glimmer 30B via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.35, + "output_per_1m": 1.5, + "cache_read_per_1m": 0.04 + } + }, + "mistralai/Ministral-3-14B-Instruct-2512": { + "provider": "together", + "model_type": "llm", + "description": "mistralai Ministral 3 14B Instruct 2512 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 0.2 + } + }, + "mistralai/Mistral-7B-Instruct-v0.1": { + "provider": "together", + "model_type": "llm", + "description": "mistralai Mistral 7B Instruct v0.1 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 0.2 + } + }, + "mistralai/Mistral-7B-Instruct-v0.3": { + "provider": "together", + "model_type": "llm", + "description": "mistralai Mistral 7B Instruct v0.3 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 0.2 + } + }, + "mistralai/Mistral-Small-24B-Instruct-2501": { + "provider": "together", + "model_type": "llm", + "description": "mistralai Mistral Small 24B Instruct 2501 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.1, + "output_per_1m": 0.3 + } + }, + "mistralai/Mixtral-8x7B-Instruct-v0.1": { + "provider": "together", + "model_type": "llm", + "description": "mistralai Mixtral 8x7B Instruct v0.1 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 0.6 + } + }, + "moonshotai/Kimi-K2-Instruct": { + "provider": "together", + "model_type": "llm", + "description": "moonshotai Kimi K2 Instruct via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.0, + "output_per_1m": 3.0 + } + }, + "moonshotai/Kimi-K2.5-fp4": { + "provider": "together", + "model_type": "llm", + "description": "moonshotai Kimi K2.5 fp4 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.5, + "output_per_1m": 2.8 + } + }, + "moonshotai/Kimi-K2.6": { + "provider": "together", + "model_type": "llm", + "description": "moonshotai Kimi K2.6 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.2, + "output_per_1m": 4.5, + "cache_read_per_1m": 0.2 + } + }, + "moonshotai/Kimi-K2.7-Code": { + "provider": "together", + "model_type": "llm", + "description": "moonshotai Kimi K2.7 Code via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.95, + "output_per_1m": 4.0, + "cache_read_per_1m": 0.19 + } + }, + "moonshotai/Kimi-K3": { + "provider": "together", + "model_type": "llm", + "description": "moonshotai Kimi K3 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 3.0, + "output_per_1m": 15.0, + "cache_read_per_1m": 0.3 + } + }, + "nvidia/Llama-3.1-Nemotron-70B-Instruct-HF": { + "provider": "together", + "model_type": "llm", + "description": "nvidia Llama 3.1 Nemotron 70B Instruct HF via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.88, + "output_per_1m": 0.88 + } + }, + "nvidia/NVIDIA-Nemotron-Nano-9B-v2": { + "provider": "together", + "model_type": "llm", + "description": "nvidia NVIDIA Nemotron Nano 9B v2 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.06, + "output_per_1m": 0.25 + } + }, + "nvidia/nemotron-3-ultra-550b-a55b": { + "provider": "together", + "model_type": "llm", + "description": "nvidia nemotron 3 ultra 550b a55b via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 3.6, + "cache_read_per_1m": 0.2 + } + }, + "openai/gpt-oss-120b": { + "provider": "together", + "model_type": "llm", + "description": "openai gpt oss 120b via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.15, + "output_per_1m": 0.6 + } + }, + "openai/gpt-oss-20b": { + "provider": "together", + "model_type": "llm", + "description": "openai gpt oss 20b via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.05, + "output_per_1m": 0.2 + } + }, + "thinkingmachines/Inkling": { + "provider": "together", + "model_type": "llm", + "description": "thinkingmachines Inkling via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.0, + "output_per_1m": 4.05, + "cache_read_per_1m": 0.17 + } + }, + "together/Tev1-4B-experimental": { + "provider": "together", + "model_type": "llm", + "description": "together Tev1 4B experimental via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.042, + "cache_read_per_1m": 0.042 + } + }, + "zai-org/GLM-4.5-Air-FP8": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 4.5 Air FP8 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.2, + "output_per_1m": 1.1 + } + }, + "zai-org/GLM-4.6": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 4.6 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.6, + "output_per_1m": 2.2 + } + }, + "zai-org/GLM-4.7": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 4.7 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.45, + "output_per_1m": 2.0 + } + }, + "zai-org/GLM-5": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 5 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.0, + "output_per_1m": 3.2 + } + }, + "zai-org/GLM-5.1": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 5.1 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.4, + "output_per_1m": 4.4, + "cache_read_per_1m": 0.26 + } + }, + "zai-org/GLM-5.2": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 5.2 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.4, + "output_per_1m": 4.4, + "cache_read_per_1m": 0.26 + } + }, + "zai-org/GLM-5.3": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 5.3 via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 1.4, + "output_per_1m": 4.4, + "cache_read_per_1m": 0.26 + } + }, + "zai-org/GLM-5.3-Flash": { + "provider": "together", + "model_type": "llm", + "description": "zai org GLM 5.3 Flash via Together", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.15, + "output_per_1m": 0.5, + "cache_read_per_1m": 0.03 + } + }, + "jev-1.13.0": { + "provider": "typesafe", + "model_type": "llm", + "description": "jev 1.13.0 via TypeSafe", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.042 + } + }, + "jev-latest": { + "provider": "typesafe", + "model_type": "llm", + "description": "jev latest via TypeSafe", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.042 + } + }, + "jev-preview": { + "provider": "typesafe", + "model_type": "llm", + "description": "jev preview via TypeSafe", + "pricing": { + "source": "litellm_import", + "usage_kind": "llm", + "input_per_1m": 0.042 + } + }, "google-speech-v2": { "provider": "google", "model_type": "stt", diff --git a/app/models/enums.py b/app/models/enums.py index 020acecf..dd4519f0 100644 --- a/app/models/enums.py +++ b/app/models/enums.py @@ -145,6 +145,7 @@ class ModelProvider(str, enum.Enum): MISTRAL = "mistral" META = "meta" TOGETHER = "together" + TYPESAFE = "typesafe" PERPLEXITY = "perplexity" AZURE = "azure" AWS = "aws" diff --git a/app/services/ai/llm_service.py b/app/services/ai/llm_service.py index 95e64488..e310194e 100644 --- a/app/services/ai/llm_service.py +++ b/app/services/ai/llm_service.py @@ -45,6 +45,8 @@ "groq": "groq", "xai": "xai", "fireworks": "fireworks_ai", + "together": "together_ai", + "typesafe": "typesafe", "sarvam": "sarvam", } diff --git a/app/services/judge_alignment/model_catalog.py b/app/services/judge_alignment/model_catalog.py index 6fb7416f..4c1b130c 100644 --- a/app/services/judge_alignment/model_catalog.py +++ b/app/services/judge_alignment/model_catalog.py @@ -43,6 +43,7 @@ ModelProvider.MISTRAL.value, ModelProvider.META.value, ModelProvider.TOGETHER.value, + ModelProvider.TYPESAFE.value, ModelProvider.PERPLEXITY.value, ModelProvider.AZURE.value, ModelProvider.AWS.value, @@ -51,6 +52,15 @@ ModelProvider.SARVAM.value, } +# Judge/evaluator LLM providers that are not wired into live voice pipelines. +_LLM_JUDGE_ONLY_PROVIDERS = frozenset( + { + ModelProvider.TYPESAFE.value, + } +) + +_LLM_VOICE_CAPABLE_PROVIDERS = _LLM_CAPABLE_PROVIDERS - _LLM_JUDGE_ONLY_PROVIDERS + # Voice-platform integrations that also expose LLM models (credentials # live in the Integration table rather than AIProvider). _INTEGRATION_LLM_PLATFORMS = { @@ -70,6 +80,7 @@ def _provider_label(provider_value: str) -> str: "mistral": "Mistral", "meta": "Meta", "together": "Together", + "typesafe": "TypeSafe", "perplexity": "Perplexity", "azure": "Azure", "aws": "AWS", diff --git a/docs-fumadocs/content/docs/(docs)/integrations/index.mdx b/docs-fumadocs/content/docs/(docs)/integrations/index.mdx index 58b61a93..ec914516 100644 --- a/docs-fumadocs/content/docs/(docs)/integrations/index.mdx +++ b/docs-fumadocs/content/docs/(docs)/integrations/index.mdx @@ -56,6 +56,7 @@ This section focuses on platform-level integration guides for teams connecting e { name: 'Mistral', logo: '/mistral.svg' }, { name: 'Meta', logo: '/metaai.png' }, { name: 'Together', logo: '/togetherai.svg' }, + { name: 'TypeSafe', logo: '/typesafe.png' }, { name: 'Perplexity', logo: '/perplexity-ai.svg' }, { name: 'Azure', logo: '/azureai.png' }, { name: 'AWS', logo: '/AWS_logo.png' }, diff --git a/frontend/public/typesafe.png b/frontend/public/typesafe.png new file mode 100644 index 00000000..9d4a8cea Binary files /dev/null and b/frontend/public/typesafe.png differ diff --git a/frontend/src/components/providers/ProviderModelPicker.tsx b/frontend/src/components/providers/ProviderModelPicker.tsx index 81021841..e723c0c0 100644 --- a/frontend/src/components/providers/ProviderModelPicker.tsx +++ b/frontend/src/components/providers/ProviderModelPicker.tsx @@ -52,6 +52,7 @@ const PROVIDER_LABELS: Record = { mistral: 'Mistral', meta: 'Meta', together: 'Together', + typesafe: 'TypeSafe', perplexity: 'Perplexity', azure: 'Azure', aws: 'AWS', diff --git a/frontend/src/config/llmGenerationParams.ts b/frontend/src/config/llmGenerationParams.ts index 189cb8e9..277b2c25 100644 --- a/frontend/src/config/llmGenerationParams.ts +++ b/frontend/src/config/llmGenerationParams.ts @@ -107,6 +107,7 @@ const NO_TOP_K = new Set([ 'mistral', 'meta', 'together', + 'typesafe', 'perplexity', 'aws', ]) diff --git a/frontend/src/config/providers.ts b/frontend/src/config/providers.ts index 995f4091..e2070b2f 100644 --- a/frontend/src/config/providers.ts +++ b/frontend/src/config/providers.ts @@ -57,6 +57,11 @@ export const MODEL_PROVIDER_CONFIG: Record = { logo: '/togetherai.svg', description: 'Hosted open-source models via Together', }, + [ModelProvider.TYPESAFE]: { + label: 'TypeSafe', + logo: '/typesafe.png', + description: 'Jev decision classifiers via TypeSafe', + }, [ModelProvider.PERPLEXITY]: { label: 'Perplexity', logo: '/perplexity-ai.svg', diff --git a/frontend/src/pages/configurations/Integrations.tsx b/frontend/src/pages/configurations/Integrations.tsx index d492552f..d9d8dfd6 100644 --- a/frontend/src/pages/configurations/Integrations.tsx +++ b/frontend/src/pages/configurations/Integrations.tsx @@ -36,6 +36,7 @@ const AI_INTEGRATION_PROVIDERS: ModelProvider[] = [ ModelProvider.MISTRAL, ModelProvider.META, ModelProvider.TOGETHER, + ModelProvider.TYPESAFE, ModelProvider.PERPLEXITY, ModelProvider.AZURE, ModelProvider.AWS, diff --git a/frontend/src/types/api.ts b/frontend/src/types/api.ts index db86ed9b..4cf0ee23 100644 --- a/frontend/src/types/api.ts +++ b/frontend/src/types/api.ts @@ -395,6 +395,7 @@ export enum ModelProvider { MISTRAL = 'mistral', META = 'meta', TOGETHER = 'together', + TYPESAFE = 'typesafe', PERPLEXITY = 'perplexity', AZURE = 'azure', AWS = 'aws', diff --git a/scripts/sync_pricing_catalog_from_litellm.py b/scripts/sync_pricing_catalog_from_litellm.py index af1131e2..de7ee316 100644 --- a/scripts/sync_pricing_catalog_from_litellm.py +++ b/scripts/sync_pricing_catalog_from_litellm.py @@ -26,6 +26,8 @@ "groq": "groq", "xai": "xai", "fireworks": "fireworks_ai", + "together": "together_ai", + "typesafe": "typesafe", "sarvam": "sarvam", "deepgram": "deepgram", "elevenlabs": "elevenlabs", @@ -90,6 +92,21 @@ FIREWORKS_LONG_FORM_PREFIX = "fireworks_ai/accounts/fireworks/models/" +TOGETHER_SKIP_MODES = frozenset( + { + "embedding", + "image_generation", + "audio_transcription", + "moderation", + "rerank", + } +) + +TEV1_CATALOG_NAME = "together/Tev1-4B-experimental" +TEV1_LITELLM_KEY = f"together_ai/{TEV1_CATALOG_NAME}" + +TYPESAFE_SKIP_MODES = TOGETHER_SKIP_MODES + def _azure_deployment_name(catalog_model: str) -> str: if catalog_model == "azure-openai-gpt4": @@ -168,6 +185,20 @@ def _litellm_candidates( catalog_name, ] ) + elif provider == "together" and model_type == "llm": + candidates.extend( + [ + f"together_ai/{catalog_name}", + catalog_name, + ] + ) + elif provider == "typesafe" and model_type == "llm": + candidates.extend( + [ + f"typesafe/{catalog_name}", + catalog_name, + ] + ) elif provider == "elevenlabs": candidates.extend([f"elevenlabs/{catalog_name}", catalog_name]) elif provider == "google" and catalog_name.endswith("-stt"): @@ -411,6 +442,253 @@ def import_missing_fireworks(*, remote: bool = True) -> Tuple[int, List[str]]: return len(new_entries), sorted(new_entries.keys()) +def _together_catalog_name(litellm_key: str, info: Dict[str, Any]) -> Optional[str]: + """Return models.json key for an importable Together chat model, else None.""" + mode = str(info.get("mode") or "").lower() + if mode in TOGETHER_SKIP_MODES: + return None + + if not litellm_key.startswith("together_ai/"): + return None + + slug = litellm_key[len("together_ai/") :] + if not slug: + return None + + if mode and mode not in {"chat", "completion"}: + return None + + has_llm_cost = _first_cost(info, "input_cost_per_token") or _first_cost( + info, "output_cost_per_token" + ) + if not has_llm_cost: + return None + + return slug + + +def _together_description(catalog_name: str) -> str: + return f"{catalog_name.replace('/', ' ').replace('-', ' ')} via Together" + + +def discover_missing_together_models( + model_cost: Dict[str, Dict[str, Any]], + existing_models: Dict[str, Any], +) -> Dict[str, Tuple[str, Dict[str, Any]]]: + """Map catalog_name -> (litellm_key, info) for Together models to import.""" + discovered: Dict[str, Tuple[str, Dict[str, Any]]] = {} + for litellm_key, info in model_cost.items(): + if litellm_key == "sample_spec" or not isinstance(info, dict): + continue + catalog_name = _together_catalog_name(litellm_key, info) + if not catalog_name or catalog_name in existing_models: + continue + discovered[catalog_name] = (litellm_key, info) + return discovered + + +def _insert_after_together_block( + models: Dict[str, Any], new_entries: Dict[str, Any] +) -> Dict[str, Any]: + if not new_entries: + return models + + keys = [key for key in models if not key.startswith("_")] + insert_at = 0 + for index, key in enumerate(keys): + cfg = models[key] + if isinstance(cfg, dict) and cfg.get("provider") == "together": + insert_at = index + 1 + + if insert_at == 0: + for index, key in enumerate(keys): + cfg = models[key] + if isinstance(cfg, dict) and cfg.get("provider") == "fireworks": + insert_at = index + 1 + + ordered_keys = keys[:insert_at] + sorted(new_entries.keys()) + keys[insert_at:] + ordered: Dict[str, Any] = {} + for key in ordered_keys: + if key in new_entries: + ordered[key] = new_entries[key] + else: + ordered[key] = models[key] + for key, value in models.items(): + if key.startswith("_"): + ordered[key] = value + return ordered + + +def _tev1_manual_entry() -> Dict[str, Any]: + return { + "provider": "together", + "model_type": "llm", + "description": "Tev1 4B Jev-style decision classifier on Qwen3.5 4B via Together", + "pricing": { + "source": "together.ai serverless listing", + "usage_kind": "llm", + "input_per_1m": 0.042, + "output_per_1m": 0, + }, + } + + +def import_missing_together(*, remote: bool = True) -> Tuple[int, List[str]]: + """Add current Together serverless chat models missing from models.json.""" + models = json.loads(MODELS_JSON.read_text(encoding="utf-8")) + model_cost = _load_model_cost(remote=remote) + missing = discover_missing_together_models(model_cost, models) + + new_entries: Dict[str, Any] = {} + for catalog_name in sorted(missing): + _litellm_key, info = missing[catalog_name] + micro = _convert_litellm_pricing(info, usage_kind="llm") + pricing = _micro_pricing_to_plan(micro, usage_kind="llm") + if not pricing.get("input_per_1m") and not pricing.get("output_per_1m"): + continue + new_entries[catalog_name] = { + "provider": "together", + "model_type": "llm", + "description": _together_description(catalog_name), + "pricing": pricing, + } + + if TEV1_CATALOG_NAME not in models and TEV1_CATALOG_NAME not in new_entries: + tev1_info = model_cost.get(TEV1_LITELLM_KEY) + if tev1_info: + micro = _convert_litellm_pricing(tev1_info, usage_kind="llm") + pricing = _micro_pricing_to_plan(micro, usage_kind="llm") + if pricing.get("input_per_1m") or pricing.get("output_per_1m"): + new_entries[TEV1_CATALOG_NAME] = { + "provider": "together", + "model_type": "llm", + "description": _tev1_manual_entry()["description"], + "pricing": pricing, + } + else: + new_entries[TEV1_CATALOG_NAME] = _tev1_manual_entry() + else: + new_entries[TEV1_CATALOG_NAME] = _tev1_manual_entry() + + if not new_entries: + return 0, [] + + updated_models = _insert_after_together_block(models, new_entries) + MODELS_JSON.write_text( + json.dumps(updated_models, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + return len(new_entries), sorted(new_entries.keys()) + + +def _typesafe_catalog_name(litellm_key: str, info: Dict[str, Any]) -> Optional[str]: + """Return models.json key for an importable TypeSafe chat model, else None.""" + mode = str(info.get("mode") or "").lower() + if mode in TYPESAFE_SKIP_MODES: + return None + + if not litellm_key.startswith("typesafe/"): + return None + + slug = litellm_key[len("typesafe/") :] + if not slug: + return None + + if mode and mode not in {"chat", "completion", "evaluation"}: + return None + + has_llm_cost = _first_cost(info, "input_cost_per_token") or _first_cost( + info, "output_cost_per_token" + ) + if not has_llm_cost: + return None + + return slug + + +def _typesafe_description(catalog_name: str) -> str: + return f"{catalog_name.replace('-', ' ')} via TypeSafe" + + +def discover_missing_typesafe_models( + model_cost: Dict[str, Dict[str, Any]], + existing_models: Dict[str, Any], +) -> Dict[str, Tuple[str, Dict[str, Any]]]: + """Map catalog_name -> (litellm_key, info) for TypeSafe models to import.""" + discovered: Dict[str, Tuple[str, Dict[str, Any]]] = {} + for litellm_key, info in model_cost.items(): + if litellm_key == "sample_spec" or not isinstance(info, dict): + continue + catalog_name = _typesafe_catalog_name(litellm_key, info) + if not catalog_name or catalog_name in existing_models: + continue + discovered[catalog_name] = (litellm_key, info) + return discovered + + +def _insert_after_typesafe_block( + models: Dict[str, Any], new_entries: Dict[str, Any] +) -> Dict[str, Any]: + if not new_entries: + return models + + keys = [key for key in models if not key.startswith("_")] + insert_at = 0 + for index, key in enumerate(keys): + cfg = models[key] + if isinstance(cfg, dict) and cfg.get("provider") == "typesafe": + insert_at = index + 1 + + if insert_at == 0: + for index, key in enumerate(keys): + cfg = models[key] + if isinstance(cfg, dict) and cfg.get("provider") == "together": + insert_at = index + 1 + + ordered_keys = keys[:insert_at] + sorted(new_entries.keys()) + keys[insert_at:] + ordered: Dict[str, Any] = {} + for key in ordered_keys: + if key in new_entries: + ordered[key] = new_entries[key] + else: + ordered[key] = models[key] + for key, value in models.items(): + if key.startswith("_"): + ordered[key] = value + return ordered + + +def import_missing_typesafe(*, remote: bool = True) -> Tuple[int, List[str]]: + """Add current TypeSafe chat models missing from models.json.""" + models = json.loads(MODELS_JSON.read_text(encoding="utf-8")) + model_cost = _load_model_cost(remote=remote) + missing = discover_missing_typesafe_models(model_cost, models) + + new_entries: Dict[str, Any] = {} + for catalog_name in sorted(missing): + _litellm_key, info = missing[catalog_name] + micro = _convert_litellm_pricing(info, usage_kind="llm") + pricing = _micro_pricing_to_plan(micro, usage_kind="llm") + if not pricing.get("input_per_1m") and not pricing.get("output_per_1m"): + continue + new_entries[catalog_name] = { + "provider": "typesafe", + "model_type": "llm", + "description": _typesafe_description(catalog_name), + "pricing": pricing, + } + + if not new_entries: + return 0, [] + + updated_models = _insert_after_typesafe_block(models, new_entries) + MODELS_JSON.write_text( + json.dumps(updated_models, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + return len(new_entries), sorted(new_entries.keys()) + + def _load_model_cost(*, remote: bool) -> Dict[str, Dict[str, Any]]: if remote: from litellm.litellm_core_utils.get_model_cost_map import get_model_cost_map @@ -553,6 +831,16 @@ def main() -> int: action="store_true", help="Add current Fireworks serverless chat models missing from models.json", ) + parser.add_argument( + "--import-missing-together", + action="store_true", + help="Add current Together serverless chat models missing from models.json", + ) + parser.add_argument( + "--import-missing-typesafe", + action="store_true", + help="Add current TypeSafe chat models missing from models.json", + ) args = parser.parse_args() if args.import_missing_fireworks: @@ -564,6 +852,24 @@ def main() -> int: for name in names: print(f" added: {name}", file=sys.stderr) + if args.import_missing_together: + imported, names = import_missing_together(remote=not args.local) + print( + f"together import: {imported} model(s) added to models.json", + file=sys.stderr, + ) + for name in names: + print(f" added: {name}", file=sys.stderr) + + if args.import_missing_typesafe: + imported, names = import_missing_typesafe(remote=not args.local) + print( + f"typesafe import: {imported} model(s) added to models.json", + file=sys.stderr, + ) + for name in names: + print(f" added: {name}", file=sys.stderr) + catalog, meta = build_catalog(remote=not args.local) payload = json.dumps(catalog, indent=2, sort_keys=True) + "\n" if args.stdout: diff --git a/tests/test_scripts/test_sync_together_import.py b/tests/test_scripts/test_sync_together_import.py new file mode 100644 index 00000000..aa17b409 --- /dev/null +++ b/tests/test_scripts/test_sync_together_import.py @@ -0,0 +1,104 @@ +"""Tests for Together model import filter in sync_pricing_catalog_from_litellm.""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[2] +SYNC_SCRIPT = REPO_ROOT / "scripts" / "sync_pricing_catalog_from_litellm.py" + + +def _load_sync_module(): + spec = importlib.util.spec_from_file_location( + "sync_pricing_catalog_from_litellm", SYNC_SCRIPT + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +@pytest.fixture(name="sync_mod") +def fixture_sync_mod(): + return _load_sync_module() + + +def _chat_cost(**overrides): + base = { + "mode": "chat", + "input_cost_per_token": 0.000000042, + "output_cost_per_token": 0.0, + } + base.update(overrides) + return base + + +def test_together_catalog_name_accepts_slash_slug(sync_mod): + key = "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo" + assert ( + sync_mod._together_catalog_name(key, _chat_cost()) + == "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo" + ) + + +def test_together_catalog_name_accepts_tev1(sync_mod): + key = "together_ai/together/Tev1-4B-experimental" + assert ( + sync_mod._together_catalog_name(key, _chat_cost()) + == "together/Tev1-4B-experimental" + ) + + +def test_together_catalog_name_skips_embeddings(sync_mod): + key = "together_ai/togethercomputer/m2-bert-80M-8k-retrieval" + info = { + "mode": "embedding", + "input_cost_per_token": 0.000000008, + "output_cost_per_token": 0, + } + assert sync_mod._together_catalog_name(key, info) is None + + +def test_discover_missing_together_models_skips_existing(sync_mod): + model_cost = { + "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": _chat_cost(), + "together_ai/together/Tev1-4B-experimental": _chat_cost(), + "together_ai/togethercomputer/m2-bert-80M-8k-retrieval": { + "mode": "embedding", + "input_cost_per_token": 0.000000008, + }, + } + existing = { + "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + }, + } + discovered = sync_mod.discover_missing_together_models(model_cost, existing) + assert set(discovered) == {"together/Tev1-4B-experimental"} + + +def test_insert_after_together_block_preserves_existing(sync_mod): + models = { + "gpt-4o": {"provider": "openai", "model_type": "llm"}, + "deepseek-v4-pro": {"provider": "fireworks", "model_type": "llm"}, + "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo": { + "provider": "together", + "model_type": "llm", + }, + "google-speech-v2": {"provider": "google", "model_type": "stt"}, + } + new_entries = { + "together/Tev1-4B-experimental": {"provider": "together", "model_type": "llm"}, + } + ordered = sync_mod._insert_after_together_block(models, new_entries) + assert list(ordered.keys()) == [ + "gpt-4o", + "deepseek-v4-pro", + "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", + "together/Tev1-4B-experimental", + "google-speech-v2", + ] diff --git a/tests/test_scripts/test_sync_typesafe_import.py b/tests/test_scripts/test_sync_typesafe_import.py new file mode 100644 index 00000000..c94330e1 --- /dev/null +++ b/tests/test_scripts/test_sync_typesafe_import.py @@ -0,0 +1,73 @@ +"""Tests for TypeSafe model import filter in sync_pricing_catalog_from_litellm.""" + +from __future__ import annotations + +import importlib.util +from pathlib import Path + +import pytest + +REPO_ROOT = Path(__file__).resolve().parents[2] +SYNC_SCRIPT = REPO_ROOT / "scripts" / "sync_pricing_catalog_from_litellm.py" + + +def _load_sync_module(): + spec = importlib.util.spec_from_file_location( + "sync_pricing_catalog_from_litellm", SYNC_SCRIPT + ) + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +@pytest.fixture(name="sync_mod") +def fixture_sync_mod(): + return _load_sync_module() + + +def _chat_cost(**overrides): + base = { + "mode": "chat", + "input_cost_per_token": 0.00000015, + "output_cost_per_token": 0.0000006, + } + base.update(overrides) + return base + + +def test_typesafe_catalog_name_accepts_jev_slug(sync_mod): + key = "typesafe/jev-1.13.0" + assert sync_mod._typesafe_catalog_name(key, _chat_cost()) == "jev-1.13.0" + + +def test_typesafe_catalog_name_accepts_evaluation_mode(sync_mod): + key = "typesafe/jev-1.13.0" + info = { + "mode": "evaluation", + "input_cost_per_token": 0.000000042, + "output_cost_per_token": 0.0, + } + assert sync_mod._typesafe_catalog_name(key, info) == "jev-1.13.0" + + +def test_typesafe_catalog_name_skips_embeddings(sync_mod): + key = "typesafe/jev-embed-v1" + info = { + "mode": "embedding", + "input_cost_per_token": 0.000000008, + "output_cost_per_token": 0, + } + assert sync_mod._typesafe_catalog_name(key, info) is None + + +def test_discover_missing_typesafe_models_skips_existing(sync_mod): + model_cost = { + "typesafe/jev-1.13.0": _chat_cost(), + "typesafe/jev-1.12.0": _chat_cost(), + } + existing = { + "jev-1.13.0": {"provider": "typesafe", "model_type": "llm"}, + } + discovered = sync_mod.discover_missing_typesafe_models(model_cost, existing) + assert set(discovered) == {"jev-1.12.0"} diff --git a/tests/test_services/test_ai/test_llm_service.py b/tests/test_services/test_ai/test_llm_service.py index f9dcb666..f8aa26af 100644 --- a/tests/test_services/test_ai/test_llm_service.py +++ b/tests/test_services/test_ai/test_llm_service.py @@ -53,6 +53,22 @@ def test_litellm_model_name_maps_known_provider_prefixes(): LLMService._litellm_model_name(ModelProvider.FIREWORKS, "deepseek-v4-pro") == "fireworks_ai/accounts/fireworks/models/deepseek-v4-pro" ) + assert ( + LLMService._litellm_model_name( + ModelProvider.TOGETHER, "together/Tev1-4B-experimental" + ) + == "together_ai/together/Tev1-4B-experimental" + ) + assert ( + LLMService._litellm_model_name( + ModelProvider.TOGETHER, "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo" + ) + == "together_ai/meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo" + ) + assert ( + LLMService._litellm_model_name(ModelProvider.TYPESAFE, "jev-1.13.0") + == "typesafe/jev-1.13.0" + ) assert ( LLMService._litellm_model_name(ModelProvider.SARVAM, "sarvam-30b") == "sarvam/sarvam-30b" diff --git a/tests/test_services/test_ai/test_model_config_together.py b/tests/test_services/test_ai/test_model_config_together.py new file mode 100644 index 00000000..08859e91 --- /dev/null +++ b/tests/test_services/test_ai/test_model_config_together.py @@ -0,0 +1,10 @@ +"""Ensure Together LLM models are present in the model catalog.""" + +from app.models.database import ModelProvider +from app.services.ai.model_config_service import model_config_service + + +def test_together_llm_models_in_catalog(): + llm_models = model_config_service.get_models_by_type(ModelProvider.TOGETHER, "llm") + assert "together/Tev1-4B-experimental" in llm_models + assert "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo" in llm_models diff --git a/tests/test_services/test_ai/test_model_config_typesafe.py b/tests/test_services/test_ai/test_model_config_typesafe.py new file mode 100644 index 00000000..88f42c56 --- /dev/null +++ b/tests/test_services/test_ai/test_model_config_typesafe.py @@ -0,0 +1,9 @@ +"""Ensure TypeSafe LLM models are present in the model catalog.""" + +from app.models.database import ModelProvider +from app.services.ai.model_config_service import model_config_service + + +def test_typesafe_llm_models_in_catalog(): + llm_models = model_config_service.get_models_by_type(ModelProvider.TYPESAFE, "llm") + assert "jev-1.13.0" in llm_models diff --git a/tests/test_services/test_voice_agent/test_llm_voice_providers.py b/tests/test_services/test_voice_agent/test_llm_voice_providers.py index 999b95ab..75ae5286 100644 --- a/tests/test_services/test_voice_agent/test_llm_voice_providers.py +++ b/tests/test_services/test_voice_agent/test_llm_voice_providers.py @@ -1,6 +1,6 @@ """Tests for live voice pipeline LLM provider registry.""" -from app.services.judge_alignment.model_catalog import _LLM_CAPABLE_PROVIDERS +from app.services.judge_alignment.model_catalog import _LLM_VOICE_CAPABLE_PROVIDERS from app.services.voice_agent.llm_voice_providers import ( LLM_VOICE_PROVIDER_KEYS, get_llm_provider_registry, @@ -9,7 +9,7 @@ def test_voice_llm_registry_covers_configurable_providers(): - assert LLM_VOICE_PROVIDER_KEYS == _LLM_CAPABLE_PROVIDERS + assert LLM_VOICE_PROVIDER_KEYS == _LLM_VOICE_CAPABLE_PROVIDERS def _fake_get_service(name: str): return lambda **kwargs: (name, kwargs)