diff --git a/assistant/assistant.py b/assistant/assistant.py index 4b578ad..4048351 100644 --- a/assistant/assistant.py +++ b/assistant/assistant.py @@ -228,7 +228,7 @@ class Assistant(commands.Cog): @assistant.command(name="models") async def assistant_models(self, ctx: commands.Context): - """List available models sorted by cost (cheapest first).""" + """List available chat models sorted by cost (cheapest first).""" # Pricing per 1M tokens (EUR) from GreenPT docs MODEL_PRICING = { "deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35}, @@ -253,6 +253,23 @@ class Assistant(commands.Cog): "llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10}, "mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00}, "kimi-k3": {"input": 3.30, "output": 16.50}, + # Compression variants (same price as glm-5.2) + "glm-5.2-caveman": {"input": 1.10, "output": 4.40}, + "glm-5.2-caveman-lite": {"input": 1.10, "output": 4.40}, + "glm-5.2-caveman-ultra": {"input": 1.10, "output": 4.40}, + "glm-5.2-honey": {"input": 1.10, "output": 4.40}, + "glm-5.2-honey-lite": {"input": 1.10, "output": 4.40}, + "glm-5.2-honey-ultra": {"input": 1.10, "output": 4.40}, + "glm-5.2-ponytail": {"input": 1.10, "output": 4.40}, + "glm-5.2-ponytail-lite": {"input": 1.10, "output": 4.40}, + "glm-5.2-ponytail-ultra": {"input": 1.10, "output": 4.40}, + } + + # Models that aren't usable for chat completions + NON_CHAT_MODELS = { + "green-embedding", "green-embeddings", "green-rerank", + "green-s", "green-s-pro", "bge-multilingual-gemma2", + "qwen3-embedding-8b", } api_key = await self.config.api_key() @@ -280,23 +297,24 @@ class Assistant(commands.Cog): await ctx.send("No models returned by the API.") return - # Build list with pricing, sorted by combined cost (input + output) + # Build list with pricing, sorted by output cost (most relevant for chat) current_model = await self.config.guild(ctx.guild).model() - lines = [] model_entries = [] for m in models: model_id = m["id"] + if model_id in NON_CHAT_MODELS: + continue pricing = MODEL_PRICING.get(model_id) if pricing: - total = pricing["input"] + pricing["output"] - model_entries.append((total, model_id, pricing)) + sort_key = pricing["output"] + model_entries.append((sort_key, model_id, pricing)) else: - # Unknown pricing, put at the end model_entries.append((9999, model_id, None)) - model_entries.sort(key=lambda x: x[0]) + model_entries.sort(key=lambda x: (x[0], x[1])) - for total, model_id, pricing in model_entries: + lines = [] + for sort_key, model_id, pricing in model_entries: marker = " ◀" if model_id == current_model else "" if pricing: lines.append( @@ -310,7 +328,7 @@ class Assistant(commands.Cog): description="\n".join(lines), color=await ctx.embed_color(), ) - embed.set_footer(text="Prices in EUR per 1M tokens. ◀ = current model.") + embed.set_footer(text="Prices in EUR per 1M tokens, sorted by output cost. ◀ = current model.") await ctx.send(embed=embed) @assistant.command(name="temp")