Fix models sort (by output cost), filter non-chat models, add compression variants
This commit is contained in:
+27
-9
@@ -228,7 +228,7 @@ class Assistant(commands.Cog):
|
||||
|
||||
@assistant.command(name="models")
|
||||
async def assistant_models(self, ctx: commands.Context):
|
||||
"""List available models sorted by cost (cheapest first)."""
|
||||
"""List available chat models sorted by cost (cheapest first)."""
|
||||
# Pricing per 1M tokens (EUR) from GreenPT docs
|
||||
MODEL_PRICING = {
|
||||
"deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35},
|
||||
@@ -253,6 +253,23 @@ class Assistant(commands.Cog):
|
||||
"llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10},
|
||||
"mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00},
|
||||
"kimi-k3": {"input": 3.30, "output": 16.50},
|
||||
# Compression variants (same price as glm-5.2)
|
||||
"glm-5.2-caveman": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-caveman-lite": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-caveman-ultra": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-honey": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-honey-lite": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-honey-ultra": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-ponytail": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-ponytail-lite": {"input": 1.10, "output": 4.40},
|
||||
"glm-5.2-ponytail-ultra": {"input": 1.10, "output": 4.40},
|
||||
}
|
||||
|
||||
# Models that aren't usable for chat completions
|
||||
NON_CHAT_MODELS = {
|
||||
"green-embedding", "green-embeddings", "green-rerank",
|
||||
"green-s", "green-s-pro", "bge-multilingual-gemma2",
|
||||
"qwen3-embedding-8b",
|
||||
}
|
||||
|
||||
api_key = await self.config.api_key()
|
||||
@@ -280,23 +297,24 @@ class Assistant(commands.Cog):
|
||||
await ctx.send("No models returned by the API.")
|
||||
return
|
||||
|
||||
# Build list with pricing, sorted by combined cost (input + output)
|
||||
# Build list with pricing, sorted by output cost (most relevant for chat)
|
||||
current_model = await self.config.guild(ctx.guild).model()
|
||||
lines = []
|
||||
model_entries = []
|
||||
for m in models:
|
||||
model_id = m["id"]
|
||||
if model_id in NON_CHAT_MODELS:
|
||||
continue
|
||||
pricing = MODEL_PRICING.get(model_id)
|
||||
if pricing:
|
||||
total = pricing["input"] + pricing["output"]
|
||||
model_entries.append((total, model_id, pricing))
|
||||
sort_key = pricing["output"]
|
||||
model_entries.append((sort_key, model_id, pricing))
|
||||
else:
|
||||
# Unknown pricing, put at the end
|
||||
model_entries.append((9999, model_id, None))
|
||||
|
||||
model_entries.sort(key=lambda x: x[0])
|
||||
model_entries.sort(key=lambda x: (x[0], x[1]))
|
||||
|
||||
for total, model_id, pricing in model_entries:
|
||||
lines = []
|
||||
for sort_key, model_id, pricing in model_entries:
|
||||
marker = " ◀" if model_id == current_model else ""
|
||||
if pricing:
|
||||
lines.append(
|
||||
@@ -310,7 +328,7 @@ class Assistant(commands.Cog):
|
||||
description="\n".join(lines),
|
||||
color=await ctx.embed_color(),
|
||||
)
|
||||
embed.set_footer(text="Prices in EUR per 1M tokens. ◀ = current model.")
|
||||
embed.set_footer(text="Prices in EUR per 1M tokens, sorted by output cost. ◀ = current model.")
|
||||
await ctx.send(embed=embed)
|
||||
|
||||
@assistant.command(name="temp")
|
||||
|
||||
Reference in New Issue
Block a user