Fix models sort (by output cost), filter non-chat models, add compression variants

This commit is contained in:
2026-08-16 23:45:23 -05:00
parent 7eb96c6e65
commit d959d18f41
+27 -9
View File
@@ -228,7 +228,7 @@ class Assistant(commands.Cog):
@assistant.command(name="models")
async def assistant_models(self, ctx: commands.Context):
"""List available models sorted by cost (cheapest first)."""
"""List available chat models sorted by cost (cheapest first)."""
# Pricing per 1M tokens (EUR) from GreenPT docs
MODEL_PRICING = {
"deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35},
@@ -253,6 +253,23 @@ class Assistant(commands.Cog):
"llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10},
"mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00},
"kimi-k3": {"input": 3.30, "output": 16.50},
# Compression variants (same price as glm-5.2)
"glm-5.2-caveman": {"input": 1.10, "output": 4.40},
"glm-5.2-caveman-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-caveman-ultra": {"input": 1.10, "output": 4.40},
"glm-5.2-honey": {"input": 1.10, "output": 4.40},
"glm-5.2-honey-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-honey-ultra": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail-ultra": {"input": 1.10, "output": 4.40},
}
# Models that aren't usable for chat completions
NON_CHAT_MODELS = {
"green-embedding", "green-embeddings", "green-rerank",
"green-s", "green-s-pro", "bge-multilingual-gemma2",
"qwen3-embedding-8b",
}
api_key = await self.config.api_key()
@@ -280,23 +297,24 @@ class Assistant(commands.Cog):
await ctx.send("No models returned by the API.")
return
# Build list with pricing, sorted by combined cost (input + output)
# Build list with pricing, sorted by output cost (most relevant for chat)
current_model = await self.config.guild(ctx.guild).model()
lines = []
model_entries = []
for m in models:
model_id = m["id"]
if model_id in NON_CHAT_MODELS:
continue
pricing = MODEL_PRICING.get(model_id)
if pricing:
total = pricing["input"] + pricing["output"]
model_entries.append((total, model_id, pricing))
sort_key = pricing["output"]
model_entries.append((sort_key, model_id, pricing))
else:
# Unknown pricing, put at the end
model_entries.append((9999, model_id, None))
model_entries.sort(key=lambda x: x[0])
model_entries.sort(key=lambda x: (x[0], x[1]))
for total, model_id, pricing in model_entries:
lines = []
for sort_key, model_id, pricing in model_entries:
marker = " ◀" if model_id == current_model else ""
if pricing:
lines.append(
@@ -310,7 +328,7 @@ class Assistant(commands.Cog):
description="\n".join(lines),
color=await ctx.embed_color(),
)
embed.set_footer(text="Prices in EUR per 1M tokens. ◀ = current model.")
embed.set_footer(text="Prices in EUR per 1M tokens, sorted by output cost. ◀ = current model.")
await ctx.send(embed=embed)
@assistant.command(name="temp")