Fix models sort (by output cost), filter non-chat models, add compression variants

This commit is contained in:
2026-08-16 23:45:23 -05:00
parent 7eb96c6e65
commit d959d18f41
+27 -9
View File
@@ -228,7 +228,7 @@ class Assistant(commands.Cog):
@assistant.command(name="models") @assistant.command(name="models")
async def assistant_models(self, ctx: commands.Context): async def assistant_models(self, ctx: commands.Context):
"""List available models sorted by cost (cheapest first).""" """List available chat models sorted by cost (cheapest first)."""
# Pricing per 1M tokens (EUR) from GreenPT docs # Pricing per 1M tokens (EUR) from GreenPT docs
MODEL_PRICING = { MODEL_PRICING = {
"deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35}, "deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35},
@@ -253,6 +253,23 @@ class Assistant(commands.Cog):
"llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10}, "llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10},
"mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00}, "mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00},
"kimi-k3": {"input": 3.30, "output": 16.50}, "kimi-k3": {"input": 3.30, "output": 16.50},
# Compression variants (same price as glm-5.2)
"glm-5.2-caveman": {"input": 1.10, "output": 4.40},
"glm-5.2-caveman-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-caveman-ultra": {"input": 1.10, "output": 4.40},
"glm-5.2-honey": {"input": 1.10, "output": 4.40},
"glm-5.2-honey-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-honey-ultra": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail-lite": {"input": 1.10, "output": 4.40},
"glm-5.2-ponytail-ultra": {"input": 1.10, "output": 4.40},
}
# Models that aren't usable for chat completions
NON_CHAT_MODELS = {
"green-embedding", "green-embeddings", "green-rerank",
"green-s", "green-s-pro", "bge-multilingual-gemma2",
"qwen3-embedding-8b",
} }
api_key = await self.config.api_key() api_key = await self.config.api_key()
@@ -280,23 +297,24 @@ class Assistant(commands.Cog):
await ctx.send("No models returned by the API.") await ctx.send("No models returned by the API.")
return return
# Build list with pricing, sorted by combined cost (input + output) # Build list with pricing, sorted by output cost (most relevant for chat)
current_model = await self.config.guild(ctx.guild).model() current_model = await self.config.guild(ctx.guild).model()
lines = []
model_entries = [] model_entries = []
for m in models: for m in models:
model_id = m["id"] model_id = m["id"]
if model_id in NON_CHAT_MODELS:
continue
pricing = MODEL_PRICING.get(model_id) pricing = MODEL_PRICING.get(model_id)
if pricing: if pricing:
total = pricing["input"] + pricing["output"] sort_key = pricing["output"]
model_entries.append((total, model_id, pricing)) model_entries.append((sort_key, model_id, pricing))
else: else:
# Unknown pricing, put at the end
model_entries.append((9999, model_id, None)) model_entries.append((9999, model_id, None))
model_entries.sort(key=lambda x: x[0]) model_entries.sort(key=lambda x: (x[0], x[1]))
for total, model_id, pricing in model_entries: lines = []
for sort_key, model_id, pricing in model_entries:
marker = " ◀" if model_id == current_model else "" marker = " ◀" if model_id == current_model else ""
if pricing: if pricing:
lines.append( lines.append(
@@ -310,7 +328,7 @@ class Assistant(commands.Cog):
description="\n".join(lines), description="\n".join(lines),
color=await ctx.embed_color(), color=await ctx.embed_color(),
) )
embed.set_footer(text="Prices in EUR per 1M tokens. ◀ = current model.") embed.set_footer(text="Prices in EUR per 1M tokens, sorted by output cost. ◀ = current model.")
await ctx.send(embed=embed) await ctx.send(embed=embed)
@assistant.command(name="temp") @assistant.command(name="temp")