Fix models sort (by output cost), filter non-chat models, add compression variants
This commit is contained in:
+27
-9
@@ -228,7 +228,7 @@ class Assistant(commands.Cog):
|
|||||||
|
|
||||||
@assistant.command(name="models")
|
@assistant.command(name="models")
|
||||||
async def assistant_models(self, ctx: commands.Context):
|
async def assistant_models(self, ctx: commands.Context):
|
||||||
"""List available models sorted by cost (cheapest first)."""
|
"""List available chat models sorted by cost (cheapest first)."""
|
||||||
# Pricing per 1M tokens (EUR) from GreenPT docs
|
# Pricing per 1M tokens (EUR) from GreenPT docs
|
||||||
MODEL_PRICING = {
|
MODEL_PRICING = {
|
||||||
"deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35},
|
"deepseek-v4-flash-0731": {"input": 0.14, "output": 0.35},
|
||||||
@@ -253,6 +253,23 @@ class Assistant(commands.Cog):
|
|||||||
"llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10},
|
"llama-3.3-70b-instruct": {"input": 1.10, "output": 1.10},
|
||||||
"mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00},
|
"mistral-medium-3.5-128b": {"input": 1.80, "output": 9.00},
|
||||||
"kimi-k3": {"input": 3.30, "output": 16.50},
|
"kimi-k3": {"input": 3.30, "output": 16.50},
|
||||||
|
# Compression variants (same price as glm-5.2)
|
||||||
|
"glm-5.2-caveman": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-caveman-lite": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-caveman-ultra": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-honey": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-honey-lite": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-honey-ultra": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-ponytail": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-ponytail-lite": {"input": 1.10, "output": 4.40},
|
||||||
|
"glm-5.2-ponytail-ultra": {"input": 1.10, "output": 4.40},
|
||||||
|
}
|
||||||
|
|
||||||
|
# Models that aren't usable for chat completions
|
||||||
|
NON_CHAT_MODELS = {
|
||||||
|
"green-embedding", "green-embeddings", "green-rerank",
|
||||||
|
"green-s", "green-s-pro", "bge-multilingual-gemma2",
|
||||||
|
"qwen3-embedding-8b",
|
||||||
}
|
}
|
||||||
|
|
||||||
api_key = await self.config.api_key()
|
api_key = await self.config.api_key()
|
||||||
@@ -280,23 +297,24 @@ class Assistant(commands.Cog):
|
|||||||
await ctx.send("No models returned by the API.")
|
await ctx.send("No models returned by the API.")
|
||||||
return
|
return
|
||||||
|
|
||||||
# Build list with pricing, sorted by combined cost (input + output)
|
# Build list with pricing, sorted by output cost (most relevant for chat)
|
||||||
current_model = await self.config.guild(ctx.guild).model()
|
current_model = await self.config.guild(ctx.guild).model()
|
||||||
lines = []
|
|
||||||
model_entries = []
|
model_entries = []
|
||||||
for m in models:
|
for m in models:
|
||||||
model_id = m["id"]
|
model_id = m["id"]
|
||||||
|
if model_id in NON_CHAT_MODELS:
|
||||||
|
continue
|
||||||
pricing = MODEL_PRICING.get(model_id)
|
pricing = MODEL_PRICING.get(model_id)
|
||||||
if pricing:
|
if pricing:
|
||||||
total = pricing["input"] + pricing["output"]
|
sort_key = pricing["output"]
|
||||||
model_entries.append((total, model_id, pricing))
|
model_entries.append((sort_key, model_id, pricing))
|
||||||
else:
|
else:
|
||||||
# Unknown pricing, put at the end
|
|
||||||
model_entries.append((9999, model_id, None))
|
model_entries.append((9999, model_id, None))
|
||||||
|
|
||||||
model_entries.sort(key=lambda x: x[0])
|
model_entries.sort(key=lambda x: (x[0], x[1]))
|
||||||
|
|
||||||
for total, model_id, pricing in model_entries:
|
lines = []
|
||||||
|
for sort_key, model_id, pricing in model_entries:
|
||||||
marker = " ◀" if model_id == current_model else ""
|
marker = " ◀" if model_id == current_model else ""
|
||||||
if pricing:
|
if pricing:
|
||||||
lines.append(
|
lines.append(
|
||||||
@@ -310,7 +328,7 @@ class Assistant(commands.Cog):
|
|||||||
description="\n".join(lines),
|
description="\n".join(lines),
|
||||||
color=await ctx.embed_color(),
|
color=await ctx.embed_color(),
|
||||||
)
|
)
|
||||||
embed.set_footer(text="Prices in EUR per 1M tokens. ◀ = current model.")
|
embed.set_footer(text="Prices in EUR per 1M tokens, sorted by output cost. ◀ = current model.")
|
||||||
await ctx.send(embed=embed)
|
await ctx.send(embed=embed)
|
||||||
|
|
||||||
@assistant.command(name="temp")
|
@assistant.command(name="temp")
|
||||||
|
|||||||
Reference in New Issue
Block a user