Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  

This file was deleted.

2 changes: 2 additions & 0 deletions providers/nano-gpt/models/Baichuan-M2.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
name = "Baichuan M2 32B Medical"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "baichuan"
release_date = "2025-08-19"
last_updated = "2025-08-19"
attachment = false
Expand All @@ -11,6 +12,7 @@ open_weights = false
[cost]
input = 15.73
output = 15.73
cache_read = 7.865

[limit]
context = 32_768
Expand Down
2 changes: 2 additions & 0 deletions providers/nano-gpt/models/Baichuan4-Air.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
name = "Baichuan 4 Air"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "baichuan"
release_date = "2025-08-19"
last_updated = "2025-08-19"
attachment = false
Expand All @@ -11,6 +12,7 @@ open_weights = false
[cost]
input = 0.157
output = 0.157
cache_read = 0.0785

[limit]
context = 32_768
Expand Down
2 changes: 2 additions & 0 deletions providers/nano-gpt/models/Baichuan4-Turbo.toml
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
name = "Baichuan 4 Turbo"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "baichuan"
release_date = "2025-08-19"
last_updated = "2025-08-19"
attachment = false
Expand All @@ -11,6 +12,7 @@ open_weights = false
[cost]
input = 2.42
output = 2.42
cache_read = 1.21

[limit]
context = 128_000
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@ attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.49299999999999994
output = 0.49299999999999994
input = 0.493
output = 0.493
cache_read = 0.2465

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@ attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 2.006
output = 2.006
cache_read = 1.003

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@ attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 2.006
output = 2.006
cache_read = 1.003

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@ attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.7989999999999999
output = 0.7989999999999999
input = 0.799
output = 0.799
cache_read = 0.3995

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,12 @@ attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.7989999999999999
output = 0.7989999999999999
input = 0.799
output = 0.799
cache_read = 0.3995

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
@@ -1,17 +1,18 @@
name = "Llama 3.05 Storybreaker Ministral 70b"
description = "Compact Mistral model for edge, latency-sensitive, and cost-efficient workloads"
family = "llama"
release_date = "2024-12-01"
release_date = "2024-01-01"
last_updated = "2024-12-01"
attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.49299999999999994
output = 0.49299999999999994
input = 0.493
output = 0.493
cache_read = 0.2465

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
@@ -1,17 +1,18 @@
name = "Nemotron Tenyxchat Storybreaker 70b"
description = "Nemotron model for efficient reasoning, coding, and specialized AI agents"
family = "nemotron"
release_date = "2024-12-01"
release_date = "2024-01-01"
last_updated = "2024-12-01"
attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.49299999999999994
output = 0.49299999999999994
input = 0.493
output = 0.493
cache_read = 0.2465

[limit]
context = 16_384
Expand Down
7 changes: 5 additions & 2 deletions providers/nano-gpt/models/GLM-4.6-Derestricted-v5.toml
Original file line number Diff line number Diff line change
@@ -1,16 +1,19 @@
name = "GLM 4.6 Derestricted v5"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "glm"
release_date = "2025-12-23"
last_updated = "2025-12-23"
attachment = false
reasoning = false
reasoning = true
tool_call = false
structured_output = false
open_weights = false
open_weights = true
reasoning_options = []

[cost]
input = 0.4
output = 1.5
cache_read = 0.2

[limit]
context = 131_072
Expand Down
Original file line number Diff line number Diff line change
@@ -1,17 +1,18 @@
name = "MN-LooseCannon-12B-v1"
description = "Mistral model for multilingual chat, reasoning, and tool-assisted workflows"
family = "mistral-nemo"
release_date = "2024-07-01"
release_date = "2024-01-01"
last_updated = "2024-07-01"
attachment = false
reasoning = false
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[cost]
input = 0.49299999999999994
output = 0.49299999999999994
input = 0.493
output = 0.493
cache_read = 0.2465

[limit]
context = 16_384
Expand Down
Original file line number Diff line number Diff line change
@@ -1,25 +1,26 @@
# Included in subscription
name = "Gemma 4 31B Claude 4.6 Opus Reasoning Distilled"
description = "O-series reasoning model for hard analysis, math, coding, and planning"
family = "claude"
release_date = "2026-05-01"
last_updated = "2026-05-01"
attachment = true
reasoning = true
reasoning_options = []
tool_call = false
structured_output = false
open_weights = false
open_weights = true
reasoning_options = []

[cost]
input = 0.306
output = 0.306
cache_read = 0.0306

[limit]
context = 262144
input = 262144
output = 16384
context = 262_144
input = 262_144
output = 16_384

[modalities]
input = ["text", "image", "video"]
input = ["text", "image"]
output = ["text"]
16 changes: 10 additions & 6 deletions providers/nano-gpt/models/Gemma-4-31B-Cognitive-Unshackled.toml
Original file line number Diff line number Diff line change
@@ -1,24 +1,28 @@
# Included in subscription
name = "Gemma 4 31B Cognitive Unshackled"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "gemma"
release_date = "2026-05-01"
last_updated = "2026-05-01"
attachment = true
reasoning = true
reasoning_options = [{ type = "toggle" }]
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[[reasoning_options]]
type = "toggle"

[cost]
input = 0.306
output = 0.306
cache_read = 0.153

[limit]
context = 262144
input = 262144
output = 16384
context = 262_144
input = 262_144
output = 16_384

[modalities]
input = ["text", "image", "video"]
input = ["text", "image"]
output = ["text"]
16 changes: 10 additions & 6 deletions providers/nano-gpt/models/Gemma-4-31B-DarkIdol.toml
Original file line number Diff line number Diff line change
@@ -1,24 +1,28 @@
# Included in subscription
name = "Gemma 4 31B DarkIdol"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "gemma"
release_date = "2026-05-01"
last_updated = "2026-05-01"
attachment = true
reasoning = true
reasoning_options = [{ type = "toggle" }]
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[[reasoning_options]]
type = "toggle"

[cost]
input = 0.306
output = 0.306
cache_read = 0.153

[limit]
context = 262144
input = 262144
output = 16384
context = 262_144
input = 262_144
output = 16_384

[modalities]
input = ["text", "image", "video"]
input = ["text", "image"]
output = ["text"]
16 changes: 10 additions & 6 deletions providers/nano-gpt/models/Gemma-4-31B-GarnetV2.toml
Original file line number Diff line number Diff line change
@@ -1,24 +1,28 @@
# Included in subscription
name = "Gemma 4 31B Garnet V2"
description = "Compact GPT model for low-latency assistance and high-volume workloads"
family = "gemma"
release_date = "2026-05-01"
last_updated = "2026-05-01"
attachment = true
reasoning = true
reasoning_options = [{ type = "toggle" }]
tool_call = false
structured_output = false
open_weights = false
open_weights = true

[[reasoning_options]]
type = "toggle"

[cost]
input = 0.306
output = 0.306
cache_read = 0.153

[limit]
context = 262144
input = 262144
output = 16384
context = 262_144
input = 262_144
output = 16_384

[modalities]
input = ["text", "image", "video"]
input = ["text", "image"]
output = ["text"]
Loading
Loading