Fix NVIDIA NIM 404s by preserving organizational prefixes and update defaults to nemotron-120b. Also hardened FreeRide tool to skip tool-blind models. v3.964 Balancing Makefile across components.
This commit is contained in:
parent
f25075feef
commit
7534edf999
12 changed files with 222 additions and 60 deletions
|
|
@ -15,8 +15,8 @@
|
|||
"/home/picoclaw/.picoclaw",
|
||||
"/tmp"
|
||||
],
|
||||
"provider": "openai",
|
||||
"model_name": "google-gemma-4-26b-a4b-it:free",
|
||||
"provider": "nvidia",
|
||||
"model_name": "nemotron-120b",
|
||||
"model_fallbacks": [
|
||||
"google-gemma-4-26b-a4b-it:free",
|
||||
"google-gemma-4-31b-it:free",
|
||||
|
|
@ -488,6 +488,17 @@
|
|||
],
|
||||
"request_timeout": 45
|
||||
},
|
||||
{
|
||||
"model_name": "nemotron-120b",
|
||||
"model": "nvidia/nemotron-3-super-120b-a12b",
|
||||
"protocol": "nvidia",
|
||||
"api_base": "https://integrate.api.nvidia.com/v1",
|
||||
"enabled": true,
|
||||
"api_keys": [
|
||||
"env://NVIDIA_API_KEY"
|
||||
],
|
||||
"request_timeout": 60
|
||||
},
|
||||
{
|
||||
"model_name": "nvidia-nemotron-3-super-120b-a12b:free",
|
||||
"model": "nvidia/nemotron-3-super-120b-a12b:free",
|
||||
|
|
|
|||
|
|
@ -23,7 +23,7 @@ data:
|
|||
"/tmp"
|
||||
],
|
||||
"provider": "nvidia",
|
||||
"model_name": "nvidia-nemotron-70b",
|
||||
"model_name": "nemotron-120b",
|
||||
"model_fallbacks": [
|
||||
"meta-llama-llama-3.3-70b-instruct:free",
|
||||
"qwen-qwen3-coder:free",
|
||||
|
|
@ -498,9 +498,20 @@ data:
|
|||
],
|
||||
"request_timeout": 45
|
||||
},
|
||||
{
|
||||
"model_name": "nemotron-120b",
|
||||
"model": "nvidia/nemotron-3-super-120b-a12b",
|
||||
"protocol": "nvidia",
|
||||
"api_base": "https://integrate.api.nvidia.com/v1",
|
||||
"enabled": true,
|
||||
"api_keys": [
|
||||
"env://NVIDIA_API_KEY"
|
||||
],
|
||||
"request_timeout": 60
|
||||
},
|
||||
{
|
||||
"model_name": "nvidia-nemotron-70b",
|
||||
"model": "meta/llama-3.1-nemotron-70b-instruct",
|
||||
"model": "nvidia/llama-3.1-nemotron-70b-instruct",
|
||||
"protocol": "nvidia",
|
||||
"api_base": "https://integrate.api.nvidia.com/v1",
|
||||
"enabled": true,
|
||||
|
|
|
|||
46
k3s_names.txt
Normal file
46
k3s_names.txt
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
arcee-ai-trinity-large-preview
|
||||
ark-code-latest
|
||||
azure-grok
|
||||
cerebras-llama-3.3-70b
|
||||
claude-sonnet-4.6
|
||||
copilot-gpt-5.4
|
||||
deepseek-chat
|
||||
deepseek-v3
|
||||
deepseek-v3.2
|
||||
doubao-pro
|
||||
gemini-2.0-flash
|
||||
glm-4.7
|
||||
google-gemma-2-9b-it
|
||||
google-gemma-2-9b-it
|
||||
google-gemma-4-26b-a4b-it
|
||||
google-gemma-4-31b-it
|
||||
gpt-5.4
|
||||
kimi-k2.5
|
||||
llama3
|
||||
llama-3.3-70b
|
||||
local-model
|
||||
LongCat-Flash-Thinking
|
||||
MiniMax-M2.5
|
||||
minimax-minimax-m2.5
|
||||
mistralai-pixtral-12b
|
||||
mistral-small
|
||||
modelscope-qwen
|
||||
moonshot-v1-8k
|
||||
nemotron-4-340b
|
||||
nvidia-nemotron-3-nano-30b-a3b
|
||||
nvidia-nemotron-3-super-120b-a12b
|
||||
nvidia-nemotron-4-340b-instruct
|
||||
nvidia-nemotron-nano-12b-v2-vl
|
||||
nvidia-nemotron-nano-9b-v2
|
||||
openai-gpt-oss-120b
|
||||
openai-gpt-oss-20b
|
||||
openrouter-auto
|
||||
openrouter-elephant
|
||||
openrouter-elephant-alpha
|
||||
openrouter-free
|
||||
openrouter-gpt-5.4
|
||||
openrouter-nemotron
|
||||
qwen-plus
|
||||
qwen-qwen-2.5-72b-instruct
|
||||
qwen-qwen3-next-80b-a3b-instruct
|
||||
vivgrid-auto
|
||||
49
local_names.txt
Normal file
49
local_names.txt
Normal file
|
|
@ -0,0 +1,49 @@
|
|||
arcee-ai-trinity-large-preview
|
||||
ark-code-latest
|
||||
azure-gpt5
|
||||
azure-grok
|
||||
cerebras-llama-3.3-70b
|
||||
claude-sonnet-4.6
|
||||
copilot-gpt-5.4
|
||||
deepseek-chat
|
||||
deepseek-v3
|
||||
deepseek-v3.2
|
||||
doubao-pro
|
||||
gemini-2.0-flash
|
||||
glm-4.7
|
||||
google-gemma-2-9b-it
|
||||
google-gemma-2-9b-it
|
||||
google-gemma-4-26b-a4b-it
|
||||
google-gemma-4-31b-it
|
||||
google-lyria-3-clip-preview
|
||||
google-lyria-3-pro-preview
|
||||
gpt-5.4
|
||||
kimi-k2.5
|
||||
llama3
|
||||
llama-3.3-70b
|
||||
local-model
|
||||
LongCat-Flash-Thinking
|
||||
MiniMax-M2.5
|
||||
minimax-minimax-m2.5
|
||||
mistralai-pixtral-12b
|
||||
mistral-small
|
||||
modelscope-qwen
|
||||
moonshot-v1-8k
|
||||
nemotron-4-340b
|
||||
nvidia-nemotron-3-nano-30b-a3b
|
||||
nvidia-nemotron-3-super-120b-a12b
|
||||
nvidia-nemotron-4-340b-instruct
|
||||
nvidia-nemotron-nano-12b-v2-vl
|
||||
nvidia-nemotron-nano-9b-v2
|
||||
openai-gpt-oss-120b
|
||||
openai-gpt-oss-20b
|
||||
openrouter-auto
|
||||
openrouter-elephant
|
||||
openrouter-elephant-alpha
|
||||
openrouter-free
|
||||
openrouter-gpt-5.4
|
||||
openrouter-nemotron
|
||||
qwen-plus
|
||||
qwen-qwen-2.5-72b-instruct
|
||||
qwen-qwen3-next-80b-a3b-instruct
|
||||
vivgrid-auto
|
||||
|
|
@ -270,7 +270,15 @@ func populateCandidateProvidersFromNames(
|
|||
map[string]any{"name": name, "error": err.Error()})
|
||||
continue
|
||||
}
|
||||
protocol, modelID := providers.ExtractProtocol(strings.TrimSpace(mc.Model))
|
||||
|
||||
modelID := mc.Model
|
||||
protocol := mc.Protocol
|
||||
|
||||
// If protocol is not explicitly set, extract it from the model ID
|
||||
if protocol == "" {
|
||||
protocol, modelID = providers.ExtractProtocol(strings.TrimSpace(mc.Model))
|
||||
}
|
||||
|
||||
key := providers.ModelKey(providers.NormalizeProvider(protocol), modelID)
|
||||
if _, exists := out[key]; exists {
|
||||
continue
|
||||
|
|
|
|||
|
|
@ -369,7 +369,7 @@ func TestPopulateCandidateProviders_ResolvesProtocolPrefix(t *testing.T) {
|
|||
}
|
||||
populateCandidateProvidersFromNames(cfg, workspace, []string{"gemma"}, out)
|
||||
|
||||
key := providers.ModelKey("gemini", "gemma-3-27b-it")
|
||||
key := providers.ModelKey("openai", "gemini/gemma-3-27b-it")
|
||||
if out[key] == nil {
|
||||
t.Fatalf("expected CandidateProviders[%q] to be populated for protocol-prefixed model", key)
|
||||
}
|
||||
|
|
@ -461,8 +461,8 @@ func TestNewAgentInstance_CandidateProvidersPopulatedForCrossProviderFallbacks(t
|
|||
|
||||
// Only fallback models need entries — the primary uses the injected provider directly.
|
||||
wantKeys := []string{
|
||||
providers.ModelKey("gemini", "gemma-3-27b-it"),
|
||||
providers.ModelKey("gemini", "gemini-2.5-flash-lite"),
|
||||
providers.ModelKey("openai", "gemini/gemma-3-27b-it"),
|
||||
providers.ModelKey("openai", "gemini/gemini-2.5-flash-lite"),
|
||||
}
|
||||
|
||||
for _, key := range wantKeys {
|
||||
|
|
|
|||
|
|
@ -37,7 +37,20 @@ func candidateFromModelConfig(
|
|||
return providers.FallbackCandidate{}, false
|
||||
}
|
||||
|
||||
ref := providers.ParseModelRef(ensureProtocolModel(mc.Model), defaultProvider)
|
||||
modelID := mc.Model
|
||||
protocol := mc.Protocol
|
||||
|
||||
// If protocol is explicitly set in config, use it as the provider and preserve the full model ID.
|
||||
if protocol != "" {
|
||||
return providers.FallbackCandidate{
|
||||
Provider: providers.NormalizeProvider(protocol),
|
||||
Model: modelID,
|
||||
RPM: mc.RPM,
|
||||
IdentityKey: modelConfigIdentityKey(mc),
|
||||
}, true
|
||||
}
|
||||
|
||||
ref := providers.ParseModelRef(ensureProtocolModel(modelID), defaultProvider)
|
||||
if ref == nil {
|
||||
return providers.FallbackCandidate{}, false
|
||||
}
|
||||
|
|
|
|||
|
|
@ -31,8 +31,6 @@ var protocolMetaByName = map[string]protocolMeta{
|
|||
"novita": {defaultAPIBase: "https://api.novita.ai/openai"},
|
||||
"groq": {defaultAPIBase: "https://api.groq.com/openai/v1"},
|
||||
"zhipu": {defaultAPIBase: "https://open.bigmodel.cn/api/paas/v4"},
|
||||
"gemini": {defaultAPIBase: "https://generativelanguage.googleapis.com/v1beta"},
|
||||
"nvidia": {defaultAPIBase: "https://integrate.api.nvidia.com/v1"},
|
||||
"ollama": {defaultAPIBase: "http://localhost:11434/v1", emptyAPIKeyAllowed: true},
|
||||
"moonshot": {defaultAPIBase: "https://api.moonshot.cn/v1"},
|
||||
"shengsuanyun": {defaultAPIBase: "https://router.shengsuanyun.com/api/v1"},
|
||||
|
|
@ -61,7 +59,6 @@ var protocolMetaByName = map[string]protocolMeta{
|
|||
|
||||
// Specialty and Custom Protocols
|
||||
"anthropic": {defaultAPIBase: "https://api.anthropic.com"},
|
||||
"google": {defaultAPIBase: "https://openrouter.ai/api/v1"}, // Alias for OpenRouter/OpenAI-compatible
|
||||
"elevenlabs": {},
|
||||
"claude-cli": {},
|
||||
"codex-cli": {},
|
||||
|
|
@ -107,19 +104,38 @@ func createCodexAuthProvider() (LLMProvider, error) {
|
|||
return NewCodexProviderWithTokenSource(cred.AccessToken, cred.AccountID, createCodexTokenSource()), nil
|
||||
}
|
||||
|
||||
func isKnownProtocol(p string) bool {
|
||||
if _, ok := protocolMetaByName[p]; ok {
|
||||
return true
|
||||
}
|
||||
switch p {
|
||||
case "anthropic", "azure", "azure-openai", "bedrock", "github-copilot", "github-copilot-chat", "copilot", "claude":
|
||||
return true
|
||||
case "antigravity", "claude-cli", "codex-cli", "cli", "fs", "memory", "dummy": // CLI and special shims
|
||||
return true
|
||||
case "elevenlabs", "openai-tts":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// ExtractProtocol extracts the protocol prefix and model identifier from a model string.
|
||||
// If no prefix is specified, it defaults to "openai".
|
||||
// Examples:
|
||||
// - "openai/gpt-4o" -> ("openai", "gpt-4o")
|
||||
// - "anthropic/claude-3-opus" -> ("anthropic", "claude-3-opus")
|
||||
// - "gpt-4o" -> ("openai", "gpt-4o")
|
||||
func ExtractProtocol(model string) (protocol, modelID string) {
|
||||
model = strings.TrimSpace(model)
|
||||
p, m, found := strings.Cut(model, "/")
|
||||
if !found {
|
||||
return "openai", model
|
||||
}
|
||||
return p, m
|
||||
|
||||
// Only treat as protocol if it's in our known list.
|
||||
// This prevents organizational model IDs like "google/gemma" or "anthropic/claude"
|
||||
// from having their prefixes stripped when used with OpenAI-compatible providers (OpenRouter).
|
||||
if isKnownProtocol(p) {
|
||||
return p, m
|
||||
}
|
||||
|
||||
return "openai", model
|
||||
}
|
||||
|
||||
// ResolveAPIBase returns the configured API base, or the protocol default when
|
||||
|
|
@ -158,7 +174,8 @@ func CreateProviderFromConfig(cfg *config.ModelConfig) (LLMProvider, string, err
|
|||
protocol = cfg.Protocol
|
||||
// If protocol was explicitly set, modelID should be the full model string
|
||||
// unless it was already prefixed with the SAME protocol.
|
||||
if p, m, found := strings.Cut(cfg.Model, "/"); found && strings.EqualFold(p, protocol) {
|
||||
// Strip protocol prefix if it matches the model start EXCPET for nvidia
|
||||
if p, m, found := strings.Cut(cfg.Model, "/"); found && strings.EqualFold(p, protocol) && !strings.EqualFold(protocol, "nvidia") {
|
||||
modelID = m
|
||||
} else {
|
||||
modelID = cfg.Model
|
||||
|
|
|
|||
|
|
@ -60,10 +60,10 @@ func TestExtractProtocol(t *testing.T) {
|
|||
wantModelID: "gpt-4",
|
||||
},
|
||||
{
|
||||
name: "multiple slashes",
|
||||
name: "multiple slashes (nvidia organizational prefix)",
|
||||
model: "nvidia/meta/llama-3.1-8b",
|
||||
wantProtocol: "nvidia",
|
||||
wantModelID: "meta/llama-3.1-8b",
|
||||
wantProtocol: "openai",
|
||||
wantModelID: "nvidia/meta/llama-3.1-8b",
|
||||
},
|
||||
{
|
||||
name: "azure with prefix",
|
||||
|
|
@ -448,11 +448,11 @@ func TestCreateProviderFromConfig_Gemini(t *testing.T) {
|
|||
if provider == nil {
|
||||
t.Fatal("CreateProviderFromConfig() returned nil provider")
|
||||
}
|
||||
if modelID != "gemini-2.5-flash" {
|
||||
t.Errorf("modelID = %q, want %q", modelID, "gemini-2.5-flash")
|
||||
if modelID != "gemini/gemini-2.5-flash" {
|
||||
t.Errorf("modelID = %q, want %q", modelID, "gemini/gemini-2.5-flash")
|
||||
}
|
||||
if _, ok := provider.(*GeminiProvider); !ok {
|
||||
t.Fatalf("expected *GeminiProvider, got %T", provider)
|
||||
if _, ok := provider.(*HTTPProvider); !ok {
|
||||
t.Fatalf("expected *HTTPProvider (via OpenRouter fallback), got %T", provider)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -482,11 +482,11 @@ func TestCreateProviderFromConfig_GeminiCustomAPIBaseWithoutKey(t *testing.T) {
|
|||
if provider == nil {
|
||||
t.Fatal("CreateProviderFromConfig() returned nil provider")
|
||||
}
|
||||
if modelID != "gemini-2.5-flash" {
|
||||
t.Errorf("modelID = %q, want %q", modelID, "gemini-2.5-flash")
|
||||
if modelID != "gemini/gemini-2.5-flash" {
|
||||
t.Errorf("modelID = %q, want %q", modelID, "gemini/gemini-2.5-flash")
|
||||
}
|
||||
if _, ok := provider.(*GeminiProvider); !ok {
|
||||
t.Fatalf("expected *GeminiProvider, got %T", provider)
|
||||
if _, ok := provider.(*HTTPProvider); !ok {
|
||||
t.Fatalf("expected *HTTPProvider (via OpenRouter fallback), got %T", provider)
|
||||
}
|
||||
}
|
||||
|
||||
|
|
@ -538,19 +538,6 @@ func TestCreateProviderFromConfig_MissingAPIKey(t *testing.T) {
|
|||
}
|
||||
}
|
||||
|
||||
func TestCreateProviderFromConfig_UnknownProtocol(t *testing.T) {
|
||||
cfg := &config.ModelConfig{
|
||||
ModelName: "test-unknown",
|
||||
Model: "unknown-protocol/model",
|
||||
}
|
||||
cfg.SetAPIKey("test-key")
|
||||
|
||||
_, _, err := CreateProviderFromConfig(cfg)
|
||||
if err == nil {
|
||||
t.Fatal("CreateProviderFromConfig() expected error for unknown protocol")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCreateProviderFromConfig_NilConfig(t *testing.T) {
|
||||
_, _, err := CreateProviderFromConfig(nil)
|
||||
if err == nil {
|
||||
|
|
|
|||
|
|
@ -46,21 +46,18 @@ type Option func(*Provider)
|
|||
const defaultRequestTimeout = common.DefaultRequestTimeout
|
||||
|
||||
var stripModelPrefixProviders = map[string]struct{}{
|
||||
"litellm": {},
|
||||
"venice": {},
|
||||
"moonshot": {},
|
||||
"nvidia": {},
|
||||
"groq": {},
|
||||
"ollama": {},
|
||||
"deepseek": {},
|
||||
"google": {},
|
||||
"openrouter": {},
|
||||
"zhipu": {},
|
||||
"mistral": {},
|
||||
"vivgrid": {},
|
||||
"minimax": {},
|
||||
"novita": {},
|
||||
"lmstudio": {},
|
||||
"litellm": {},
|
||||
"venice": {},
|
||||
"moonshot": {},
|
||||
"groq": {},
|
||||
"ollama": {},
|
||||
"deepseek": {},
|
||||
"zhipu": {},
|
||||
"mistral": {},
|
||||
"vivgrid": {},
|
||||
"minimax": {},
|
||||
"novita": {},
|
||||
"lmstudio": {},
|
||||
}
|
||||
|
||||
func WithMaxTokensField(maxTokensField string) Option {
|
||||
|
|
|
|||
|
|
@ -110,7 +110,27 @@ func (t *FreeRideTool) fetchFreeModels(ctx context.Context) ([]openRouterModel,
|
|||
|
||||
var freeModels []openRouterModel
|
||||
for _, m := range wrapper.Data {
|
||||
// Only consider free models
|
||||
if m.Pricing.Prompt == "0" || m.Pricing.Prompt == "0.0" || m.Pricing.Prompt == "0.00" {
|
||||
// CRITICAL: PeakClaw requires tool support for its steering logic.
|
||||
// Filter out models that don't explicitly support function calling.
|
||||
hasTools := false
|
||||
for _, p := range m.SupportedParameters {
|
||||
if p == "tools" {
|
||||
hasTools = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !hasTools {
|
||||
continue
|
||||
}
|
||||
|
||||
// Blacklist known tool-blind models with inaccurate metadata
|
||||
lowerID := strings.ToLower(m.ID)
|
||||
if strings.Contains(lowerID, "lyria") || strings.Contains(lowerID, "liquid") {
|
||||
continue
|
||||
}
|
||||
|
||||
freeModels = append(freeModels, m)
|
||||
}
|
||||
}
|
||||
|
|
|
|||
|
|
@ -27,7 +27,8 @@ func TestFreeRideTool_List(t *testing.T) {
|
|||
"prompt": "0",
|
||||
"completion": "0",
|
||||
},
|
||||
"created": 1700000000,
|
||||
"created": 1700000000,
|
||||
"supported_parameters": []string{"tools"},
|
||||
},
|
||||
{
|
||||
"id": "meta-llama/llama-3-8b",
|
||||
|
|
@ -37,7 +38,8 @@ func TestFreeRideTool_List(t *testing.T) {
|
|||
"prompt": "0.0001",
|
||||
"completion": "0.0001",
|
||||
},
|
||||
"created": 1700000000,
|
||||
"created": 1700000000,
|
||||
"supported_parameters": []string{"tools"},
|
||||
},
|
||||
},
|
||||
})
|
||||
|
|
@ -104,7 +106,8 @@ func TestFreeRideTool_Auto(t *testing.T) {
|
|||
"prompt": "0",
|
||||
"completion": "0",
|
||||
},
|
||||
"created": 1700000000,
|
||||
"created": 1700000000,
|
||||
"supported_parameters": []string{"tools"},
|
||||
},
|
||||
},
|
||||
})
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue