2026-02-15 13:04:07 +00:00
|
|
|
package openai_compat
|
|
|
|
|
|
|
|
|
|
import (
|
|
|
|
|
"bytes"
|
|
|
|
|
"context"
|
|
|
|
|
"encoding/json"
|
|
|
|
|
"fmt"
|
|
|
|
|
"net/http"
|
|
|
|
|
"net/url"
|
|
|
|
|
"strings"
|
|
|
|
|
"time"
|
|
|
|
|
|
2026-03-14 14:52:34 +00:00
|
|
|
"github.com/sipeed/picoclaw/pkg/providers/common"
|
2026-02-17 16:13:10 +00:00
|
|
|
"github.com/sipeed/picoclaw/pkg/providers/protocoltypes"
|
|
|
|
|
)
|
2026-02-15 13:04:07 +00:00
|
|
|
|
2026-02-20 18:03:11 +00:00
|
|
|
type (
|
|
|
|
|
ToolCall = protocoltypes.ToolCall
|
|
|
|
|
FunctionCall = protocoltypes.FunctionCall
|
|
|
|
|
LLMResponse = protocoltypes.LLMResponse
|
|
|
|
|
UsageInfo = protocoltypes.UsageInfo
|
|
|
|
|
Message = protocoltypes.Message
|
|
|
|
|
ToolDefinition = protocoltypes.ToolDefinition
|
|
|
|
|
ToolFunctionDefinition = protocoltypes.ToolFunctionDefinition
|
|
|
|
|
ExtraContent = protocoltypes.ExtraContent
|
|
|
|
|
GoogleExtra = protocoltypes.GoogleExtra
|
2026-02-26 05:24:51 +00:00
|
|
|
ReasoningDetail = protocoltypes.ReasoningDetail
|
2026-02-20 18:03:11 +00:00
|
|
|
)
|
2026-02-15 13:04:07 +00:00
|
|
|
|
|
|
|
|
type Provider struct {
|
2026-02-19 16:12:01 +00:00
|
|
|
apiKey string
|
|
|
|
|
apiBase string
|
|
|
|
|
maxTokensField string // Field name for max tokens (e.g., "max_completion_tokens" for o1/glm models)
|
|
|
|
|
httpClient *http.Client
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
|
|
|
|
|
2026-02-26 08:08:19 +00:00
|
|
|
type Option func(*Provider)
|
|
|
|
|
|
2026-03-14 14:52:34 +00:00
|
|
|
const defaultRequestTimeout = common.DefaultRequestTimeout
|
2026-02-26 08:08:19 +00:00
|
|
|
|
|
|
|
|
func WithMaxTokensField(maxTokensField string) Option {
|
|
|
|
|
return func(p *Provider) {
|
|
|
|
|
p.maxTokensField = maxTokensField
|
|
|
|
|
}
|
2026-02-19 16:12:01 +00:00
|
|
|
}
|
|
|
|
|
|
2026-02-26 08:08:19 +00:00
|
|
|
func WithRequestTimeout(timeout time.Duration) Option {
|
|
|
|
|
return func(p *Provider) {
|
|
|
|
|
if timeout > 0 {
|
|
|
|
|
p.httpClient.Timeout = timeout
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func NewProvider(apiKey, apiBase, proxy string, opts ...Option) *Provider {
|
|
|
|
|
p := &Provider{
|
|
|
|
|
apiKey: apiKey,
|
|
|
|
|
apiBase: strings.TrimRight(apiBase, "/"),
|
2026-03-14 14:52:34 +00:00
|
|
|
httpClient: common.NewHTTPClient(proxy),
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
2026-02-26 08:08:19 +00:00
|
|
|
|
|
|
|
|
for _, opt := range opts {
|
|
|
|
|
if opt != nil {
|
|
|
|
|
opt(p)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
return p
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func NewProviderWithMaxTokensField(apiKey, apiBase, proxy, maxTokensField string) *Provider {
|
|
|
|
|
return NewProvider(apiKey, apiBase, proxy, WithMaxTokensField(maxTokensField))
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func NewProviderWithMaxTokensFieldAndTimeout(
|
|
|
|
|
apiKey, apiBase, proxy, maxTokensField string,
|
|
|
|
|
requestTimeoutSeconds int,
|
|
|
|
|
) *Provider {
|
|
|
|
|
return NewProvider(
|
|
|
|
|
apiKey,
|
|
|
|
|
apiBase,
|
|
|
|
|
proxy,
|
|
|
|
|
WithMaxTokensField(maxTokensField),
|
|
|
|
|
WithRequestTimeout(time.Duration(requestTimeoutSeconds)*time.Second),
|
|
|
|
|
)
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
|
|
|
|
|
2026-02-20 18:03:11 +00:00
|
|
|
func (p *Provider) Chat(
|
|
|
|
|
ctx context.Context,
|
|
|
|
|
messages []Message,
|
|
|
|
|
tools []ToolDefinition,
|
|
|
|
|
model string,
|
|
|
|
|
options map[string]any,
|
|
|
|
|
) (*LLMResponse, error) {
|
2026-02-15 13:04:07 +00:00
|
|
|
if p.apiBase == "" {
|
|
|
|
|
return nil, fmt.Errorf("API base not configured")
|
|
|
|
|
}
|
|
|
|
|
|
2026-02-17 16:13:10 +00:00
|
|
|
model = normalizeModel(model, p.apiBase)
|
2026-02-15 13:04:07 +00:00
|
|
|
|
2026-02-20 18:03:11 +00:00
|
|
|
requestBody := map[string]any{
|
2026-02-15 13:04:07 +00:00
|
|
|
"model": model,
|
2026-03-14 14:52:34 +00:00
|
|
|
"messages": common.SerializeMessages(messages),
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
|
|
|
|
|
2026-03-18 03:55:30 +00:00
|
|
|
// When fallback uses a different provider (e.g. DeepSeek), that provider must not inject web_search_preview.
|
|
|
|
|
nativeSearch, _ := options["native_search"].(bool)
|
|
|
|
|
nativeSearch = nativeSearch && isNativeSearchHost(p.apiBase)
|
|
|
|
|
if len(tools) > 0 || nativeSearch {
|
|
|
|
|
requestBody["tools"] = buildToolsList(tools, nativeSearch)
|
2026-02-15 13:04:07 +00:00
|
|
|
requestBody["tool_choice"] = "auto"
|
|
|
|
|
}
|
|
|
|
|
|
2026-03-14 14:52:34 +00:00
|
|
|
if maxTokens, ok := common.AsInt(options["max_tokens"]); ok {
|
2026-02-19 16:12:01 +00:00
|
|
|
// Use configured maxTokensField if specified, otherwise fallback to model-based detection
|
|
|
|
|
fieldName := p.maxTokensField
|
|
|
|
|
if fieldName == "" {
|
|
|
|
|
// Fallback: detect from model name for backward compatibility
|
|
|
|
|
lowerModel := strings.ToLower(model)
|
2026-02-20 18:03:11 +00:00
|
|
|
if strings.Contains(lowerModel, "glm") || strings.Contains(lowerModel, "o1") ||
|
|
|
|
|
strings.Contains(lowerModel, "gpt-5") {
|
2026-02-19 16:12:01 +00:00
|
|
|
fieldName = "max_completion_tokens"
|
|
|
|
|
} else {
|
|
|
|
|
fieldName = "max_tokens"
|
|
|
|
|
}
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
2026-02-19 16:12:01 +00:00
|
|
|
requestBody[fieldName] = maxTokens
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
|
|
|
|
|
2026-03-14 14:52:34 +00:00
|
|
|
if temperature, ok := common.AsFloat(options["temperature"]); ok {
|
2026-02-15 13:04:07 +00:00
|
|
|
lowerModel := strings.ToLower(model)
|
|
|
|
|
// Kimi k2 models only support temperature=1.
|
|
|
|
|
if strings.Contains(lowerModel, "kimi") && strings.Contains(lowerModel, "k2") {
|
|
|
|
|
requestBody["temperature"] = 1.0
|
|
|
|
|
} else {
|
|
|
|
|
requestBody["temperature"] = temperature
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
fix: cache system prompt with mtime-based auto-invalidation (#607)
Avoid rebuilding the entire system prompt on every BuildMessages() call
by caching the static portion (identity, bootstrap, skills summary,
memory) and only recomputing it when workspace source files change.
Key changes:
- ContextBuilder caches the static prompt behind an RWMutex with
double-checked locking. Source file changes are detected via cheap
os.Stat mtime checks so no explicit invalidation is needed.
- Track file existence at cache time (existedAtCache map) so that
newly created or deleted bootstrap/memory files also trigger a
rebuild — the old modifiedSince() silently returned false on
os.IsNotExist.
- Walk the skills directory recursively with filepath.WalkDir to
catch content-only edits at any nesting depth; directory mtime
alone misses in-place file modifications on most filesystems.
- ToolRegistry.sortedToolNames() sorts tool names before iteration,
ensuring deterministic tool definition order across calls — a
prerequisite for LLM-side prefix/KV cache reuse.
- Merge all context (static + dynamic + summary) into a single
system message for provider compatibility: the Anthropic adapter
extracts messages[0] as the top-level system parameter, and Codex
reads only the first system message as instructions.
- Fix a data race in BuildMessages() where cachedSystemPrompt was
read without holding the lock in a debug log statement.
- Add tests: single system message invariant, mtime auto-invalidation,
new-file creation detection, skill file content change, explicit
InvalidateCache, cache stability, concurrent access (20 goroutines
x 50 iterations, passes go test -race), and a benchmark.
2026-02-25 02:34:54 +00:00
|
|
|
// Prompt caching: pass a stable cache key so OpenAI can bucket requests
|
|
|
|
|
// with the same key and reuse prefix KV cache across calls.
|
|
|
|
|
// The key is typically the agent ID — stable per agent, shared across requests.
|
|
|
|
|
// See: https://platform.openai.com/docs/guides/prompt-caching
|
2026-02-25 15:09:46 +00:00
|
|
|
// Prompt caching is only supported by OpenAI-native endpoints.
|
2026-03-11 17:21:54 +00:00
|
|
|
// Non-OpenAI providers (Mistral, Gemini, DeepSeek, etc.) reject unknown
|
|
|
|
|
// fields with 422 errors, so only include it for OpenAI APIs.
|
fix: cache system prompt with mtime-based auto-invalidation (#607)
Avoid rebuilding the entire system prompt on every BuildMessages() call
by caching the static portion (identity, bootstrap, skills summary,
memory) and only recomputing it when workspace source files change.
Key changes:
- ContextBuilder caches the static prompt behind an RWMutex with
double-checked locking. Source file changes are detected via cheap
os.Stat mtime checks so no explicit invalidation is needed.
- Track file existence at cache time (existedAtCache map) so that
newly created or deleted bootstrap/memory files also trigger a
rebuild — the old modifiedSince() silently returned false on
os.IsNotExist.
- Walk the skills directory recursively with filepath.WalkDir to
catch content-only edits at any nesting depth; directory mtime
alone misses in-place file modifications on most filesystems.
- ToolRegistry.sortedToolNames() sorts tool names before iteration,
ensuring deterministic tool definition order across calls — a
prerequisite for LLM-side prefix/KV cache reuse.
- Merge all context (static + dynamic + summary) into a single
system message for provider compatibility: the Anthropic adapter
extracts messages[0] as the top-level system parameter, and Codex
reads only the first system message as instructions.
- Fix a data race in BuildMessages() where cachedSystemPrompt was
read without holding the lock in a debug log statement.
- Add tests: single system message invariant, mtime auto-invalidation,
new-file creation detection, skill file content change, explicit
InvalidateCache, cache stability, concurrent access (20 goroutines
x 50 iterations, passes go test -race), and a benchmark.
2026-02-25 02:34:54 +00:00
|
|
|
if cacheKey, ok := options["prompt_cache_key"].(string); ok && cacheKey != "" {
|
2026-03-11 17:21:54 +00:00
|
|
|
if supportsPromptCacheKey(p.apiBase) {
|
2026-02-25 15:09:46 +00:00
|
|
|
requestBody["prompt_cache_key"] = cacheKey
|
|
|
|
|
}
|
fix: cache system prompt with mtime-based auto-invalidation (#607)
Avoid rebuilding the entire system prompt on every BuildMessages() call
by caching the static portion (identity, bootstrap, skills summary,
memory) and only recomputing it when workspace source files change.
Key changes:
- ContextBuilder caches the static prompt behind an RWMutex with
double-checked locking. Source file changes are detected via cheap
os.Stat mtime checks so no explicit invalidation is needed.
- Track file existence at cache time (existedAtCache map) so that
newly created or deleted bootstrap/memory files also trigger a
rebuild — the old modifiedSince() silently returned false on
os.IsNotExist.
- Walk the skills directory recursively with filepath.WalkDir to
catch content-only edits at any nesting depth; directory mtime
alone misses in-place file modifications on most filesystems.
- ToolRegistry.sortedToolNames() sorts tool names before iteration,
ensuring deterministic tool definition order across calls — a
prerequisite for LLM-side prefix/KV cache reuse.
- Merge all context (static + dynamic + summary) into a single
system message for provider compatibility: the Anthropic adapter
extracts messages[0] as the top-level system parameter, and Codex
reads only the first system message as instructions.
- Fix a data race in BuildMessages() where cachedSystemPrompt was
read without holding the lock in a debug log statement.
- Add tests: single system message invariant, mtime auto-invalidation,
new-file creation detection, skill file content change, explicit
InvalidateCache, cache stability, concurrent access (20 goroutines
x 50 iterations, passes go test -race), and a benchmark.
2026-02-25 02:34:54 +00:00
|
|
|
}
|
|
|
|
|
|
2026-02-15 13:04:07 +00:00
|
|
|
jsonData, err := json.Marshal(requestBody)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to marshal request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req, err := http.NewRequestWithContext(ctx, "POST", p.apiBase+"/chat/completions", bytes.NewReader(jsonData))
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to create request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
req.Header.Set("Content-Type", "application/json")
|
|
|
|
|
if p.apiKey != "" {
|
|
|
|
|
req.Header.Set("Authorization", "Bearer "+p.apiKey)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
resp, err := p.httpClient.Do(req)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return nil, fmt.Errorf("failed to send request: %w", err)
|
|
|
|
|
}
|
|
|
|
|
defer resp.Body.Close()
|
|
|
|
|
|
2026-03-07 07:50:08 +00:00
|
|
|
if resp.StatusCode != http.StatusOK {
|
2026-03-14 14:52:34 +00:00
|
|
|
return nil, common.HandleErrorResponse(resp, p.apiBase)
|
2026-02-15 13:04:07 +00:00
|
|
|
}
|
|
|
|
|
|
2026-03-14 14:52:34 +00:00
|
|
|
return common.ReadAndParseResponse(resp, p.apiBase)
|
fix: cache system prompt with mtime-based auto-invalidation (#607)
Avoid rebuilding the entire system prompt on every BuildMessages() call
by caching the static portion (identity, bootstrap, skills summary,
memory) and only recomputing it when workspace source files change.
Key changes:
- ContextBuilder caches the static prompt behind an RWMutex with
double-checked locking. Source file changes are detected via cheap
os.Stat mtime checks so no explicit invalidation is needed.
- Track file existence at cache time (existedAtCache map) so that
newly created or deleted bootstrap/memory files also trigger a
rebuild — the old modifiedSince() silently returned false on
os.IsNotExist.
- Walk the skills directory recursively with filepath.WalkDir to
catch content-only edits at any nesting depth; directory mtime
alone misses in-place file modifications on most filesystems.
- ToolRegistry.sortedToolNames() sorts tool names before iteration,
ensuring deterministic tool definition order across calls — a
prerequisite for LLM-side prefix/KV cache reuse.
- Merge all context (static + dynamic + summary) into a single
system message for provider compatibility: the Anthropic adapter
extracts messages[0] as the top-level system parameter, and Codex
reads only the first system message as instructions.
- Fix a data race in BuildMessages() where cachedSystemPrompt was
read without holding the lock in a debug log statement.
- Add tests: single system message invariant, mtime auto-invalidation,
new-file creation detection, skill file content change, explicit
InvalidateCache, cache stability, concurrent access (20 goroutines
x 50 iterations, passes go test -race), and a benchmark.
2026-02-25 02:34:54 +00:00
|
|
|
}
|
|
|
|
|
|
2026-02-17 16:13:10 +00:00
|
|
|
func normalizeModel(model, apiBase string) string {
|
2026-02-27 08:35:07 +00:00
|
|
|
before, after, ok := strings.Cut(model, "/")
|
|
|
|
|
if !ok {
|
2026-02-17 16:13:10 +00:00
|
|
|
return model
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
if strings.Contains(strings.ToLower(apiBase), "openrouter.ai") {
|
|
|
|
|
return model
|
|
|
|
|
}
|
|
|
|
|
|
2026-02-27 08:35:07 +00:00
|
|
|
prefix := strings.ToLower(before)
|
2026-02-17 16:13:10 +00:00
|
|
|
switch prefix {
|
2026-03-06 12:20:22 +00:00
|
|
|
case "litellm", "moonshot", "nvidia", "groq", "ollama", "deepseek", "google",
|
2026-03-18 10:29:27 +00:00
|
|
|
"openrouter", "zhipu", "mistral", "vivgrid", "minimax", "novita":
|
2026-02-27 08:35:07 +00:00
|
|
|
return after
|
2026-02-17 16:13:10 +00:00
|
|
|
default:
|
|
|
|
|
return model
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
|
2026-03-18 03:55:30 +00:00
|
|
|
func buildToolsList(tools []ToolDefinition, nativeSearch bool) []any {
|
|
|
|
|
result := make([]any, 0, len(tools)+1)
|
|
|
|
|
for _, t := range tools {
|
|
|
|
|
if nativeSearch && strings.EqualFold(t.Function.Name, "web_search") {
|
|
|
|
|
continue
|
|
|
|
|
}
|
|
|
|
|
result = append(result, t)
|
|
|
|
|
}
|
|
|
|
|
if nativeSearch {
|
|
|
|
|
result = append(result, map[string]any{"type": "web_search_preview"})
|
|
|
|
|
}
|
|
|
|
|
return result
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func (p *Provider) SupportsNativeSearch() bool {
|
|
|
|
|
return isNativeSearchHost(p.apiBase)
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
func isNativeSearchHost(apiBase string) bool {
|
|
|
|
|
u, err := url.Parse(apiBase)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
host := u.Hostname()
|
|
|
|
|
return host == "api.openai.com" || strings.HasSuffix(host, ".openai.azure.com")
|
|
|
|
|
}
|
|
|
|
|
|
2026-03-11 17:21:54 +00:00
|
|
|
// supportsPromptCacheKey reports whether the given API base is known to
|
|
|
|
|
// support the prompt_cache_key request field. Currently only OpenAI's own
|
2026-03-11 19:24:31 +00:00
|
|
|
// API and Azure OpenAI support this. All other OpenAI-compatible providers
|
|
|
|
|
// (Mistral, Gemini, DeepSeek, Groq, etc.) reject unknown fields with 422 errors.
|
2026-03-11 17:21:54 +00:00
|
|
|
func supportsPromptCacheKey(apiBase string) bool {
|
2026-03-11 19:24:31 +00:00
|
|
|
u, err := url.Parse(apiBase)
|
|
|
|
|
if err != nil {
|
|
|
|
|
return false
|
|
|
|
|
}
|
|
|
|
|
host := u.Hostname()
|
|
|
|
|
return host == "api.openai.com" || strings.HasSuffix(host, ".openai.azure.com")
|
2026-03-11 17:21:54 +00:00
|
|
|
}
|