* refactor: consolidate relay protocol converters * refactor relayconvert text converters * feat: refine relay converters and advanced custom routing * refactor: enhance logging and add thought signature handling for Gemini requests * refactor: enhance channel cache and pricing endpoint handling for advanced custom models * feat: preserve billing usage semantics * feat: add protocol-aware billing usage * Delete useless files * chore: update action versions in workflow files * chore: update Docker action versions in workflow files * fix: harden billing usage settlement and hot-path route matching - estimate Gemini completion tokens locally when billable usageMetadata is prompt-only but output content was received (e.g. client aborts the stream before the final chunk), and rebuild the attached billing_usage as estimated so settlement does not bill zero output tokens - guard NewClaudeMessagesBillingUsage against all-zero ClaudeUsage, matching the OpenAI/Gemini constructors, so a zero billing_usage cannot override a non-zero top-level usage during settlement - cache compiled advanced-custom route model regexes; they run on the request hot path and were recompiled per request - move the effectiveBillingUsage remap to PostTextConsumeQuota only, and document that calculateTextQuotaSummary expects remapped usage - document the updatePricingLock -> channelSyncLock lock ordering that InitChannelCache/CacheUpdateChannel rely on, and the aux-struct pitfall in GeminiChatResponse.UnmarshalJSON
269 lines
7.2 KiB
Go
269 lines
7.2 KiB
Go
package gemini
|
|
|
|
import (
|
|
"strconv"
|
|
"strings"
|
|
|
|
"github.com/QuantumNous/new-api/common"
|
|
"github.com/QuantumNous/new-api/dto"
|
|
relaycommon "github.com/QuantumNous/new-api/relay/common"
|
|
relaymeta "github.com/QuantumNous/new-api/service/relayconvert/internal/meta"
|
|
"github.com/QuantumNous/new-api/setting/model_setting"
|
|
"github.com/QuantumNous/new-api/setting/reasoning"
|
|
)
|
|
|
|
var SupportedMimeTypes = map[string]bool{
|
|
"application/pdf": true,
|
|
"audio/mpeg": true,
|
|
"audio/mp3": true,
|
|
"audio/wav": true,
|
|
"image/png": true,
|
|
"image/jpeg": true,
|
|
"image/jpg": true,
|
|
"image/webp": true,
|
|
"image/heic": true,
|
|
"image/heif": true,
|
|
"text/plain": true,
|
|
"video/mov": true,
|
|
"video/mpeg": true,
|
|
"video/mp4": true,
|
|
"video/mpg": true,
|
|
"video/avi": true,
|
|
"video/wmv": true,
|
|
"video/mpegps": true,
|
|
"video/flv": true,
|
|
}
|
|
|
|
var SafetySettingCategories = []string{
|
|
"HARM_CATEGORY_HARASSMENT",
|
|
"HARM_CATEGORY_HATE_SPEECH",
|
|
"HARM_CATEGORY_SEXUALLY_EXPLICIT",
|
|
"HARM_CATEGORY_DANGEROUS_CONTENT",
|
|
}
|
|
|
|
const ThoughtSignatureBypassValue = "context_engineering_is_the_way_to_go"
|
|
|
|
const (
|
|
pro25MinBudget = 128
|
|
pro25MaxBudget = 32768
|
|
flash25MaxBudget = 24576
|
|
flash25LiteMinBudget = 512
|
|
flash25LiteMaxBudget = 24576
|
|
)
|
|
|
|
func ShouldAttachThoughtSignature() bool {
|
|
return model_setting.GetGeminiSettings().FunctionCallThoughtSignatureEnabled
|
|
}
|
|
|
|
func AttachThoughtSignatureBypass(part *dto.GeminiPart) bool {
|
|
if part == nil || len(part.ThoughtSignature) > 0 || !ShouldAttachThoughtSignature() {
|
|
return false
|
|
}
|
|
part.ThoughtSignature = []byte(strconv.Quote(ThoughtSignatureBypassValue))
|
|
return true
|
|
}
|
|
|
|
func AttachFunctionCallThoughtSignature(part *dto.GeminiPart) bool {
|
|
if part == nil || !HasFunctionCallContent(part.FunctionCall) {
|
|
return false
|
|
}
|
|
return AttachThoughtSignatureBypass(part)
|
|
}
|
|
|
|
func AttachFirstTextThoughtSignature(parts []dto.GeminiPart) bool {
|
|
if !ShouldAttachThoughtSignature() {
|
|
return false
|
|
}
|
|
for i := range parts {
|
|
if parts[i].Text != "" && len(parts[i].ThoughtSignature) == 0 {
|
|
parts[i].ThoughtSignature = []byte(strconv.Quote(ThoughtSignatureBypassValue))
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func ApplyThinkingConfig(geminiRequest *dto.GeminiChatRequest, info *relaycommon.RelayInfo, oaiRequest ...dto.GeneralOpenAIRequest) {
|
|
if geminiRequest == nil || info == nil || !model_setting.GetGeminiSettings().ThinkingAdapterEnabled {
|
|
return
|
|
}
|
|
|
|
modelName := relaymeta.RelayInfoUpstreamModelName(info)
|
|
isNew25Pro := strings.HasPrefix(modelName, "gemini-2.5-pro") &&
|
|
!strings.HasPrefix(modelName, "gemini-2.5-pro-preview-05-06") &&
|
|
!strings.HasPrefix(modelName, "gemini-2.5-pro-preview-03-25")
|
|
|
|
if strings.Contains(modelName, "-thinking-") {
|
|
parts := strings.SplitN(modelName, "-thinking-", 2)
|
|
if len(parts) == 2 && parts[1] != "" {
|
|
if budgetTokens, err := strconv.Atoi(parts[1]); err == nil {
|
|
clampedBudget := clampThinkingBudget(modelName, budgetTokens)
|
|
geminiRequest.GenerationConfig.ThinkingConfig = &dto.GeminiThinkingConfig{
|
|
ThinkingBudget: common.GetPointer(clampedBudget),
|
|
IncludeThoughts: true,
|
|
}
|
|
}
|
|
}
|
|
} else if strings.HasSuffix(modelName, "-thinking") {
|
|
unsupportedModels := []string{
|
|
"gemini-2.5-pro-preview-05-06",
|
|
"gemini-2.5-pro-preview-03-25",
|
|
}
|
|
isUnsupported := false
|
|
for _, unsupportedModel := range unsupportedModels {
|
|
if strings.HasPrefix(modelName, unsupportedModel) {
|
|
isUnsupported = true
|
|
break
|
|
}
|
|
}
|
|
|
|
if isUnsupported {
|
|
geminiRequest.GenerationConfig.ThinkingConfig = &dto.GeminiThinkingConfig{
|
|
IncludeThoughts: true,
|
|
}
|
|
} else {
|
|
geminiRequest.GenerationConfig.ThinkingConfig = &dto.GeminiThinkingConfig{
|
|
IncludeThoughts: true,
|
|
}
|
|
if geminiRequest.GenerationConfig.MaxOutputTokens != nil && *geminiRequest.GenerationConfig.MaxOutputTokens > 0 {
|
|
budgetTokens := model_setting.GetGeminiSettings().ThinkingAdapterBudgetTokensPercentage * float64(*geminiRequest.GenerationConfig.MaxOutputTokens)
|
|
clampedBudget := clampThinkingBudget(modelName, int(budgetTokens))
|
|
geminiRequest.GenerationConfig.ThinkingConfig.ThinkingBudget = common.GetPointer(clampedBudget)
|
|
} else if len(oaiRequest) > 0 {
|
|
geminiRequest.GenerationConfig.ThinkingConfig.ThinkingBudget = common.GetPointer(clampThinkingBudgetByEffort(modelName, oaiRequest[0].ReasoningEffort))
|
|
}
|
|
}
|
|
} else if strings.HasSuffix(modelName, "-nothinking") {
|
|
if !isNew25Pro {
|
|
geminiRequest.GenerationConfig.ThinkingConfig = &dto.GeminiThinkingConfig{
|
|
ThinkingBudget: common.GetPointer(0),
|
|
}
|
|
}
|
|
} else if _, level, ok := reasoning.TrimEffortSuffix(modelName); ok && level != "" {
|
|
geminiRequest.GenerationConfig.ThinkingConfig = &dto.GeminiThinkingConfig{
|
|
IncludeThoughts: true,
|
|
ThinkingLevel: level,
|
|
}
|
|
info.ReasoningEffort = level
|
|
}
|
|
}
|
|
|
|
func ParseStopSequences(stop any) []string {
|
|
if stop == nil {
|
|
return nil
|
|
}
|
|
|
|
switch v := stop.(type) {
|
|
case string:
|
|
if v != "" {
|
|
return []string{v}
|
|
}
|
|
case []string:
|
|
return v
|
|
case []interface{}:
|
|
sequences := make([]string, 0, len(v))
|
|
for _, item := range v {
|
|
if str, ok := item.(string); ok && str != "" {
|
|
sequences = append(sequences, str)
|
|
}
|
|
}
|
|
return sequences
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func HasFunctionCallContent(call *dto.FunctionCall) bool {
|
|
if call == nil {
|
|
return false
|
|
}
|
|
if strings.TrimSpace(call.FunctionName) != "" {
|
|
return true
|
|
}
|
|
|
|
switch v := call.Arguments.(type) {
|
|
case nil:
|
|
return false
|
|
case string:
|
|
return strings.TrimSpace(v) != ""
|
|
case map[string]interface{}:
|
|
return len(v) > 0
|
|
case []interface{}:
|
|
return len(v) > 0
|
|
default:
|
|
return true
|
|
}
|
|
}
|
|
|
|
func SupportedMimeTypesList() []string {
|
|
keys := make([]string, 0, len(SupportedMimeTypes))
|
|
for key := range SupportedMimeTypes {
|
|
keys = append(keys, key)
|
|
}
|
|
return keys
|
|
}
|
|
|
|
func isNew25ProModel(modelName string) bool {
|
|
return strings.HasPrefix(modelName, "gemini-2.5-pro") &&
|
|
!strings.HasPrefix(modelName, "gemini-2.5-pro-preview-05-06") &&
|
|
!strings.HasPrefix(modelName, "gemini-2.5-pro-preview-03-25")
|
|
}
|
|
|
|
func is25FlashLiteModel(modelName string) bool {
|
|
return strings.HasPrefix(modelName, "gemini-2.5-flash-lite")
|
|
}
|
|
|
|
func clampThinkingBudget(modelName string, budget int) int {
|
|
isNew25Pro := isNew25ProModel(modelName)
|
|
is25FlashLite := is25FlashLiteModel(modelName)
|
|
|
|
if is25FlashLite {
|
|
if budget < flash25LiteMinBudget {
|
|
return flash25LiteMinBudget
|
|
}
|
|
if budget > flash25LiteMaxBudget {
|
|
return flash25LiteMaxBudget
|
|
}
|
|
} else if isNew25Pro {
|
|
if budget < pro25MinBudget {
|
|
return pro25MinBudget
|
|
}
|
|
if budget > pro25MaxBudget {
|
|
return pro25MaxBudget
|
|
}
|
|
} else {
|
|
if budget < 0 {
|
|
return 0
|
|
}
|
|
if budget > flash25MaxBudget {
|
|
return flash25MaxBudget
|
|
}
|
|
}
|
|
return budget
|
|
}
|
|
|
|
func clampThinkingBudgetByEffort(modelName string, effort string) int {
|
|
isNew25Pro := isNew25ProModel(modelName)
|
|
is25FlashLite := is25FlashLiteModel(modelName)
|
|
|
|
maxBudget := 0
|
|
if is25FlashLite {
|
|
maxBudget = flash25LiteMaxBudget
|
|
}
|
|
if isNew25Pro {
|
|
maxBudget = pro25MaxBudget
|
|
} else {
|
|
maxBudget = flash25MaxBudget
|
|
}
|
|
switch effort {
|
|
case "high":
|
|
maxBudget = maxBudget * 80 / 100
|
|
case "medium":
|
|
maxBudget = maxBudget * 50 / 100
|
|
case "low":
|
|
maxBudget = maxBudget * 20 / 100
|
|
case "minimal":
|
|
maxBudget = maxBudget * 5 / 100
|
|
}
|
|
return clampThinkingBudget(modelName, maxBudget)
|
|
}
|