feat: vllm thinking_token_budget (#7027)

This commit is contained in:
Seefs
2026-08-26 21:06:22 +08:00
committed by GitHub
parent 8c25eee71b
commit 8f6961c675
+10 -8
View File
@@ -106,6 +106,8 @@ type GeneralOpenAIRequest struct {
SearchMode json.RawMessage `json:"search_mode,omitempty"` SearchMode json.RawMessage `json:"search_mode,omitempty"`
// Minimax // Minimax
ReasoningSplit json.RawMessage `json:"reasoning_split,omitempty"` ReasoningSplit json.RawMessage `json:"reasoning_split,omitempty"`
// vLLM
ThinkingTokenBudget json.RawMessage `json:"thinking_token_budget,omitempty"`
} }
func (r GeneralOpenAIRequest) MarshalJSON() ([]byte, error) { func (r GeneralOpenAIRequest) MarshalJSON() ([]byte, error) {
@@ -859,14 +861,14 @@ type OpenAIResponsesRequest struct {
Include json.RawMessage `json:"include,omitempty"` Include json.RawMessage `json:"include,omitempty"`
// 在后台运行推理,暂时还不支持依赖的接口 // 在后台运行推理,暂时还不支持依赖的接口
// Background json.RawMessage `json:"background,omitempty"` // Background json.RawMessage `json:"background,omitempty"`
Conversation json.RawMessage `json:"conversation,omitempty"` Conversation json.RawMessage `json:"conversation,omitempty"`
ContextManagement json.RawMessage `json:"context_management,omitempty"` ContextManagement json.RawMessage `json:"context_management,omitempty"`
Instructions json.RawMessage `json:"instructions,omitempty"` Instructions json.RawMessage `json:"instructions,omitempty"`
MaxOutputTokens *uint `json:"max_output_tokens,omitempty"` MaxOutputTokens *uint `json:"max_output_tokens,omitempty"`
TopLogProbs *int `json:"top_logprobs,omitempty"` TopLogProbs *int `json:"top_logprobs,omitempty"`
Metadata json.RawMessage `json:"metadata,omitempty"` Metadata json.RawMessage `json:"metadata,omitempty"`
Moderation json.RawMessage `json:"moderation,omitempty"` Moderation json.RawMessage `json:"moderation,omitempty"`
ParallelToolCalls json.RawMessage `json:"parallel_tool_calls,omitempty"` ParallelToolCalls json.RawMessage `json:"parallel_tool_calls,omitempty"`
// FrequencyPenalty/PresencePenalty are not part of the official OpenAI // FrequencyPenalty/PresencePenalty are not part of the official OpenAI
// Responses API; they are forwarded verbatim for OpenAI-compatible upstreams // Responses API; they are forwarded verbatim for OpenAI-compatible upstreams
// (e.g. vLLM) that accept them. // (e.g. vLLM) that accept them.