adaptor.go 12 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402
  1. package gemini
  2. import (
  3. "encoding/json"
  4. "errors"
  5. "fmt"
  6. "io"
  7. "net/http"
  8. "strings"
  9. "github.com/QuantumNous/new-api/dto"
  10. "github.com/QuantumNous/new-api/relay/channel"
  11. "github.com/QuantumNous/new-api/relay/channel/openai"
  12. relaycommon "github.com/QuantumNous/new-api/relay/common"
  13. "github.com/QuantumNous/new-api/relay/constant"
  14. "github.com/QuantumNous/new-api/setting/model_setting"
  15. "github.com/QuantumNous/new-api/types"
  16. "github.com/gin-gonic/gin"
  17. )
  18. type Adaptor struct {
  19. }
  20. func (a *Adaptor) ConvertGeminiRequest(c *gin.Context, info *relaycommon.RelayInfo, request *dto.GeminiChatRequest) (any, error) {
  21. if len(request.Contents) > 0 {
  22. for i, content := range request.Contents {
  23. if i == 0 {
  24. if request.Contents[0].Role == "" {
  25. request.Contents[0].Role = "user"
  26. }
  27. }
  28. for _, part := range content.Parts {
  29. if part.FileData != nil {
  30. if part.FileData.MimeType == "" && strings.Contains(part.FileData.FileUri, "www.youtube.com") {
  31. part.FileData.MimeType = "video/webm"
  32. }
  33. }
  34. }
  35. }
  36. }
  37. return request, nil
  38. }
  39. func (a *Adaptor) ConvertClaudeRequest(c *gin.Context, info *relaycommon.RelayInfo, req *dto.ClaudeRequest) (any, error) {
  40. adaptor := openai.Adaptor{}
  41. oaiReq, err := adaptor.ConvertClaudeRequest(c, info, req)
  42. if err != nil {
  43. return nil, err
  44. }
  45. return a.ConvertOpenAIRequest(c, info, oaiReq.(*dto.GeneralOpenAIRequest))
  46. }
  47. func (a *Adaptor) ConvertAudioRequest(c *gin.Context, info *relaycommon.RelayInfo, request dto.AudioRequest) (io.Reader, error) {
  48. //TODO implement me
  49. return nil, errors.New("not implemented")
  50. }
  51. type ImageConfig struct {
  52. AspectRatio string `json:"aspectRatio,omitempty"`
  53. ImageSize string `json:"imageSize,omitempty"`
  54. }
  55. type SizeMapping struct {
  56. AspectRatio string
  57. ImageSize string
  58. }
  59. type QualityMapping struct {
  60. Standard string
  61. HD string
  62. High string
  63. FourK string
  64. Auto string
  65. }
  66. func getImageSizeMapping() QualityMapping {
  67. return QualityMapping{
  68. Standard: "1K",
  69. HD: "2K",
  70. High: "2K",
  71. FourK: "4K",
  72. Auto: "1K",
  73. }
  74. }
  75. func getSizeMappings() map[string]SizeMapping {
  76. return map[string]SizeMapping{
  77. // Gemini 2.5 Flash Image - default 1K resolutions
  78. "1024x1024": {AspectRatio: "1:1", ImageSize: ""},
  79. "832x1248": {AspectRatio: "2:3", ImageSize: ""},
  80. "1248x832": {AspectRatio: "3:2", ImageSize: ""},
  81. "864x1184": {AspectRatio: "3:4", ImageSize: ""},
  82. "1184x864": {AspectRatio: "4:3", ImageSize: ""},
  83. "896x1152": {AspectRatio: "4:5", ImageSize: ""},
  84. "1152x896": {AspectRatio: "5:4", ImageSize: ""},
  85. "768x1344": {AspectRatio: "9:16", ImageSize: ""},
  86. "1344x768": {AspectRatio: "16:9", ImageSize: ""},
  87. "1536x672": {AspectRatio: "21:9", ImageSize: ""},
  88. // Gemini 3 Pro Image Preview resolutions
  89. "1536x1024": {AspectRatio: "3:2", ImageSize: ""},
  90. "1024x1536": {AspectRatio: "2:3", ImageSize: ""},
  91. "1024x1792": {AspectRatio: "9:16", ImageSize: ""},
  92. "1792x1024": {AspectRatio: "16:9", ImageSize: ""},
  93. "2048x2048": {AspectRatio: "1:1", ImageSize: "2K"},
  94. "4096x4096": {AspectRatio: "1:1", ImageSize: "4K"},
  95. }
  96. }
  97. func processSizeParameters(size, quality string) ImageConfig {
  98. config := ImageConfig{} // 默认为空值
  99. if size != "" {
  100. if strings.Contains(size, ":") {
  101. config.AspectRatio = size // 直接设置,不与默认值比较
  102. } else {
  103. if mapping, exists := getSizeMappings()[size]; exists {
  104. if mapping.AspectRatio != "" {
  105. config.AspectRatio = mapping.AspectRatio
  106. }
  107. if mapping.ImageSize != "" {
  108. config.ImageSize = mapping.ImageSize
  109. }
  110. }
  111. }
  112. }
  113. if quality != "" {
  114. qualityMapping := getImageSizeMapping()
  115. switch strings.ToLower(strings.TrimSpace(quality)) {
  116. case "hd", "high":
  117. config.ImageSize = qualityMapping.HD
  118. case "4k":
  119. config.ImageSize = qualityMapping.FourK
  120. case "standard", "medium", "low", "auto", "1k":
  121. config.ImageSize = qualityMapping.Standard
  122. }
  123. }
  124. return config
  125. }
  126. func (a *Adaptor) ConvertImageRequest(c *gin.Context, info *relaycommon.RelayInfo, request dto.ImageRequest) (any, error) {
  127. if model_setting.IsGeminiModelSupportImagine(info.UpstreamModelName) {
  128. var content any
  129. if base64Data, err := relaycommon.GetImageBase64sFromForm(c); err == nil {
  130. content = []any{
  131. dto.MediaContent{
  132. Type: dto.ContentTypeText,
  133. Text: request.Prompt,
  134. },
  135. dto.MediaContent{
  136. Type: dto.ContentTypeFile,
  137. File: &dto.MessageFile{
  138. FileData: base64Data.String(),
  139. },
  140. },
  141. }
  142. } else {
  143. content = request.Prompt
  144. }
  145. chatRequest := dto.GeneralOpenAIRequest{
  146. Model: request.Model,
  147. Messages: []dto.Message{
  148. {Role: "user", Content: content},
  149. },
  150. N: int(request.N),
  151. }
  152. config := processSizeParameters(strings.TrimSpace(request.Size), request.Quality)
  153. googleGenerationConfig := map[string]interface{}{
  154. "responseModalities": []string{"TEXT", "IMAGE"},
  155. "imageConfig": config,
  156. }
  157. extraBody := map[string]interface{}{
  158. "google": map[string]interface{}{
  159. "generationConfig": googleGenerationConfig,
  160. },
  161. }
  162. chatRequest.ExtraBody, _ = json.Marshal(extraBody)
  163. return a.ConvertOpenAIRequest(c, info, &chatRequest)
  164. }
  165. // convert size to aspect ratio but allow user to specify aspect ratio
  166. aspectRatio := "1:1" // default aspect ratio
  167. size := strings.TrimSpace(request.Size)
  168. if size != "" {
  169. if strings.Contains(size, ":") {
  170. aspectRatio = size
  171. } else {
  172. if mapping, exists := getSizeMappings()[size]; exists && mapping.AspectRatio != "" {
  173. aspectRatio = mapping.AspectRatio
  174. }
  175. }
  176. }
  177. // build gemini imagen request
  178. geminiRequest := dto.GeminiImageRequest{
  179. Instances: []dto.GeminiImageInstance{
  180. {
  181. Prompt: request.Prompt,
  182. },
  183. },
  184. Parameters: dto.GeminiImageParameters{
  185. SampleCount: int(request.N),
  186. AspectRatio: aspectRatio,
  187. PersonGeneration: "allow_adult", // default allow adult
  188. },
  189. }
  190. // Set imageSize when quality parameter is specified
  191. // Map quality parameter to imageSize (only supported by Standard and Ultra models)
  192. // quality values: auto, high, medium, low (for gpt-image-1), hd, standard (for dall-e-3)
  193. // imageSize values: 1K (default), 2K
  194. // https://ai.google.dev/gemini-api/docs/imagen
  195. // https://platform.openai.com/docs/api-reference/images/create
  196. if request.Quality != "" {
  197. imageSize := "1K" // default
  198. switch request.Quality {
  199. case "hd", "high":
  200. imageSize = "2K"
  201. case "2K":
  202. imageSize = "2K"
  203. case "standard", "medium", "low", "auto", "1K":
  204. imageSize = "1K"
  205. default:
  206. // unknown quality value, default to 1K
  207. imageSize = "1K"
  208. }
  209. geminiRequest.Parameters.ImageSize = imageSize
  210. }
  211. return geminiRequest, nil
  212. }
  213. func (a *Adaptor) Init(info *relaycommon.RelayInfo) {
  214. }
  215. func (a *Adaptor) GetRequestURL(info *relaycommon.RelayInfo) (string, error) {
  216. if model_setting.GetGeminiSettings().ThinkingAdapterEnabled &&
  217. !model_setting.ShouldPreserveThinkingSuffix(info.OriginModelName) {
  218. // 新增逻辑:处理 -thinking-<budget> 格式
  219. if strings.Contains(info.UpstreamModelName, "-thinking-") {
  220. parts := strings.Split(info.UpstreamModelName, "-thinking-")
  221. info.UpstreamModelName = parts[0]
  222. } else if strings.HasSuffix(info.UpstreamModelName, "-thinking") { // 旧的适配
  223. info.UpstreamModelName = strings.TrimSuffix(info.UpstreamModelName, "-thinking")
  224. } else if strings.HasSuffix(info.UpstreamModelName, "-nothinking") {
  225. info.UpstreamModelName = strings.TrimSuffix(info.UpstreamModelName, "-nothinking")
  226. }
  227. }
  228. version := model_setting.GetGeminiVersionSetting(info.UpstreamModelName)
  229. if strings.HasPrefix(info.UpstreamModelName, "imagen") {
  230. return fmt.Sprintf("%s/%s/models/%s:predict", info.ChannelBaseUrl, version, info.UpstreamModelName), nil
  231. }
  232. if strings.HasPrefix(info.UpstreamModelName, "text-embedding") ||
  233. strings.HasPrefix(info.UpstreamModelName, "embedding") ||
  234. strings.HasPrefix(info.UpstreamModelName, "gemini-embedding") {
  235. action := "embedContent"
  236. if info.IsGeminiBatchEmbedding {
  237. action = "batchEmbedContents"
  238. }
  239. return fmt.Sprintf("%s/%s/models/%s:%s", info.ChannelBaseUrl, version, info.UpstreamModelName, action), nil
  240. }
  241. action := "generateContent"
  242. if info.IsStream {
  243. action = "streamGenerateContent?alt=sse"
  244. if info.RelayMode == constant.RelayModeGemini {
  245. info.DisablePing = true
  246. }
  247. }
  248. return fmt.Sprintf("%s/%s/models/%s:%s", info.ChannelBaseUrl, version, info.UpstreamModelName, action), nil
  249. }
  250. func (a *Adaptor) SetupRequestHeader(c *gin.Context, req *http.Header, info *relaycommon.RelayInfo) error {
  251. channel.SetupApiRequestHeader(info, c, req)
  252. req.Set("x-goog-api-key", info.ApiKey)
  253. return nil
  254. }
  255. func (a *Adaptor) ConvertOpenAIRequest(c *gin.Context, info *relaycommon.RelayInfo, request *dto.GeneralOpenAIRequest) (any, error) {
  256. if request == nil {
  257. return nil, errors.New("request is nil")
  258. }
  259. geminiRequest, err := CovertOpenAI2Gemini(c, *request, info)
  260. if err != nil {
  261. return nil, err
  262. }
  263. return geminiRequest, nil
  264. }
  265. func (a *Adaptor) ConvertRerankRequest(c *gin.Context, relayMode int, request dto.RerankRequest) (any, error) {
  266. return nil, nil
  267. }
  268. func (a *Adaptor) ConvertEmbeddingRequest(c *gin.Context, info *relaycommon.RelayInfo, request dto.EmbeddingRequest) (any, error) {
  269. if request.Input == nil {
  270. return nil, errors.New("input is required")
  271. }
  272. inputs := request.ParseInput()
  273. if len(inputs) == 0 {
  274. return nil, errors.New("input is empty")
  275. }
  276. // We always build a batch-style payload with `requests`, so ensure we call the
  277. // batch endpoint upstream to avoid payload/endpoint mismatches.
  278. info.IsGeminiBatchEmbedding = true
  279. // process all inputs
  280. geminiRequests := make([]map[string]interface{}, 0, len(inputs))
  281. for _, input := range inputs {
  282. geminiRequest := map[string]interface{}{
  283. "model": fmt.Sprintf("models/%s", info.UpstreamModelName),
  284. "content": dto.GeminiChatContent{
  285. Parts: []dto.GeminiPart{
  286. {
  287. Text: input,
  288. },
  289. },
  290. },
  291. }
  292. // set specific parameters for different models
  293. // https://ai.google.dev/api/embeddings?hl=zh-cn#method:-models.embedcontent
  294. switch info.UpstreamModelName {
  295. case "text-embedding-004", "gemini-embedding-exp-03-07", "gemini-embedding-001":
  296. // Only newer models introduced after 2024 support OutputDimensionality
  297. if request.Dimensions > 0 {
  298. geminiRequest["outputDimensionality"] = request.Dimensions
  299. }
  300. }
  301. geminiRequests = append(geminiRequests, geminiRequest)
  302. }
  303. return map[string]interface{}{
  304. "requests": geminiRequests,
  305. }, nil
  306. }
  307. func (a *Adaptor) ConvertOpenAIResponsesRequest(c *gin.Context, info *relaycommon.RelayInfo, request dto.OpenAIResponsesRequest) (any, error) {
  308. // TODO implement me
  309. return nil, errors.New("not implemented")
  310. }
  311. func (a *Adaptor) DoRequest(c *gin.Context, info *relaycommon.RelayInfo, requestBody io.Reader) (any, error) {
  312. return channel.DoApiRequest(a, c, info, requestBody)
  313. }
  314. func (a *Adaptor) DoResponse(c *gin.Context, resp *http.Response, info *relaycommon.RelayInfo) (usage any, err *types.NewAPIError) {
  315. if info.RelayMode == constant.RelayModeGemini {
  316. if strings.Contains(info.RequestURLPath, ":embedContent") ||
  317. strings.Contains(info.RequestURLPath, ":batchEmbedContents") {
  318. return NativeGeminiEmbeddingHandler(c, resp, info)
  319. }
  320. if info.IsStream {
  321. return GeminiTextGenerationStreamHandler(c, info, resp)
  322. } else {
  323. return GeminiTextGenerationHandler(c, info, resp)
  324. }
  325. }
  326. if strings.HasPrefix(info.UpstreamModelName, "imagen") {
  327. return GeminiImageHandler(c, info, resp)
  328. }
  329. if model_setting.IsGeminiModelSupportImagine(info.UpstreamModelName) {
  330. return ChatImageHandler(c, info, resp)
  331. }
  332. // check if the model is an embedding model
  333. if strings.HasPrefix(info.UpstreamModelName, "text-embedding") ||
  334. strings.HasPrefix(info.UpstreamModelName, "embedding") ||
  335. strings.HasPrefix(info.UpstreamModelName, "gemini-embedding") {
  336. return GeminiEmbeddingHandler(c, info, resp)
  337. }
  338. if info.IsStream {
  339. return GeminiChatStreamHandler(c, info, resp)
  340. } else {
  341. return GeminiChatHandler(c, info, resp)
  342. }
  343. }
  344. func (a *Adaptor) GetModelList() []string {
  345. return ModelList
  346. }
  347. func (a *Adaptor) GetChannelName() string {
  348. return ChannelName
  349. }