|
| 1 | +// ABOUTME: Together AI provider — fast open-source model inference |
| 2 | +// ABOUTME: Extends OpenAICompatibleProvider with Together-specific reasoning and thinking extraction |
| 3 | +package org.nlogo.extensions.llm.providers |
| 4 | + |
| 5 | +import org.nlogo.extensions.llm.models.ChatRequest |
| 6 | +import scala.concurrent.ExecutionContext |
| 7 | + |
| 8 | +/** |
| 9 | + * Together AI provider implementation. |
| 10 | + * |
| 11 | + * Together AI provides fast inference for open-source models (Llama, DeepSeek, |
| 12 | + * Qwen, Gemma, Mistral, etc.) via an OpenAI-compatible API. |
| 13 | + * |
| 14 | + * Key differences from direct OpenAI: |
| 15 | + * - Model names are vendor-prefixed (e.g. "meta-llama/Llama-3.3-70B-Instruct-Turbo") |
| 16 | + * - Hybrid reasoning models use { reasoning: { enabled: true } } |
| 17 | + * - DeepSeek-R1 embeds thinking in <think>...</think> tags within content |
| 18 | + */ |
| 19 | +class TogetherProvider(implicit ec: ExecutionContext) extends OpenAICompatibleProvider { |
| 20 | + |
| 21 | + override def providerName: String = "together" |
| 22 | + |
| 23 | + override def defaultModel: String = "meta-llama/Llama-3.3-70B-Instruct-Turbo" |
| 24 | + |
| 25 | + override protected def defaultBaseUrl: String = "https://api.together.xyz/v1" |
| 26 | + |
| 27 | + override protected def baseUrlConfigKey: String = "together_base_url" |
| 28 | + |
| 29 | + override protected def apiKeyConfigKey: String = "together_api_key" |
| 30 | + |
| 31 | + override protected def defaultMaxTokens: String = "1000" |
| 32 | + |
| 33 | + override protected def requiresApiKey: Boolean = true |
| 34 | + |
| 35 | + // No extra headers needed (unlike OpenRouter) |
| 36 | + |
| 37 | + /** |
| 38 | + * Together hybrid models use { reasoning: { enabled: true } }. |
| 39 | + * Adjustable-effort models use top-level reasoning_effort (the default from base class). |
| 40 | + * We send both when thinking is enabled — the API ignores unrecognized fields. |
| 41 | + */ |
| 42 | + override protected def applyReasoningFields(baseObj: ujson.Obj, request: ChatRequest): Unit = { |
| 43 | + // Enable reasoning for hybrid models |
| 44 | + baseObj("reasoning") = ujson.Obj("enabled" -> true) |
| 45 | + // Also pass effort level if specified (for adjustable-effort models) |
| 46 | + request.thinkingConfig.flatMap(_.reasoningEffort).foreach { effort => |
| 47 | + baseObj("reasoning_effort") = effort |
| 48 | + } |
| 49 | + } |
| 50 | + |
| 51 | + /** |
| 52 | + * Extract thinking text from Together AI responses. |
| 53 | + * |
| 54 | + * Two extraction paths: |
| 55 | + * 1. message.reasoning field (most reasoning models) |
| 56 | + * 2. <think>...</think> tags in message.content (DeepSeek-R1) |
| 57 | + */ |
| 58 | + override protected def extractThinking(message: ujson.Value): Option[String] = { |
| 59 | + // Path 1: check message.reasoning field |
| 60 | + val fromReasoning = try { |
| 61 | + message.obj.get("reasoning").flatMap { v => |
| 62 | + val text = v.str.trim |
| 63 | + if (text.nonEmpty) Some(text) else None |
| 64 | + } |
| 65 | + } catch { |
| 66 | + case _: Exception => None |
| 67 | + } |
| 68 | + |
| 69 | + if (fromReasoning.isDefined) return fromReasoning |
| 70 | + |
| 71 | + // Path 2: parse <think>...</think> tags from content (DeepSeek-R1) |
| 72 | + try { |
| 73 | + val content = message("content").str |
| 74 | + val thinkPattern = """(?s)<think>(.*?)</think>""".r |
| 75 | + thinkPattern.findFirstMatchIn(content).map(_.group(1).trim).filter(_.nonEmpty) |
| 76 | + } catch { |
| 77 | + case _: Exception => None |
| 78 | + } |
| 79 | + } |
| 80 | +} |
0 commit comments