feat: llm-proxy — OpenAI-прослойка перед RouterAI (provider.ignore + -no-think)

This commit is contained in:
Hermes Agent
2026-08-20 13:14:52 +03:00
commit 5f3ae73e36
12 changed files with 610 additions and 0 deletions
+196
View File
@@ -0,0 +1,196 @@
package pw.binom.llmproxy
import io.ktor.http.ContentType
import io.ktor.http.HttpStatusCode
import io.ktor.server.application.Application
import io.ktor.server.application.ApplicationCall
import io.ktor.server.application.install
import io.ktor.server.cio.CIO
import io.ktor.server.engine.embeddedServer
import io.ktor.server.request.receiveText
import io.ktor.server.response.respondBytes
import io.ktor.server.response.respondOutputStream
import io.ktor.server.routing.get
import io.ktor.server.routing.post
import io.ktor.server.routing.routing
import kotlinx.serialization.json.Json
import kotlinx.serialization.json.JsonArray
import kotlinx.serialization.json.JsonObject
import kotlinx.serialization.json.JsonPrimitive
import kotlinx.serialization.json.jsonArray
import kotlinx.serialization.json.jsonObject
import kotlinx.serialization.json.jsonPrimitive
import java.net.URI
import java.net.http.HttpClient
import java.net.http.HttpRequest
import java.net.http.HttpResponse
import java.time.Duration
/**
* Прослойка OpenAI API: принимает chat/completions, добавляет в JSON
* `provider.ignore` (исключение дорогих провайдеров) + `allow_fallbacks: false`
* и прозрачно проксирует на RouterAI (routerai.ru/api/v1). Ответ — как есть,
* включая SSE-стрим.
*
* Трюк «-no-think»: модели из [THINKING_MODELS] в каталоге /v1/models дублируются
* с суффиксом `-no-think`; запрос на такой id получает `reasoning: {"enabled": false}`
* (модель не думает — не жрёт токены на reasoning). Проверено на
* deepseek/deepseek-v4-flash-0731: reasoning.enabled=false глушит думанье.
*
* Конфиг (env):
* - PORT (default 8100)
* - UPSTREAM_URL (default https://routerai.ru/api/v1)
* - ROUTER_API_KEY — Bearer-ключ RouterAI (обязателен)
* - EXCLUDED_PROVIDERS — slug'и провайдеров через запятую (напр. "deepseek")
* - THINKING_MODELS — подстроки id «думающих» моделей через запятую
* (напр. "deepseek/deepseek-v4-flash-0731,deepseek/deepseek-r1")
*/
fun main() {
val port = System.getenv("PORT")?.toIntOrNull() ?: 8100
val upstream = System.getenv("UPSTREAM_URL") ?: "https://routerai.ru/api/v1"
val apiKey = System.getenv("ROUTER_API_KEY")
?: throw IllegalStateException("ROUTER_API_KEY required")
val excluded = (System.getenv("EXCLUDED_PROVIDERS") ?: "")
.split(",").map { it.trim() }.filter { it.isNotEmpty() }.distinct()
val thinking = (System.getenv("THINKING_MODELS") ?: "deepseek/deepseek-v4-flash-0731,deepseek/deepseek-v4-flash")
.split(",").map { it.trim() }.filter { it.isNotEmpty() }.distinct()
embeddedServer(CIO, port = port, host = "0.0.0.0") {
proxyModule(upstream, apiKey, excluded, thinking)
}.start(wait = true)
}
private val json = Json { ignoreUnknownKeys = true }
/** Прокси-клиент: один на процесс (HttpClient потокобезопасен). */
private val http = HttpClient.newBuilder()
.connectTimeout(Duration.ofSeconds(15))
.build()
fun Application.proxyModule(upstream: String, apiKey: String, excluded: List<String>, thinking: List<String>) {
routing {
post("/v1/chat/completions") {
handleChat(call, upstream, apiKey, excluded, thinking)
}
get("/v1/models") {
handleModels(call, upstream, apiKey, thinking)
}
}
}
private suspend fun handleChat(
call: ApplicationCall,
upstream: String,
apiKey: String,
excluded: List<String>,
thinking: List<String>,
) {
val raw = call.receiveText()
if (raw.isBlank()) {
call.respondBytes(
"""{"error":{"message":"empty body"}}""".toByteArray(),
ContentType.Application.Json, HttpStatusCode.BadRequest,
)
return
}
val patched = try {
patchBody(raw, excluded, thinking)
} catch (e: Exception) {
call.respondBytes(
"""{"error":{"message":"bad json: ${e.message}"}}""".toByteArray(),
ContentType.Application.Json, HttpStatusCode.BadRequest,
)
return
}
val req = HttpRequest.newBuilder()
.uri(URI.create(upstream.trimEnd('/') + "/chat/completions"))
.header("Authorization", "Bearer $apiKey")
.header("Content-Type", "application/json")
.POST(HttpRequest.BodyPublishers.ofString(patched))
.build()
val stream = json.parseToJsonElement(raw).jsonObject["stream"]?.jsonPrimitive?.content == "true"
if (stream) {
// SSE-стрим: транслируем как есть, чанк за чанком.
val resp = http.send(req, HttpResponse.BodyHandlers.ofInputStream())
val ct = resp.headers().firstValue("content-type").orElse("text/event-stream")
call.respondOutputStream(ContentType.parse(ct), HttpStatusCode.fromValue(resp.statusCode())) {
resp.body().use { input -> input.copyTo(this, 8192) }
}
} else {
val resp = http.send(req, HttpResponse.BodyHandlers.ofByteArray())
val ct = resp.headers().firstValue("content-type").orElse("application/json")
call.respondBytes(resp.body(), ContentType.parse(ct), HttpStatusCode.fromValue(resp.statusCode()))
}
}
/**
* Патч запроса: (1) если модель оканчивается на "-no-think" — снять суффикс и
* добавить `reasoning: {"enabled": false}` (не думать); (2) добавить
* `provider.ignore` (объединяя с присланным клиентом) и `allow_fallbacks: false` —
* иначе ignore не жёсткий: RouterAI может уйти на исключённого провайдера
* резервной попыткой (см. гайд provider-selection).
*/
internal fun patchBody(raw: String, excluded: List<String>, thinking: List<String>): String {
val root = json.parseToJsonElement(raw).jsonObject.toMutableMap()
val model = root["model"]?.jsonPrimitive?.content ?: ""
if (model.endsWith("-no-think")) {
root["model"] = JsonPrimitive(model.removeSuffix("-no-think"))
if (root["reasoning"] == null) {
root["reasoning"] = JsonObject(mapOf("enabled" to JsonPrimitive(false)))
}
}
val provider = root["provider"]?.jsonObject?.toMutableMap() ?: mutableMapOf()
val existing = provider["ignore"]?.jsonArray?.map { it.jsonPrimitive.content } ?: emptyList()
provider["ignore"] = JsonArray((existing + excluded).distinct().map { JsonPrimitive(it) })
if (provider["allow_fallbacks"] == null) {
provider["allow_fallbacks"] = JsonPrimitive(false)
}
root["provider"] = JsonObject(provider)
return JsonObject(root).toString()
}
private suspend fun handleModels(
call: ApplicationCall,
upstream: String,
apiKey: String,
thinking: List<String>,
) {
val req = HttpRequest.newBuilder()
.uri(URI.create(upstream.trimEnd('/') + "/models"))
.header("Authorization", "Bearer $apiKey")
.GET()
.build()
val resp = http.send(req, HttpResponse.BodyHandlers.ofByteArray())
val ct = resp.headers().firstValue("content-type").orElse("application/json")
val body = if (thinking.isNotEmpty()) {
patchModelsCatalog(String(resp.body(), Charsets.UTF_8), thinking).toByteArray(Charsets.UTF_8)
} else {
resp.body()
}
call.respondBytes(body, ContentType.parse(ct), HttpStatusCode.fromValue(resp.statusCode()))
}
/**
* Дублировать «думающие» модели в каталоге с суффиксом "-no-think":
* каждая модель, чей id содержит любую из подстрок [thinking], получает копию
* с id = "<оригинал>-no-think".
*/
internal fun patchModelsCatalog(raw: String, thinking: List<String>): String {
val root = json.parseToJsonElement(raw).jsonObject.toMutableMap()
val data = root["data"]?.jsonArray?.map { it.jsonObject } ?: emptyList()
if (data.isEmpty()) return raw
val copies = data.filter { m ->
val id = m["id"]?.jsonPrimitive?.content ?: ""
thinking.any { id.contains(it) }
}.map { m ->
val id = m["id"]?.jsonPrimitive?.content ?: ""
JsonObject(m.toMutableMap().apply { this["id"] = JsonPrimitive(id + "-no-think") })
}
if (copies.isEmpty()) return raw
root["data"] = JsonArray(data + copies)
return JsonObject(root).toString()
}