Compare commits
4 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 350ef663c0 | |||
| f45ed0d11c | |||
| 8fc743860c | |||
| 0843db20db |
+1
-1
@@ -1,5 +1,5 @@
|
||||
FROM docker.io/library/eclipse-temurin:17-jre-alpine
|
||||
COPY llm-proxy.jar /app/llm-proxy.jar
|
||||
COPY build/libs/llm-proxy.jar /app/llm-proxy.jar
|
||||
WORKDIR /app
|
||||
EXPOSE 8100
|
||||
ENTRYPOINT ["java", "-jar", "/app/llm-proxy.jar"]
|
||||
|
||||
+8
-1
@@ -3,15 +3,22 @@
|
||||
# моделей с суффиксом "-no-think", проксирует на RouterAI. Ответ — как есть.
|
||||
#
|
||||
# Запуск на 76.179 (llm-router): podman-compose up -d
|
||||
# Образ тянется из нашего реестра (zot): images.binom.pw/llm-proxy:<tag>
|
||||
services:
|
||||
llm-proxy:
|
||||
image: images.binom.pw/llm-proxy:latest
|
||||
container_name: llm-proxy
|
||||
restart: unless-stopped
|
||||
network_mode: bifrost_default
|
||||
networks:
|
||||
- bifrost
|
||||
environment:
|
||||
- PORT=8100
|
||||
- UPSTREAM_URL=https://routerai.ru/api/v1
|
||||
- ROUTER_API_KEY=__SET_FROM_CONFIG_DB__
|
||||
- EXCLUDED_PROVIDERS=deepseek
|
||||
- THINKING_MODELS=deepseek/deepseek-v4-flash-0731,deepseek/deepseek-v4-flash
|
||||
|
||||
networks:
|
||||
bifrost:
|
||||
external: true
|
||||
name: bifrost_default
|
||||
|
||||
@@ -103,6 +103,10 @@ private suspend fun handleChat(
|
||||
return
|
||||
}
|
||||
|
||||
val start = System.currentTimeMillis()
|
||||
val model = try {
|
||||
json.parseToJsonElement(patched).jsonObject["model"]?.jsonPrimitive?.content ?: "?"
|
||||
} catch (e: Exception) { "?" }
|
||||
val req = HttpRequest.newBuilder()
|
||||
.uri(URI.create(upstream.trimEnd('/') + "/chat/completions"))
|
||||
.header("Authorization", "Bearer $apiKey")
|
||||
@@ -111,6 +115,7 @@ private suspend fun handleChat(
|
||||
.build()
|
||||
|
||||
val stream = json.parseToJsonElement(raw).jsonObject["stream"]?.jsonPrimitive?.content == "true"
|
||||
try {
|
||||
if (stream) {
|
||||
// SSE-стрим: транслируем как есть, чанк за чанком.
|
||||
val resp = http.send(req, HttpResponse.BodyHandlers.ofInputStream())
|
||||
@@ -118,10 +123,19 @@ private suspend fun handleChat(
|
||||
call.respondOutputStream(ContentType.parse(ct), HttpStatusCode.fromValue(resp.statusCode())) {
|
||||
resp.body().use { input -> input.copyTo(this, 8192) }
|
||||
}
|
||||
println("[llm-proxy] chat model=$model status=${resp.statusCode()} в ${System.currentTimeMillis() - start}ms stream=true")
|
||||
} else {
|
||||
val resp = http.send(req, HttpResponse.BodyHandlers.ofByteArray())
|
||||
val ct = resp.headers().firstValue("content-type").orElse("application/json")
|
||||
call.respondBytes(resp.body(), ContentType.parse(ct), HttpStatusCode.fromValue(resp.statusCode()))
|
||||
println("[llm-proxy] chat model=$model status=${resp.statusCode()} в ${System.currentTimeMillis() - start}ms stream=false")
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
println("[llm-proxy] chat model=$model ОШИБКА: ${e.message} в ${System.currentTimeMillis() - start}ms")
|
||||
call.respondBytes(
|
||||
"""{"error":{"message":"upstream: ${e.message}"}}""".toByteArray(),
|
||||
ContentType.Application.Json, HttpStatusCode.BadGateway,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -141,6 +155,12 @@ internal fun patchBody(raw: String, excluded: List<String>, thinking: List<Strin
|
||||
if (root["reasoning"] == null) {
|
||||
root["reasoning"] = JsonObject(mapOf("enabled" to JsonPrimitive(false)))
|
||||
}
|
||||
// SGLANG_COMPAT: Qwen3.8 глушится только через chat_template_kwargs.enable_thinking=false
|
||||
if (System.getenv("SGLANG_COMPAT") == "true") {
|
||||
val ctk = root["chat_template_kwargs"]?.jsonObject?.toMutableMap() ?: mutableMapOf()
|
||||
ctk["enable_thinking"] = JsonPrimitive(false)
|
||||
root["chat_template_kwargs"] = JsonObject(ctk)
|
||||
}
|
||||
}
|
||||
|
||||
val provider = root["provider"]?.jsonObject?.toMutableMap() ?: mutableMapOf()
|
||||
|
||||
Reference in New Issue
Block a user