Upgrade litert-api/litert-google/litert-openai to v7 (litertlm-jvm 0.17.0), drop workarounds
This commit is contained in:
@@ -71,7 +71,7 @@ class ChatAgent(
|
||||
)
|
||||
}
|
||||
}
|
||||
val conv = ChatConversation(record = rec, stores = stores, llm = llm, systemPrompt = llmConfig.systemPrompt, foldSystemIntoFirstUser = llmConfig.foldSystemIntoFirstUser)
|
||||
val conv = ChatConversation(record = rec, stores = stores, llm = llm, systemPrompt = llmConfig.systemPrompt)
|
||||
runBlocking {
|
||||
liveLock.withLock { live[conv.id] = conv }
|
||||
}
|
||||
@@ -82,7 +82,7 @@ class ChatAgent(
|
||||
override suspend fun getConversation(id: String): ProtoConversation? {
|
||||
liveLock.withLock { live[id] }?.let { if (!it.isClosed) return it }
|
||||
val rec = stores.conversations.get(id) ?: return null
|
||||
return ChatConversation(record = rec, stores = stores, llm = llm, systemPrompt = llmConfig.systemPrompt, foldSystemIntoFirstUser = llmConfig.foldSystemIntoFirstUser).also {
|
||||
return ChatConversation(record = rec, stores = stores, llm = llm, systemPrompt = llmConfig.systemPrompt).also {
|
||||
liveLock.withLock { live[id] = it }
|
||||
}
|
||||
}
|
||||
|
||||
+3
-26
@@ -9,7 +9,6 @@ import kotlinx.coroutines.channels.BufferOverflow
|
||||
import kotlinx.coroutines.flow.Flow
|
||||
import kotlinx.coroutines.flow.MutableSharedFlow
|
||||
import kotlinx.coroutines.flow.asSharedFlow
|
||||
import kotlinx.coroutines.flow.transformWhile
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.sync.Mutex
|
||||
import kotlinx.coroutines.sync.withLock
|
||||
@@ -59,7 +58,6 @@ class ChatConversation(
|
||||
private val stores: SqliteStores,
|
||||
private val llm: LiteLlm,
|
||||
private val systemPrompt: String,
|
||||
private val foldSystemIntoFirstUser: Boolean = false,
|
||||
) : ProtoConversation, AutoCloseable {
|
||||
|
||||
private var record: ConversationRecord = record
|
||||
@@ -184,12 +182,7 @@ class ChatConversation(
|
||||
|
||||
val reply = StringBuilder()
|
||||
try {
|
||||
// Некоторые бэкенды (litert-google-jvm 0.16.1) не закрывают стрим после `isDone = true`,
|
||||
// поэтому заворачиваем в transformWhile: видим isDone → отдаём дельту и завершаем flow.
|
||||
liteConv.sendStreamContents(parts).transformWhile { delta ->
|
||||
emit(delta)
|
||||
!delta.isDone
|
||||
}.collect { delta ->
|
||||
liteConv.sendStreamContents(parts).collect { delta ->
|
||||
if (delta.text.isNotEmpty()) {
|
||||
reply.append(delta.text)
|
||||
emitEvent(ProtoEvent.AppendText(date = now(), body = delta.text))
|
||||
@@ -238,9 +231,6 @@ class ChatConversation(
|
||||
* уже записанное в audit + working memory, но ещё не отправленное в LLM — мы отдадим
|
||||
* его через [LiteConversation.sendStreamContents]). Это предотвращает дублирование
|
||||
* "user → user" в LiteConversation history.
|
||||
*
|
||||
* Если [foldSystemIntoFirstUser] — система не передаётся как `systemInstruction`,
|
||||
* а фолдится в первое user-сообщение (нужно для Gemma-3 шаблона LiteRT-LM 0.16.1).
|
||||
*/
|
||||
private suspend fun getOrCreateLiteConversation(excludeUserSourceId: String? = null): LiteConversation {
|
||||
liteConv?.let { return it }
|
||||
@@ -268,21 +258,8 @@ class ChatConversation(
|
||||
}
|
||||
|
||||
val config = LiteConversationConfig(
|
||||
systemInstruction = if (foldSystemIntoFirstUser) null else resolvedSystemPrompt.takeIf { it.isNotBlank() },
|
||||
initialMessages = if (foldSystemIntoFirstUser) {
|
||||
val firstUser = pastTurns.firstOrNull { it.role == LiteRole.USER }
|
||||
if (firstUser != null && resolvedSystemPrompt.isNotBlank()) {
|
||||
val folded = LiteMessage(
|
||||
role = LiteRole.USER,
|
||||
contents = listOf(LiteContentPart.Text(resolvedSystemPrompt + "\n\n")) + firstUser.contents,
|
||||
)
|
||||
listOf(folded) + pastTurns.drop(1)
|
||||
} else {
|
||||
pastTurns
|
||||
}
|
||||
} else {
|
||||
pastTurns
|
||||
},
|
||||
systemInstruction = resolvedSystemPrompt.takeIf { it.isNotBlank() },
|
||||
initialMessages = pastTurns,
|
||||
)
|
||||
|
||||
return llm.createConversation(config).also { liteConv = it }
|
||||
|
||||
@@ -12,7 +12,6 @@ data class LlmConfig(
|
||||
val systemPrompt: String,
|
||||
val openai: LitertOpenAiConfig? = null,
|
||||
val google: GoogleConfig? = null,
|
||||
val foldSystemIntoFirstUser: Boolean = backend == LlmBackend.GOOGLE,
|
||||
) {
|
||||
init {
|
||||
when (backend) {
|
||||
|
||||
@@ -326,6 +326,7 @@ class ChatAgentTest {
|
||||
|
||||
private fun fakeLiteLlmForReload(): LiteLlm = object : LiteLlm {
|
||||
override val backendName: String = "fake"
|
||||
override val capabilities: pw.binom.litert.LiteCapabilities? = null
|
||||
override fun isInitialized(): Boolean = true
|
||||
override fun createConversation(config: LiteConversationConfig): LiteConversation =
|
||||
error("not used in reload test")
|
||||
@@ -357,6 +358,7 @@ class ChatAgentTest {
|
||||
/** Поддельный LiteLlm: возвращает fakeLlm.reply в sendStreamContents, опционально запоминает history. */
|
||||
private class FakeLiteLlm : LiteLlm {
|
||||
override val backendName: String = "fake"
|
||||
override val capabilities: pw.binom.litert.LiteCapabilities? = null
|
||||
var reply: String = ""
|
||||
var rememberHistory: Boolean = false
|
||||
var slow: Boolean = false
|
||||
|
||||
Reference in New Issue
Block a user