fix: workflow inference chain — surface llama-server errors, user-turn-last, bundle-relative prompts

- LlamaCppInferenceProvider: check HTTP status and surface the real error body instead of masking it as a ChatCompletionResponse deserialization failure
- PromptRenderer: render the live conversation layer (L1) last so the user turn is the final message (fixes Qwen 'No user query found' 500 when L3 recall is present)
- TomlWorkflowLoader: resolve stage prompts relative to the workflow file's own dir (bundle), CWD-independent; config-dir fallback for globals
- SessionOrchestrator: hard-fail a stage whose declared prompt can't resolve (was a silent user-less request)
- move healthcheck prompts next to the example workflow (examples/workflows/prompts/)
This commit is contained in:
2026-06-01 23:23:29 +04:00
parent 55a18b4b3a
commit 94f7ad0ee9
6 changed files with 50 additions and 23 deletions
@@ -12,6 +12,14 @@ data class ChatMessage(
)
object PromptRenderer {
// L1 (live conversation, ends with the current user turn) must render last so
// the model template sees the user query as the final message. Background/memory
// layers (L2, L3, L4) are injected before it.
private val nonL0RenderOrder: Comparator<ContextLayer> = Comparator { a, b ->
val priority = { layer: ContextLayer -> if (layer == ContextLayer.L1) Int.MAX_VALUE else layer.ordinal }
priority(a) - priority(b)
}
fun render(contextPack: ContextPack): List<ChatMessage> {
val sorted = contextPack.layers.entries.sortedBy { it.key.ordinal }
val systemContent = sorted
@@ -19,19 +27,22 @@ object PromptRenderer {
.flatMap { it.value }
.joinToString("\n\n") { it.content }
.takeIf { it.isNotBlank() }
val conversationMessages = sorted.filter { it.key != ContextLayer.L0 }.flatMap { (_, entries) ->
entries.map { entry ->
ChatMessage(
role = when (entry.role) {
EntryRole.SYSTEM -> "system"
EntryRole.ASSISTANT -> "assistant"
EntryRole.TOOL -> "tool"
EntryRole.USER -> "user"
},
content = entry.content,
)
val conversationMessages = contextPack.layers.entries
.filter { it.key != ContextLayer.L0 }
.sortedWith(Comparator.comparing({ it.key }, nonL0RenderOrder))
.flatMap { (_, entries) ->
entries.map { entry ->
ChatMessage(
role = when (entry.role) {
EntryRole.SYSTEM -> "system"
EntryRole.ASSISTANT -> "assistant"
EntryRole.TOOL -> "tool"
EntryRole.USER -> "user"
},
content = entry.content,
)
}
}
}
val messages = buildList {
systemContent?.let { add(ChatMessage("system", it)) }
addAll(conversationMessages)
@@ -222,7 +222,7 @@ abstract class SessionOrchestrator(
} ?: emptyList()
val promptEntries = stageConfig.metadata["prompt"]
?.let { path ->
runCatching { promptResolver.resolve(path) }
val resolvedText = runCatching { promptResolver.resolve(path) }
.onFailure {
log.error(
"[SessionOrchestrator] stage=${stageId.value}: " +
@@ -230,9 +230,11 @@ abstract class SessionOrchestrator(
)
}
.getOrNull()
}
?.takeIf { it.isNotBlank() }
?.let { text ->
val text = resolvedText?.takeIf { it.isNotBlank() }
?: error(
"[SessionOrchestrator] stage=${stageId.value}: " +
"declared prompt '$path' could not be resolved",
)
listOf(
ContextEntry(
id = ContextEntryId(UUID.randomUUID().toString()),
@@ -21,6 +21,7 @@ import io.ktor.client.request.accept
import io.ktor.client.request.get
import io.ktor.client.request.post
import io.ktor.client.request.setBody
import io.ktor.client.statement.bodyAsText
import io.ktor.http.ContentType
import io.ktor.http.contentType
import io.ktor.serialization.kotlinx.json.json
@@ -106,11 +107,16 @@ class LlamaCppInferenceProvider(
val encoded = json.encodeToString(body)
log.debug("sending request to llm: {}", encoded)
val response = httpClient.post("$baseUrl/v1/chat/completions") {
val httpResponse = httpClient.post("$baseUrl/v1/chat/completions") {
contentType(ContentType.Application.Json)
accept(ContentType.Application.Json)
setBody(encoded)
}.body<ChatCompletionResponse>()
}
if (httpResponse.status.value !in 200..299) {
val errorBody = httpResponse.bodyAsText()
error("llama-server returned ${httpResponse.status.value} ${httpResponse.status.description}: $errorBody")
}
val response = httpResponse.body<ChatCompletionResponse>()
log.debug("got response from llm: {}", response)
@@ -15,6 +15,7 @@ import com.fasterxml.jackson.dataformat.toml.TomlMapper
import com.fasterxml.jackson.module.kotlin.kotlinModule
import com.fasterxml.jackson.module.kotlin.readValue
import java.nio.file.Path
import kotlin.io.path.exists
import kotlin.io.path.readText
private data class WorkflowFile(
@@ -67,10 +68,17 @@ class TomlWorkflowLoader(
val raw = path.readText()
val file = runCatching { mapper.readValue<WorkflowFile>(raw) }
.getOrElse { throw WorkflowValidationException("Failed to parse workflow TOML: ${it.message}") }
return file.toWorkflowGraph()
return file.toWorkflowGraph(path.parent)
}
private fun WorkflowFile.toWorkflowGraph(): WorkflowGraph {
private fun resolvePromptPath(raw: String, workflowDir: Path?): String {
val p = Path.of(raw)
if (p.isAbsolute) return raw
val candidate = workflowDir?.resolve(raw)
return if (candidate != null && candidate.exists()) candidate.toString() else raw
}
private fun WorkflowFile.toWorkflowGraph(workflowDir: Path?): WorkflowGraph {
val startId = StageId(start)
val stageMap = stages.associate { s ->
StageId(s.id) to StageConfig(
@@ -84,8 +92,8 @@ class TomlWorkflowLoader(
tokenBudget = s.tokenBudget,
maxRetries = s.maxRetries,
metadata = buildMap {
s.prompt?.let { put("prompt", it) }
s.systemPrompt?.let { put("systemPrompt", it) }
s.prompt?.let { put("prompt", resolvePromptPath(it, workflowDir)) }
s.systemPrompt?.let { put("systemPrompt", resolvePromptPath(it, workflowDir)) }
},
)
}