feat: LLM phraser, shared LLM client, and LLM replier
- Add llmphraser: LFM-based phraser implementing Phraser interface with PhraseChat, PhraseNudge, PhraseReactive, and PhraseReminder methods. - Add shared internal/llm/client: llama-server completion client used by both the phraser (talking back) and router (routing), sharing one model. - Add LLMReplier in mavend: replaces StubReplier for chat/nudge/reactive replies, falls back to stub on model errors. - Update Phraser interface: add PhraseChat method, update stub to match. - Wire LLM phaser into mavend voice init, plumb LLM config from JSON.
This commit is contained in:
@@ -0,0 +1,85 @@
|
||||
// Package llm is the shared llama-server completion client — one seam both the
|
||||
// phraser (talking back) and the router (routing) call. It does NOT spawn the
|
||||
// server; the daemon owns one llama-server (spawned by the phraser) and hands
|
||||
// its base URL here, so a single resident model serves both callers.
|
||||
package llm
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
)
|
||||
|
||||
type Client struct {
|
||||
base string
|
||||
http *http.Client
|
||||
}
|
||||
|
||||
func New(baseURL string, timeout time.Duration) *Client {
|
||||
return &Client{base: baseURL, http: &http.Client{Timeout: timeout}}
|
||||
}
|
||||
|
||||
type Req struct {
|
||||
System string
|
||||
User string
|
||||
Grammar string // GBNF; empty ⇒ unconstrained
|
||||
MaxTokens int
|
||||
// RepeatPenalty > 0 ⇒ penalize token repetition (curbs the sub-1B "тоже
|
||||
// тоже тоже" loop). 0 ⇒ server default (no extra penalty).
|
||||
RepeatPenalty float64
|
||||
// Stop — sequences that end generation early (e.g. newline for a one-liner).
|
||||
Stop []string
|
||||
}
|
||||
|
||||
type msg struct {
|
||||
Role string `json:"role"`
|
||||
Content string `json:"content"`
|
||||
}
|
||||
type body struct {
|
||||
Messages []msg `json:"messages"`
|
||||
MaxTokens int `json:"max_tokens,omitempty"`
|
||||
Grammar string `json:"grammar,omitempty"`
|
||||
Temp float64 `json:"temperature"`
|
||||
RepeatPenalty float64 `json:"repeat_penalty,omitempty"`
|
||||
Stop []string `json:"stop,omitempty"`
|
||||
}
|
||||
type resp struct {
|
||||
Choices []struct {
|
||||
Message msg `json:"message"`
|
||||
} `json:"choices"`
|
||||
}
|
||||
|
||||
func (c *Client) Complete(ctx context.Context, r Req) (string, error) {
|
||||
b, _ := json.Marshal(body{
|
||||
Messages: []msg{{"system", r.System}, {"user", r.User}},
|
||||
MaxTokens: r.MaxTokens,
|
||||
Grammar: r.Grammar,
|
||||
Temp: 0,
|
||||
RepeatPenalty: r.RepeatPenalty,
|
||||
Stop: r.Stop,
|
||||
})
|
||||
req, err := http.NewRequestWithContext(ctx, "POST", c.base+"/v1/chat/completions", bytes.NewReader(b))
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
httpResp, err := c.http.Do(req)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
defer httpResp.Body.Close()
|
||||
if httpResp.StatusCode != 200 {
|
||||
return "", fmt.Errorf("llm: status %d", httpResp.StatusCode)
|
||||
}
|
||||
var out resp
|
||||
if err := json.NewDecoder(httpResp.Body).Decode(&out); err != nil {
|
||||
return "", err
|
||||
}
|
||||
if len(out.Choices) == 0 {
|
||||
return "", fmt.Errorf("llm: no choices")
|
||||
}
|
||||
return out.Choices[0].Message.Content, nil
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package llm
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestComplete(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != "POST" {
|
||||
t.Errorf("method = %q, want POST", r.Method)
|
||||
}
|
||||
if !strings.HasSuffix(r.URL.Path, "/v1/chat/completions") {
|
||||
t.Errorf("path = %q, want /v1/chat/completions", r.URL.Path)
|
||||
}
|
||||
var reqBody struct {
|
||||
Messages []struct {
|
||||
Role string `json:"role"`
|
||||
Content string `json:"content"`
|
||||
} `json:"messages"`
|
||||
Grammar string `json:"grammar"`
|
||||
MaxTokens int `json:"max_tokens"`
|
||||
}
|
||||
if err := json.NewDecoder(r.Body).Decode(&reqBody); err != nil {
|
||||
t.Fatalf("decode request body: %v", err)
|
||||
}
|
||||
if len(reqBody.Messages) < 2 {
|
||||
t.Fatalf("expected at least 2 messages, got %d", len(reqBody.Messages))
|
||||
}
|
||||
if reqBody.Messages[0].Role != "system" || reqBody.Messages[1].Content != "hi" {
|
||||
t.Errorf("unexpected messages: %+v", reqBody.Messages)
|
||||
}
|
||||
if reqBody.Grammar != `root ::= "x"` {
|
||||
t.Errorf("grammar = %q, want root ::= \"x\"", reqBody.Grammar)
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
c := New(srv.URL, 5*time.Second)
|
||||
got, err := c.Complete(context.Background(), Req{
|
||||
System: "be helpful",
|
||||
User: "hi",
|
||||
Grammar: `root ::= "x"`,
|
||||
MaxTokens: 42,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Complete: %v", err)
|
||||
}
|
||||
if got != "ok" {
|
||||
t.Errorf("got %q, want %q", got, "ok")
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user