phraser: the backend types move to server.go (V-397)
Verbatim move: backend, borrowedBackend, llamaProc, lineTail, extractPort and listenRE. The spawn functions follow in the next commit.
This commit is contained in:
@@ -12,7 +12,6 @@ import (
|
|||||||
"net/http"
|
"net/http"
|
||||||
"os"
|
"os"
|
||||||
"os/exec"
|
"os/exec"
|
||||||
"regexp"
|
|
||||||
"strings"
|
"strings"
|
||||||
"sync"
|
"sync"
|
||||||
"syscall"
|
"syscall"
|
||||||
@@ -25,8 +24,6 @@ import (
|
|||||||
"github.com/kami/maven/internal/router"
|
"github.com/kami/maven/internal/router"
|
||||||
)
|
)
|
||||||
|
|
||||||
var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
|
|
||||||
|
|
||||||
// errEmptyResponse — the server answered and said nothing. Separate from a
|
// errEmptyResponse — the server answered and said nothing. Separate from a
|
||||||
// transport failure: the model is up and produced no tokens, which is still not
|
// transport failure: the model is up and produced no tokens, which is still not
|
||||||
// an answer and must not score as one.
|
// an answer and must not score as one.
|
||||||
@@ -218,41 +215,6 @@ func loadNudgeTemplates() *NudgeTemplates {
|
|||||||
return nt
|
return nt
|
||||||
}
|
}
|
||||||
|
|
||||||
// backend — one llama-server this phraser talks to. Two implementations: a
|
|
||||||
// llamaProc we spawned and must reap, and a borrowedBackend someone else owns.
|
|
||||||
type backend interface {
|
|
||||||
BaseURL() string
|
|
||||||
Close() error
|
|
||||||
}
|
|
||||||
|
|
||||||
// borrowedBackend — a server started and owned by someone else (the phrasing
|
|
||||||
// scorer's shared llama-server). Closing it is a no-op by construction.
|
|
||||||
type borrowedBackend string
|
|
||||||
|
|
||||||
func (b borrowedBackend) BaseURL() string { return string(b) }
|
|
||||||
func (b borrowedBackend) Close() error { return nil }
|
|
||||||
|
|
||||||
// llamaProc — a llama-server child process plus the goroutine reading its
|
|
||||||
// stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer.
|
|
||||||
type llamaProc struct {
|
|
||||||
base string
|
|
||||||
cmd *exec.Cmd
|
|
||||||
cancel context.CancelFunc
|
|
||||||
wg sync.WaitGroup
|
|
||||||
}
|
|
||||||
|
|
||||||
func (l *llamaProc) BaseURL() string { return l.base }
|
|
||||||
|
|
||||||
func (l *llamaProc) Close() error {
|
|
||||||
l.cancel()
|
|
||||||
if l.cmd != nil && l.cmd.Process != nil {
|
|
||||||
_ = l.cmd.Process.Kill()
|
|
||||||
_ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
|
|
||||||
}
|
|
||||||
l.wg.Wait()
|
|
||||||
return nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
// spawnLlamaServer starts one llama-server for cfg and waits until it says which
|
||||||
// address it is listening on. ctx owns the process lifetime, so it must be the
|
// address it is listening on. ctx owns the process lifetime, so it must be the
|
||||||
// daemon's context, not a request's.
|
// daemon's context, not a request's.
|
||||||
@@ -376,34 +338,6 @@ func startLlamaProc(ctx context.Context, cfg Config) (*llamaProc, error) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// lineTail keeps the last few startup lines so a server that dies before it
|
|
||||||
// listens can say why in the error, not just "EOF". Written by the reader
|
|
||||||
// goroutine and read by whoever gives up on startup, so it takes a lock.
|
|
||||||
type lineTail struct {
|
|
||||||
mu sync.Mutex
|
|
||||||
lines []string
|
|
||||||
}
|
|
||||||
|
|
||||||
const lineTailMax = 12
|
|
||||||
|
|
||||||
func (t *lineTail) add(line string) {
|
|
||||||
t.mu.Lock()
|
|
||||||
defer t.mu.Unlock()
|
|
||||||
t.lines = append(t.lines, line)
|
|
||||||
if len(t.lines) > lineTailMax {
|
|
||||||
t.lines = t.lines[len(t.lines)-lineTailMax:]
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
func (t *lineTail) String() string {
|
|
||||||
t.mu.Lock()
|
|
||||||
defer t.mu.Unlock()
|
|
||||||
if len(t.lines) == 0 {
|
|
||||||
return "(no output)"
|
|
||||||
}
|
|
||||||
return strings.Join(t.lines, " | ")
|
|
||||||
}
|
|
||||||
|
|
||||||
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
// BaseURL is the llama-server this phraser talks to right now. It changes when
|
||||||
// the model is swapped, so callers that cache it must register an observer
|
// the model is swapped, so callers that cache it must register an observer
|
||||||
// (OnSwap) rather than keeping the string forever.
|
// (OnSwap) rather than keeping the string forever.
|
||||||
@@ -1155,11 +1089,3 @@ func buildNudgePrompt(c loop.Candidate) string {
|
|||||||
|
|
||||||
return strings.Join(ctxParts, "\n") + "\n\n" + tail
|
return strings.Join(ctxParts, "\n") + "\n\n" + tail
|
||||||
}
|
}
|
||||||
|
|
||||||
func extractPort(listen string) string {
|
|
||||||
_, port, _ := strings.Cut(listen, ":")
|
|
||||||
if port == "" {
|
|
||||||
return "0"
|
|
||||||
}
|
|
||||||
return port
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -0,0 +1,87 @@
|
|||||||
|
// phraser/server.go — the llama-server this phraser talks to: the child
|
||||||
|
// process it may own, and the startup handshake that waits for the port.
|
||||||
|
//
|
||||||
|
// Nothing here knows what a prompt is. Split out of llmphraser.go so the
|
||||||
|
// phrasing paths and the process lifetime read as two subjects.
|
||||||
|
package phraser
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"os/exec"
|
||||||
|
"regexp"
|
||||||
|
"strings"
|
||||||
|
"sync"
|
||||||
|
)
|
||||||
|
|
||||||
|
var listenRE = regexp.MustCompile(`listening on (https?://\S+)`)
|
||||||
|
|
||||||
|
// backend — one llama-server this phraser talks to. Two implementations: a
|
||||||
|
// llamaProc we spawned and must reap, and a borrowedBackend someone else owns.
|
||||||
|
type backend interface {
|
||||||
|
BaseURL() string
|
||||||
|
Close() error
|
||||||
|
}
|
||||||
|
|
||||||
|
// borrowedBackend — a server started and owned by someone else (the phrasing
|
||||||
|
// scorer's shared llama-server). Closing it is a no-op by construction.
|
||||||
|
type borrowedBackend string
|
||||||
|
|
||||||
|
func (b borrowedBackend) BaseURL() string { return string(b) }
|
||||||
|
func (b borrowedBackend) Close() error { return nil }
|
||||||
|
|
||||||
|
// llamaProc — a llama-server child process plus the goroutine reading its
|
||||||
|
// stderr. Close kills and reaps it; see the Pdeathsig note in spawnLlamaServer.
|
||||||
|
type llamaProc struct {
|
||||||
|
base string
|
||||||
|
cmd *exec.Cmd
|
||||||
|
cancel context.CancelFunc
|
||||||
|
wg sync.WaitGroup
|
||||||
|
}
|
||||||
|
|
||||||
|
func (l *llamaProc) BaseURL() string { return l.base }
|
||||||
|
|
||||||
|
func (l *llamaProc) Close() error {
|
||||||
|
l.cancel()
|
||||||
|
if l.cmd != nil && l.cmd.Process != nil {
|
||||||
|
_ = l.cmd.Process.Kill()
|
||||||
|
_ = l.cmd.Wait() // reap the process — without Wait, the child becomes a zombie
|
||||||
|
}
|
||||||
|
l.wg.Wait()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// lineTail keeps the last few startup lines so a server that dies before it
|
||||||
|
// listens can say why in the error, not just "EOF". Written by the reader
|
||||||
|
// goroutine and read by whoever gives up on startup, so it takes a lock.
|
||||||
|
type lineTail struct {
|
||||||
|
mu sync.Mutex
|
||||||
|
lines []string
|
||||||
|
}
|
||||||
|
|
||||||
|
const lineTailMax = 12
|
||||||
|
|
||||||
|
func (t *lineTail) add(line string) {
|
||||||
|
t.mu.Lock()
|
||||||
|
defer t.mu.Unlock()
|
||||||
|
t.lines = append(t.lines, line)
|
||||||
|
if len(t.lines) > lineTailMax {
|
||||||
|
t.lines = t.lines[len(t.lines)-lineTailMax:]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func (t *lineTail) String() string {
|
||||||
|
t.mu.Lock()
|
||||||
|
defer t.mu.Unlock()
|
||||||
|
if len(t.lines) == 0 {
|
||||||
|
return "(no output)"
|
||||||
|
}
|
||||||
|
return strings.Join(t.lines, " | ")
|
||||||
|
}
|
||||||
|
|
||||||
|
func extractPort(listen string) string {
|
||||||
|
_, port, _ := strings.Cut(listen, ":")
|
||||||
|
if port == "" {
|
||||||
|
return "0"
|
||||||
|
}
|
||||||
|
return port
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user