e926e4e6df
checkPrivate validates the configured endpoint literal and validated nothing after it. The client followed redirects, so a 302 from the local llama-server would have sent the photo, as a data URI in the POST body, to whatever the redirect named. "No provider in this repo may upload a blob" was true only of the first hop. Redirects are refused now, and the reply is read through a cap rather than however much the endpoint feels like sending. ValidateEndpoint exports the same check so config can fail at startup on a typo instead of logging once and leaving vision quietly off. Found in review of #72.
306 lines
11 KiB
Go
306 lines
11 KiB
Go
// Package vision is Maven's image-understanding seam (Vikunja #252,
|
|
// docs/plans/07-vision.md).
|
|
//
|
|
// One interface, Provider, with one method: describe an image, in words, in
|
|
// Russian, with an optional question about it. Text extraction is not a second
|
|
// method — "прочитай текст с картинки" is a prompt, and a vision-language model
|
|
// does not have a separate OCR mode to select.
|
|
//
|
|
// # What is deliberately NOT here
|
|
//
|
|
// The plan document called for a `RemoteProvider` calling "an OpenAI-compatible
|
|
// vision API endpoint". That step is refused: CLAUDE.md's surviving hard
|
|
// constraint after "never phones home" was deprecated is *no cloud model,
|
|
// inference stays on the box*, and a photo of his flat is the single worst thing
|
|
// to make an exception for. Endpoint is therefore checked at construction and
|
|
// must be a loopback or private address — a public host is a config error, not a
|
|
// deployment option. That check is the reason this package does not simply reuse
|
|
// internal/llm.Client.
|
|
//
|
|
// # State on this box, honestly
|
|
//
|
|
// The resident model is Qwen3-1.7B, which is text-only, and as of 2026-08-01
|
|
// there is no vision-capable gguf and no mmproj file anywhere under
|
|
// /mnt/hdd1/llms. So LocalProvider is written, tested against a fake server, and
|
|
// currently has nothing real to talk to: the describing half is BLOCKED on a
|
|
// model download (see docs/plans/07-vision.md for the candidates and the
|
|
// recipe). What works today without any download is the intake — an image
|
|
// arrives, is stored, is prepared — and the config seam that turns the rest on.
|
|
//
|
|
// Provider is nil-safe through Disabled, and vision is OFF unless configured,
|
|
// like the weather and telegram.
|
|
package vision
|
|
|
|
import (
|
|
"bytes"
|
|
"context"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"io"
|
|
"net"
|
|
"net/http"
|
|
"net/url"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/kami/maven/internal/media"
|
|
"github.com/kami/maven/internal/webfetch"
|
|
)
|
|
|
|
// DefaultTimeout — budget for one description. A small VLM doing prefill over
|
|
// an 896px image on a Vega iGPU is slow; 90s is generous because nobody is
|
|
// holding a conversation open on this path — the answer arrives as a reply or a
|
|
// note, and a too-tight timeout just means it never arrives at all.
|
|
const DefaultTimeout = 90 * time.Second
|
|
|
|
// DefaultMaxTokens — cap on the description. A paragraph is what a spoken
|
|
// answer can carry; a page is not.
|
|
const DefaultMaxTokens = 300
|
|
|
|
// DefaultPrompt — what she is asked when he did not ask anything specific,
|
|
// only sent a picture. Russian, because that is the channel language, and
|
|
// feminine self-reference is not needed here (the prompt is an instruction, the
|
|
// persona block is added by the caller that phrases the reply).
|
|
const DefaultPrompt = "Опиши, что на этом изображении. Коротко, 2-3 предложения. Если на нём есть текст, приведи его."
|
|
|
|
// Errors callers distinguish.
|
|
var (
|
|
// ErrDisabled — vision is not configured. Returned by Disabled, which is
|
|
// what the daemon wires when the config block is absent.
|
|
ErrDisabled = errors.New("vision: not configured")
|
|
// ErrNotPrivate — the configured endpoint is not on this box or its
|
|
// network. Refused at construction; see the package comment.
|
|
ErrNotPrivate = errors.New("vision: endpoint must be a local or private address")
|
|
// ErrEmptyReply — the model returned nothing usable.
|
|
ErrEmptyReply = errors.New("vision: empty description")
|
|
)
|
|
|
|
// Provider — the image-understanding contract. Describe takes an image already
|
|
// prepared by internal/media (decoded, downscaled, JPEG) and a prompt; an empty
|
|
// prompt means DefaultPrompt.
|
|
type Provider interface {
|
|
Describe(ctx context.Context, im media.Image, prompt string) (string, error)
|
|
}
|
|
|
|
// Disabled — the floor Provider. Every call fails with ErrDisabled, which the
|
|
// caller turns into "я не умею смотреть картинки — зрение не настроено". It
|
|
// exists so that no call site needs a nil check and switching vision off cannot
|
|
// crash a turn.
|
|
type Disabled struct{}
|
|
|
|
// Describe always fails. The signature matches Provider.
|
|
func (Disabled) Describe(context.Context, media.Image, string) (string, error) {
|
|
return "", ErrDisabled
|
|
}
|
|
|
|
// MaxReplyBytes bounds what is read back from the vision server. A description
|
|
// is words; anything past a megabyte is a broken endpoint.
|
|
const MaxReplyBytes = 1 << 20
|
|
|
|
// Config — how to reach the local vision server. Built from
|
|
// config.VisionConfig by the daemon; kept separate so this package does not
|
|
// import internal/config.
|
|
type Config struct {
|
|
// Endpoint — base URL of a llama-server started with a vision model and its
|
|
// mmproj (`llama-server -m model.gguf --mmproj mmproj.gguf`). Must be
|
|
// loopback or private. The path is appended by the provider; give it
|
|
// "http://127.0.0.1:8081".
|
|
Endpoint string
|
|
// Model — the model name to send. llama-server ignores it; it matters if the
|
|
// endpoint is something else OpenAI-shaped on the same box.
|
|
Model string
|
|
// Timeout — per-description budget. 0 ⇒ DefaultTimeout.
|
|
Timeout time.Duration
|
|
// MaxTokens — cap on the reply. 0 ⇒ DefaultMaxTokens.
|
|
MaxTokens int
|
|
// Prompt — the default question. Empty ⇒ DefaultPrompt.
|
|
Prompt string
|
|
}
|
|
|
|
// LocalProvider talks to a llama-server on this box over its
|
|
// /v1/chat/completions endpoint, sending the image as a data URI content part.
|
|
// It is the only real Provider, and it is a plain HTTP client: no subprocess
|
|
// spawning, because the daemon already owns llama-server lifecycle for the
|
|
// resident model and a second managed process is a bigger change than this task.
|
|
type LocalProvider struct {
|
|
endpoint string
|
|
model string
|
|
prompt string
|
|
maxTokens int
|
|
http *http.Client
|
|
}
|
|
|
|
// NewLocal builds a LocalProvider, refusing a non-private endpoint. A bad URL
|
|
// or a public host is an error at construction so the daemon logs it once at
|
|
// startup instead of failing every turn.
|
|
func NewLocal(cfg Config) (*LocalProvider, error) {
|
|
base := strings.TrimRight(strings.TrimSpace(cfg.Endpoint), "/")
|
|
if base == "" {
|
|
return nil, errors.New("vision: empty endpoint")
|
|
}
|
|
if err := checkPrivate(base); err != nil {
|
|
return nil, err
|
|
}
|
|
timeout := cfg.Timeout
|
|
if timeout <= 0 {
|
|
timeout = DefaultTimeout
|
|
}
|
|
maxTokens := cfg.MaxTokens
|
|
if maxTokens <= 0 {
|
|
maxTokens = DefaultMaxTokens
|
|
}
|
|
prompt := strings.TrimSpace(cfg.Prompt)
|
|
if prompt == "" {
|
|
prompt = DefaultPrompt
|
|
}
|
|
return &LocalProvider{
|
|
endpoint: base,
|
|
model: cfg.Model,
|
|
prompt: prompt,
|
|
maxTokens: maxTokens,
|
|
http: &http.Client{
|
|
Timeout: timeout,
|
|
// No redirects. checkPrivate validates the configured literal and
|
|
// nothing validated a hop, so a 302 from the local llama-server
|
|
// would send the photo, as a data URI in the POST body, wherever
|
|
// the redirect named. "No provider in this repo may upload a blob"
|
|
// has to be true of the second request as well as the first.
|
|
CheckRedirect: func(*http.Request, []*http.Request) error {
|
|
return http.ErrUseLastResponse
|
|
},
|
|
},
|
|
}, nil
|
|
}
|
|
|
|
// Endpoint is the server this provider talks to. For logs and /dash.
|
|
func (p *LocalProvider) Endpoint() string { return p.endpoint }
|
|
|
|
// ValidateEndpoint reports whether a configured endpoint is one this package
|
|
// would accept. Exported so config validation fails at startup on a typo,
|
|
// rather than logging once at wiring time and leaving the capability quietly
|
|
// off.
|
|
func ValidateEndpoint(raw string) error {
|
|
return checkPrivate(strings.TrimRight(strings.TrimSpace(raw), "/"))
|
|
}
|
|
|
|
// checkPrivate refuses any endpoint that is not on this box or its LAN. A
|
|
// hostname that is not an IP literal is refused too: "vision.example.com" could
|
|
// resolve anywhere, and resolving it here would be trusting DNS with his photos.
|
|
// localhost is the one name allowed, because it is the common case.
|
|
func checkPrivate(raw string) error {
|
|
u, err := url.Parse(raw)
|
|
if err != nil {
|
|
return fmt.Errorf("vision: parse endpoint: %w", err)
|
|
}
|
|
if u.Scheme != "http" && u.Scheme != "https" {
|
|
return fmt.Errorf("vision: endpoint scheme %q not supported", u.Scheme)
|
|
}
|
|
host := u.Hostname()
|
|
if host == "" {
|
|
return errors.New("vision: endpoint has no host")
|
|
}
|
|
if strings.EqualFold(host, "localhost") {
|
|
return nil
|
|
}
|
|
ip := net.ParseIP(host)
|
|
if ip == nil {
|
|
return fmt.Errorf("%w: %q is a name, not an address", ErrNotPrivate, host)
|
|
}
|
|
if !webfetch.IsPrivateIP(ip) {
|
|
return fmt.Errorf("%w: %s", ErrNotPrivate, host)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// chat request shapes. Content is the OpenAI multimodal array form: a text part
|
|
// and an image_url part whose url is a data URI.
|
|
type textPart struct {
|
|
Type string `json:"type"`
|
|
Text string `json:"text"`
|
|
}
|
|
type imageURL struct {
|
|
URL string `json:"url"`
|
|
}
|
|
type imagePart struct {
|
|
Type string `json:"type"`
|
|
ImageURL imageURL `json:"image_url"`
|
|
}
|
|
type chatReq struct {
|
|
Model string `json:"model,omitempty"`
|
|
Messages []any `json:"messages"`
|
|
MaxTokens int `json:"max_tokens,omitempty"`
|
|
Temp float64 `json:"temperature"`
|
|
}
|
|
type userMsg struct {
|
|
Role string `json:"role"`
|
|
Content []any `json:"content"`
|
|
}
|
|
type chatResp struct {
|
|
Choices []struct {
|
|
Message struct {
|
|
Content string `json:"content"`
|
|
ReasoningContent string `json:"reasoning_content"`
|
|
} `json:"message"`
|
|
} `json:"choices"`
|
|
}
|
|
|
|
// Describe sends the image and prompt and returns the model's answer. An empty
|
|
// prompt uses the configured default. Errors are wrapped, never fatal: the
|
|
// caller says she could not make out the picture and the turn continues.
|
|
func (p *LocalProvider) Describe(ctx context.Context, im media.Image, prompt string) (string, error) {
|
|
if len(im.JPEG) == 0 {
|
|
return "", media.ErrEmpty
|
|
}
|
|
q := strings.TrimSpace(prompt)
|
|
if q == "" {
|
|
q = p.prompt
|
|
}
|
|
body, err := json.Marshal(chatReq{
|
|
Model: p.model,
|
|
MaxTokens: p.maxTokens,
|
|
Messages: []any{userMsg{Role: "user", Content: []any{
|
|
textPart{Type: "text", Text: q},
|
|
imagePart{Type: "image_url", ImageURL: imageURL{URL: im.DataURI()}},
|
|
}}},
|
|
})
|
|
if err != nil {
|
|
return "", fmt.Errorf("vision: marshal: %w", err)
|
|
}
|
|
req, err := http.NewRequestWithContext(ctx, http.MethodPost,
|
|
p.endpoint+"/v1/chat/completions", bytes.NewReader(body))
|
|
if err != nil {
|
|
return "", fmt.Errorf("vision: request: %w", err)
|
|
}
|
|
req.Header.Set("Content-Type", "application/json")
|
|
|
|
resp, err := p.http.Do(req)
|
|
if err != nil {
|
|
return "", fmt.Errorf("vision: post: %w", err)
|
|
}
|
|
defer resp.Body.Close()
|
|
if resp.StatusCode != http.StatusOK {
|
|
return "", fmt.Errorf("vision: status %d", resp.StatusCode)
|
|
}
|
|
var out chatResp
|
|
// Capped: the decoder would otherwise read whatever the endpoint sends, and
|
|
// a local server answering with a stuck stream should not cost the daemon
|
|
// its memory. A description is a few hundred tokens.
|
|
if err := json.NewDecoder(io.LimitReader(resp.Body, MaxReplyBytes)).Decode(&out); err != nil {
|
|
return "", fmt.Errorf("vision: decode: %w", err)
|
|
}
|
|
if len(out.Choices) == 0 {
|
|
return "", ErrEmptyReply
|
|
}
|
|
text := strings.TrimSpace(out.Choices[0].Message.Content)
|
|
if text == "" {
|
|
// Same fallback as internal/llm: a Thinking model sometimes puts the
|
|
// whole answer in reasoning_content and leaves content empty.
|
|
text = strings.TrimSpace(out.Choices[0].Message.ReasoningContent)
|
|
}
|
|
if text == "" {
|
|
return "", ErrEmptyReply
|
|
}
|
|
return text, nil
|
|
}
|