a spoken correction lands in the label table, with or without a target (V-636)
The gesture was web-only, so the sample was skewing to the turns he happens to type. Voice is where the hard cases are. Half of it already existed: the repair rung has read "нет, это была заметка" since V-455. It taught the classifier and wrote no durable label, so the two paths disagreed about what a correction is. It now writes both. Two sinks and not one on purpose: the classifier seed makes the next turn better today, and the label is what a fitted head trains on after the transcript expires. The trace id is stamped onto the remembered turn after the fact, because the trace is written when the turn ends and recordTurn runs in the middle of it. New: the untargeted half. "нет, не так" writes the negative and redoes nothing, because there is no target to redo it as. Voice needs this more than the web does — naming an intent aloud means saying "заметка" or "факт", which is her vocabulary and not his. repair_negatives is a new closed lexicon set matched against the WHOLE utterance, never as a substring. That is what keeps it apart from repair_markers, where "это не" is a fragment that needs an intent word after it. A member that could appear inside an ordinary sentence does not belong in the set.
This commit is contained in:
@@ -7,6 +7,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/kami/maven/internal/router"
|
||||
"github.com/kami/maven/internal/store"
|
||||
)
|
||||
|
||||
func TestParseRepairReadsTheCorrectedIntent(t *testing.T) {
|
||||
@@ -149,3 +150,95 @@ func TestRepairIntentWordCollisions(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// V-636. A spoken correction lands in the same table the /chat gesture writes,
|
||||
// so the sample is not limited to the turns he happened to type.
|
||||
func TestSpokenCorrectionWritesTheLabel(t *testing.T) {
|
||||
h, st, _ := newClarifyHandler(t)
|
||||
emb := router.NewHashEmbedder(256)
|
||||
h.recall.embedder = emb
|
||||
h.router = router.New(router.Config{Classifier: router.NewClassifier(emb), Extractor: h.extractor})
|
||||
ctx := context.Background()
|
||||
|
||||
id, err := st.WriteRoutingTrace(ctx, store.RoutingTrace{
|
||||
Ts: h.now(), Utterance: "купить хлеб", Intent: "fact", Source: "tap:voice",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h.recordTurn("купить хлеб", router.IntentFact)
|
||||
h.stampLastTurn("купить хлеб", id)
|
||||
|
||||
if _, handled := h.resolveRepair(ctx, "нет, это заметка"); !handled {
|
||||
t.Fatal("the correction was not handled")
|
||||
}
|
||||
labels, err := st.RoutingLabels(ctx, 5)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(labels) != 1 || labels[0].Was != "fact" || labels[0].ShouldBe != "note" {
|
||||
t.Fatalf("labels %+v: the spoken correction did not land as a pair", labels)
|
||||
}
|
||||
}
|
||||
|
||||
// The cheap half, which voice needs more than the web does: naming an intent
|
||||
// aloud means saying "заметка", which is her vocabulary and not his.
|
||||
func TestUntargetedSpokenCorrection(t *testing.T) {
|
||||
h, st, now := newClarifyHandler(t)
|
||||
ctx := context.Background()
|
||||
seed := func(utterance string) int64 {
|
||||
id, err := st.WriteRoutingTrace(ctx, store.RoutingTrace{
|
||||
Ts: h.now(), Utterance: utterance, Intent: "query", Source: "tap:voice",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
h.recordTurn(utterance, router.IntentQuery)
|
||||
h.stampLastTurn(utterance, id)
|
||||
return id
|
||||
}
|
||||
|
||||
seed("поужинал")
|
||||
reply, handled := h.resolveUntargetedRepair(ctx, "нет, не так")
|
||||
if !handled {
|
||||
t.Fatal("«нет, не так» was not read as a correction")
|
||||
}
|
||||
if reply == "" {
|
||||
t.Error("a correction he cannot hear reads as one that was dropped")
|
||||
}
|
||||
labels, err := st.RoutingLabels(ctx, 5)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(labels) != 1 || labels[0].ShouldBe != "" || labels[0].Was != "query" {
|
||||
t.Fatalf("labels %+v: want one untargeted negative naming what she chose", labels)
|
||||
}
|
||||
|
||||
// Outside the window it is a fresh sentence, not a verdict.
|
||||
seed("поужинал ещё раз")
|
||||
*now = now.Add(repairWindow + time.Minute)
|
||||
if _, handled := h.resolveUntargetedRepair(ctx, "не так"); handled {
|
||||
t.Error("a correction outside the window was handled")
|
||||
}
|
||||
}
|
||||
|
||||
// Whole-utterance, never a substring. This is the difference between the
|
||||
// negatives and the markers, and getting it wrong would claim any sentence with
|
||||
// "не так" in it.
|
||||
func TestRepairNegativeIsTheWholeUtterance(t *testing.T) {
|
||||
for _, s := range []string{
|
||||
"не так поняла", "нет, не так", "ты ошиблась", "неправильно", "wrong", "no, that was wrong",
|
||||
} {
|
||||
if !isRepairNegative(s) {
|
||||
t.Errorf("%q is not read as a correction", s)
|
||||
}
|
||||
}
|
||||
for _, s := range []string{
|
||||
"это не важно", "напомни не так поздно", "а не завтра", "не так, а вот так — это заметка",
|
||||
"", "нет",
|
||||
} {
|
||||
if isRepairNegative(s) {
|
||||
t.Errorf("%q was read as a correction", s)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user