Three things that decide whether memory improves with use or rots with it. A RELEVANCE FLOOR. Recall returned its top five whatever they scored, so a run about shift cover was handed five memories about certifications simply because nothing better existed — and the block tells the model these are things the workspace remembered, so it reads them as pertinent. Embeddings are unit-normalised, so knowledge_dot is cosine, and 0.30 is where text is usually about something else. A judgement rather than a measurement, and the honest way to tune it is to watch what gets carried on real questions. NO DUPLICATES. The same standing preference comes up in conversation after conversation, and each run that hears it has no idea the last one wrote it down. Five recall slots spent on one fact restated five ways is the normal failure, not a rare one. A write with the same normalised text, in the same org and about the same subject, pushes the existing memory's expiry out instead of adding a row — matched on the same sentence rather than a similar one, because collapsing two genuinely different facts is the worse error. EXPIRY THAT DELETES. expires_at was set and filtered on read, and nothing ever removed anything: the row was invisible and still retained. "We keep it ninety days" has to be true of the table, not only of the query. Prune is batched, and a redaction is kept for a thirty-day grace period so an erasure stays provable shortly afterwards. It runs in the maintenance sweeper that already exists rather than a second scheduler — same ticker, same cancellation, same failure isolation. That forced one honest change: Maintenance() used to be nil without OAuth, on the reasoning that there was nothing to sweep. There is now, and a retention promise enforced only when an unrelated feature happens to be enabled is not a promise. The test that asserted the old behaviour now asserts the new one and says why. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
139 lines
5.7 KiB
Go
139 lines
5.7 KiB
Go
package memory
|
|
|
|
import (
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
)
|
|
|
|
// The rule that makes storing an observation about a person defensible: it can
|
|
// be found. A memory about somebody that names nobody cannot be shown to them
|
|
// on request and cannot be erased for them, so it is refused at the door.
|
|
func TestAPersonalMemoryWithoutASubjectIsRefused(t *testing.T) {
|
|
for _, subject := range []Subject{SubjectCandidate, SubjectUser} {
|
|
w := Write{SubjectType: subject, Text: "seemed unreliable", Author: AuthorModel}
|
|
if err := w.Validate(); err != ErrSubjectRequired {
|
|
t.Errorf("%s without a subject id: got %v, want ErrSubjectRequired", subject, err)
|
|
}
|
|
}
|
|
}
|
|
|
|
// A workspace fact has no personal subject and must not be made to invent one.
|
|
func TestAWorkspaceMemoryNeedsNoSubject(t *testing.T) {
|
|
w := Write{SubjectType: SubjectWorkspace, Text: "This venue staffs on Thursdays.", Author: AuthorModel}
|
|
if err := w.Validate(); err != nil {
|
|
t.Errorf("a workspace fact was refused: %v", err)
|
|
}
|
|
}
|
|
|
|
func TestAnEmptyMemoryIsRefused(t *testing.T) {
|
|
w := Write{SubjectType: SubjectWorkspace, Text: " ", Author: AuthorModel}
|
|
if err := w.Validate(); err != ErrEmpty {
|
|
t.Errorf("got %v, want ErrEmpty", err)
|
|
}
|
|
}
|
|
|
|
// A memory is a sentence. The long form of something belongs in the corpus,
|
|
// which has ingestion review and search; this table has neither.
|
|
func TestAMemoryLongerThanASentenceIsRefused(t *testing.T) {
|
|
w := Write{SubjectType: SubjectWorkspace, Text: strings.Repeat("x", MaxTextRunes+1), Author: AuthorModel}
|
|
if err := w.Validate(); err == nil {
|
|
t.Error("an over-long memory was accepted")
|
|
}
|
|
}
|
|
|
|
func TestAnUnknownSubjectOrAuthorIsRefused(t *testing.T) {
|
|
if err := (Write{SubjectType: "anything", Text: "x", Author: AuthorModel}).Validate(); err == nil {
|
|
t.Error("an invented subject type was accepted")
|
|
}
|
|
if err := (Write{SubjectType: SubjectWorkspace, Text: "x", Author: "nobody"}).Validate(); err == nil {
|
|
t.Error("an invented author was accepted")
|
|
}
|
|
}
|
|
|
|
// Everything written decays. A fact with no end date is read back long after
|
|
// it stopped being true.
|
|
func TestTheDefaultTTLIsBounded(t *testing.T) {
|
|
if DefaultTTL <= 0 || DefaultTTL > 365*24*time.Hour {
|
|
t.Errorf("DefaultTTL = %v; a memory must expire, and within a year", DefaultTTL)
|
|
}
|
|
}
|
|
|
|
/* ── What the model is shown ─────────────────────────────────────────────── */
|
|
|
|
func TestRenderFencesAndLabelsMemories(t *testing.T) {
|
|
out := Render([]Record{
|
|
{SubjectType: SubjectWorkspace, Author: AuthorModel, Text: "Thursdays are short-staffed."},
|
|
})
|
|
for _, want := range []string{"<memory>", "</memory>", "never", "instructions"} {
|
|
if !strings.Contains(out, want) {
|
|
t.Errorf("the memory block does not contain %q:\n%s", want, out)
|
|
}
|
|
}
|
|
}
|
|
|
|
// An inference and a recruiter's note are different kinds of claim. Flattening
|
|
// them lets "the model thought X" be read back later as "X".
|
|
func TestRenderSaysWhetherAMemoryWasInferredOrWritten(t *testing.T) {
|
|
out := Render([]Record{
|
|
{SubjectType: SubjectCandidate, SubjectID: "c1", Author: AuthorModel, Text: "A"},
|
|
{SubjectType: SubjectCandidate, SubjectID: "c2", Author: AuthorPerson, Text: "B"},
|
|
})
|
|
if !strings.Contains(out, "inferred by an agent") || !strings.Contains(out, "noted by a person") {
|
|
t.Errorf("the origin of each memory is not stated:\n%s", out)
|
|
}
|
|
}
|
|
|
|
// The block says plainly that a memory is not a reason to reject somebody.
|
|
// This is the sentence that keeps a remembered impression from being read as a
|
|
// decision, so it is pinned by a test rather than left to an edit.
|
|
func TestRenderRefusesToLetAMemoryDecide(t *testing.T) {
|
|
out := Render([]Record{{SubjectType: SubjectCandidate, SubjectID: "c1", Author: AuthorModel, Text: "A"}})
|
|
if !strings.Contains(out, "never a reason on their own to accept or reject") {
|
|
t.Errorf("the block does not say a memory cannot decide:\n%s", out)
|
|
}
|
|
}
|
|
|
|
func TestRenderIsEmptyWhenThereIsNothingToRemember(t *testing.T) {
|
|
if Render(nil) != "" {
|
|
t.Error("an empty memory set must add nothing to the prompt")
|
|
}
|
|
}
|
|
|
|
/* ── Hygiene ─────────────────────────────────────────────────────────────── */
|
|
|
|
// A floor, not just a top N. Without one, a run about shift cover is handed
|
|
// five memories about certifications simply because nothing better exists —
|
|
// and the model is told these are things the workspace remembered.
|
|
func TestThereIsARelevanceFloor(t *testing.T) {
|
|
if MinRelevance <= 0 || MinRelevance >= 1 {
|
|
t.Fatalf("MinRelevance = %v; a cosine floor belongs in (0, 1)", MinRelevance)
|
|
}
|
|
// Low enough to carry a genuinely related memory, high enough to exclude
|
|
// unrelated text. Pinned so a later "let's return more" cannot quietly
|
|
// become "let's return anything".
|
|
if MinRelevance < 0.15 || MinRelevance > 0.6 {
|
|
t.Errorf("MinRelevance = %v; outside the range where this is a filter rather than a formality", MinRelevance)
|
|
}
|
|
}
|
|
|
|
// Recall competes with the tool catalogue and the retrieved block for one
|
|
// prompt, against a per-minute ceiling.
|
|
func TestRecallIsSmallEnoughToShareAPrompt(t *testing.T) {
|
|
if DefaultRecall <= 0 || DefaultRecall > 10 {
|
|
t.Errorf("DefaultRecall = %d; memory must not crowd out the evidence", DefaultRecall)
|
|
}
|
|
}
|
|
|
|
// The retention promise is about the table, not only about the query. Reads
|
|
// already hide an expired row; Prune is what makes "we keep it ninety days"
|
|
// true of what is actually stored.
|
|
func TestPruneIsBatched(t *testing.T) {
|
|
// A first pass against a large table must not hold one transaction across
|
|
// the whole of it. The default is applied when the caller passes nothing.
|
|
s := New(nil, nil)
|
|
if s == nil {
|
|
t.Fatal("a store without a database should still construct")
|
|
}
|
|
}
|