Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/CONFIG.md
Original file line number Diff line number Diff line change
Expand Up @@ -594,6 +594,7 @@ The `memory` section controls the persistent memory system (see [docs/MEMORY.md]
"memory_budget_chars": 2000,
"decay_half_life_days": 30,
"quarantine_ttl_days": 7,
"ephemeral_ttl_days": 14,
"eviction_policy": "retention_decay",
"predictive_intents": 3,
"auto_extract_per_turn": true,
Expand Down Expand Up @@ -630,6 +631,7 @@ The `memory` section controls the persistent memory system (see [docs/MEMORY.md]
| `memory_budget_chars` | `2000` | `ODEK_MEMORY_EXTENDED_MEMORY_BUDGET_CHARS` | `--memory-extended-memory-budget-chars` | Maximum injected Extended Memory context per turn. |
| `decay_half_life_days` | `30` | — | — | Days until an atom's recall/eviction weight halves. |
| `quarantine_ttl_days` | `7` | — | — | Days before a tainted atom is auto-deleted from quarantine. |
| `ephemeral_ttl_days` | `14` | — | — | Days before ephemeral-class atoms (intent, goal, error, question, file) stop being recalled. Durable classes (preference, convention, fact, decision) never expire via TTL; pinned atoms are exempt. |
| `eviction_policy` | `"retention_decay"` | — | — | Eviction algorithm. `"retention_decay"` is the only supported value. |
| `predictive_intents` | `3` | — | — | Reserved for future predictive-intent recall. Currently accepted but ignored. |
| `auto_extract_per_turn` | `true` | — | — | Extract atoms after every user message. |
Expand Down
5 changes: 5 additions & 0 deletions internal/memory/extended/config.go
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@ type Config struct {
MemoryBudgetChars int `json:"memory_budget_chars,omitempty"`
DecayHalfLifeDays int `json:"decay_half_life_days,omitempty"`
QuarantineTTLDays int `json:"quarantine_ttl_days,omitempty"`
EphemeralTTLDays int `json:"ephemeral_ttl_days,omitempty"` // TTL for ephemeral atom classes (intent/goal/error/question/file); 0 = default
EvictionPolicy string `json:"eviction_policy,omitempty"`
PredictiveIntents int `json:"predictive_intents,omitempty"`
AutoExtractPerTurn *bool `json:"auto_extract_per_turn,omitempty"`
Expand Down Expand Up @@ -86,6 +87,7 @@ func DefaultConfig() Config {
MemoryBudgetChars: 2000,
DecayHalfLifeDays: 30,
QuarantineTTLDays: 7,
EphemeralTTLDays: DefaultEphemeralTTLDays,
EvictionPolicy: "retention_decay",
PredictiveIntents: 3,
AutoExtractPerTurn: boolPtr(true),
Expand Down Expand Up @@ -144,6 +146,9 @@ func Resolve(cfg Config) Config {
if cfg.QuarantineTTLDays > 0 {
def.QuarantineTTLDays = cfg.QuarantineTTLDays
}
if cfg.EphemeralTTLDays > 0 {
def.EphemeralTTLDays = cfg.EphemeralTTLDays
}
if cfg.EvictionPolicy != "" {
def.EvictionPolicy = cfg.EvictionPolicy
}
Expand Down
103 changes: 102 additions & 1 deletion internal/memory/extended/extractor.go
Original file line number Diff line number Diff line change
Expand Up @@ -49,9 +49,65 @@ Rules:
- Do NOT extract ephemeral details specific only to this message.
- If nothing durable is present, return an empty array.

Do NOT extract release-specific or bookkeeping facts — they rot immediately:
- Session IDs, turn numbers, timestamps, file paths, commit hashes.
- Version tags, PR numbers, CI run statuses ("tag 1.14.8", "PR #45", "merged as 768d380").
- A session ID like "20260918-3e4cb01f" is provenance, never content.
- Statements about this memory system itself (pending reviews, stored atoms).
- Restatements of something already durable — generalize instead.

Examples of REJECTS (do not emit these):
"User said merge after CI passes (turn 3, session 20260918-…)" -> REJECT (provenance in text)
"The correct version tag is 1.14.8" -> REJECT (release-ephemeral)
"Consume pending_review entry 555af9bf" -> REJECT (self-referential bookkeeping)

Examples of ACCEPTS:
"User requires CI to pass before any merge" -> generalizes across projects
"User prefers concise answers" -> durable preference

Output ONLY a JSON array. Example:
[{"text":"User prefers concise answers","type":"preference","confidence":0.9}]`

// qualityRules match atom text that violates the extractor quality
// contract: provenance tokens (session IDs, turn numbers), release
// ephemera (PR numbers, commit hashes, version tags), and self-referential
// bookkeeping. Each pattern is deliberately anchored to its noise shape to
// avoid false positives on legitimate atoms that merely contain digits.
var qualityRules = []struct {
name string
re *regexp.Regexp
}{
{"session_id", regexp.MustCompile(`\bsession\s+[0-9]{8}-[0-9a-f]{4,}`)},
{"turn_number", regexp.MustCompile(`\bturn\s+[0-9]{1,3}\s*[,)]`)},
{"pr_number", regexp.MustCompile(`(?i)\bpr\s+#[0-9]{1,6}\b`)},
{"commit_hash", regexp.MustCompile(`\b[0-9a-f]*[0-9][0-9a-f]{6,39}\b.*\b(?:merged|commit|squash)`)},
{"commit_hash_merged", regexp.MustCompile(`\b(?:merged|squash-merged|commit)\s+(?:as\s+)?[0-9a-f]*[0-9][0-9a-f]{6,39}\b`)},
{"version_tag", regexp.MustCompile(`(?i)\b(?:version\s+)?tag\s+(?:is\s+|v)?[0-9]+\.[0-9]+`)},
{"version_release", regexp.MustCompile(`\brelease\s+v?[0-9]+\.[0-9]+`)},
{"semver", regexp.MustCompile(`\bv[0-9]+\.[0-9]+\.[0-9]+\b`)},
{"pending_review_ref", regexp.MustCompile(`pending_review`)},
{"already_stored", regexp.MustCompile(`(?i)already stored`)},
}

// qualityViolation reports which rule (if any) an atom's text violates.
// The returned rule name makes drops reviewable in logs.
func qualityViolationRule(text string) (string, bool) {
for _, r := range qualityRules {
if r.re.MatchString(text) {
return r.name, true
}
}
return "", false
}

// qualityViolation reports whether atom text violates the extractor
// quality contract (provenance-in-text, release ephemera, or
// self-referential bookkeeping).
func qualityViolation(text string) bool {
_, ok := qualityViolationRule(text)
return ok
}

// untrustedRe matches nonce'd untrusted content wrappers so they can be
// stripped before extraction.
var untrustedRe = regexp.MustCompile(`(?s)<untrusted_content_[^\s>]*\s+source="[^"]*"\s*>.*?</untrusted_content_[^\s>]*>`)
Expand Down Expand Up @@ -158,6 +214,45 @@ func normalizeAtomText(text string) string {
// their retention score. Explicit LLM-provided values in (0,1] are kept.
const defaultExtractionConfidence = 0.7

// ExtractionTypeQuota caps atoms minted per atom type in one extraction
// run, so a verbose model cannot fill the store with one class of atom.
const ExtractionTypeQuota = 3

// ExtractionRunCap caps total atoms minted in one extraction run.
const ExtractionRunCap = 8

// applyExtractionQuotas quality-ranks candidate atoms and trims them to
// the per-type quota and the overall run cap. Ranking is by confidence,
// then stable order (first-seen wins ties).
func applyExtractionQuotas(atoms []MemoryAtom) []MemoryAtom {
if len(atoms) <= ExtractionRunCap && len(atoms) <= ExtractionTypeQuota {
return atoms
}
// Stable sort by confidence descending.
idx := make([]int, len(atoms))
for i := range idx {
idx[i] = i
}
for i := 1; i < len(idx); i++ {
for j := i; j > 0 && atoms[idx[j]].Confidence > atoms[idx[j-1]].Confidence; j-- {
idx[j], idx[j-1] = idx[j-1], idx[j]
}
}
perType := make(map[string]int, len(atoms))
out := make([]MemoryAtom, 0, ExtractionRunCap)
for _, i := range idx {
if len(out) >= ExtractionRunCap {
break
}
if perType[atoms[i].Type] >= ExtractionTypeQuota {
continue
}
perType[atoms[i].Type]++
out = append(out, atoms[i])
}
return out
}

// Extract atoms from text. Returns nil if the LLM is unavailable, the output
// is unparseable, or no atoms are found. Extracted atoms are sourced from the
// user ("user_said").
Expand Down Expand Up @@ -208,6 +303,12 @@ func (e *Extractor) Extract(ctx context.Context, text string) ([]MemoryAtom, err
if txt == "" {
continue
}
// Quality contract: drop atoms whose text embeds provenance or
// release ephemera. The prompt nudges; this filter enforces.
if rule, bad := qualityViolationRule(txt); bad {
log.Printf("extended memory: dropped atom violating quality contract (rule %s): %.80s", rule, txt)
continue
}
typ := r.Type
if !validType(typ) {
typ = TypeObservation
Expand All @@ -230,7 +331,7 @@ func (e *Extractor) Extract(ctx context.Context, text string) ([]MemoryAtom, err
Confidence: conf,
})
}
return atoms, nil
return applyExtractionQuotas(atoms), nil
}

func validType(t string) bool {
Expand Down
91 changes: 91 additions & 0 deletions internal/memory/extended/extractor_quality_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
package extended

import (
"context"
"strings"
"testing"
)

// TestQualityViolationDetectsProvenanceInText pins the write-time quality
// validator: atom text that embeds provenance tokens (session IDs, turn
// numbers) or release ephemera (PR numbers, commit hashes, version tags)
// must be flagged so the extractor can drop it.
func TestQualityViolationDetectsProvenanceInText(t *testing.T) {
cases := []struct {
name string
text string
want bool
}{
{"session id", "User confirmed merge in session 20260918-3e4cb01f", true},
{"turn number parenthetical", "User said merge after CI (turn 3)", true},
{"turn number comma", "decided, at turn 12, to proceed", true},
{"turn prose not provenance", "per turn 100 requests are allowed", false},
{"pr number", "PR #45 was merged", true},
{"pr number lowercase", "pr #45 was merged", true},
{"commit hash", "squash-merged as 768d380 on main", true},
{"version tag", "the correct version tag is 1.14.8", true},
{"version tag v prefix", "release v1.42.3 shipped", true},
{"semver bare", "shipped v1.42.1 yesterday", true},
{"go version not a tag", "User works with Go 1.24", false},
{"pending_review self-reference", "consume/resolve pending_review entries 555af9bf", true},
{"already stored restatement", "already stored, no change: user prefers concise answers", true},

{"clean preference", "User prefers concise answers", false},
{"clean convention", "User requires CI checks to pass before any merge", false},
{"clean fact", "User maintains odek under the BackendStack21 organization", false},
{"number that is not a tag", "User keeps sub-agent concurrency capped at 2", false},
{"hash-like word", "User likes the hashing approach", false},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got := qualityViolation(tc.text)
if got != tc.want {
t.Errorf("qualityViolation(%q) = %v, want %v", tc.text, got, tc.want)
}
})
}
}

// TestExtractorDropsProvenanceAtoms pins the extractor contract: even when
// the LLM returns atoms with provenance or release ephemera baked into the
// text, Extract must not emit them. The prompt is a mitigation, not the
// mechanism — this filter is.
func TestExtractorDropsProvenanceAtoms(t *testing.T) {
resp := `[` + strings.Join([]string{
`{"text":"User confirmed merge in session 20260918-3e4cb01f","type":"fact","confidence":0.9}`,
`{"text":"PR #45 was squash-merged as 768d380","type":"fact","confidence":0.8}`,
`{"text":"The correct version tag is 1.14.8","type":"fact","confidence":0.9}`,
`{"text":"Consume pending_review entries 555af9bf","type":"goal","confidence":0.85}`,
`{"text":"User requires CI to pass before any merge","type":"convention","confidence":0.95}`,
}, ",") + `]`
llm := newMockLLM(resp)
ex := NewExtractor(llm, DefaultConfig())
atoms, err := ex.Extract(context.Background(), "we merged it")
if err != nil {
t.Fatalf("Extract failed: %v", err)
}
if len(atoms) != 1 {
t.Fatalf("expected 1 surviving atom, got %d: %+v", len(atoms), atoms)
}
if !strings.Contains(atoms[0].Text, "CI to pass") {
t.Errorf("surviving atom should be the generalizing one, got %q", atoms[0].Text)
}
}

// TestExtractionPromptBansNoise pins that the prompt itself carries the
// negative guidance (example-first rejects), so the model is nudged before
// the write-time filter has to drop anything.
func TestExtractionPromptBansNoise(t *testing.T) {
for _, want := range []string{
"Do NOT extract",
"session ID",
"turn number",
"version",
"PR",
"REJECT",
} {
if !strings.Contains(extractionPrompt, want) {
t.Errorf("extractionPrompt missing negative-example guidance %q", want)
}
}
}
92 changes: 92 additions & 0 deletions internal/memory/extended/extractor_quota_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,92 @@
package extended

import (
"context"
"strings"
"testing"
)

// TestExtractorEnforcesTypeQuotas pins that a single extraction run cannot
// mint an unbounded number of atoms of one type: quality-ranked quota per
// type plus an overall per-run cap. A model that emits 6 goal atoms in one
// run must not store them all.
func TestExtractorEnforcesTypeQuotas(t *testing.T) {
mk := func(typ string) string {
return `{"text":"User goal ` + typ + ` number placeholder unique","type":"` + typ + `","confidence":0.9}`
}
var items []string
for i := 0; i < 6; i++ {
items = append(items, strings.Replace(mk("goal"), "placeholder", string(rune('a'+i)), 1))
}
for i := 0; i < 6; i++ {
items = append(items, strings.Replace(mk("fact"), "placeholder", string(rune('a'+i)), 1))
}
resp := `[` + strings.Join(items, ",") + `]`
llm := newMockLLM(resp)
ex := NewExtractor(llm, DefaultConfig())
atoms, err := ex.Extract(context.Background(), "lots of things")
if err != nil {
t.Fatalf("Extract failed: %v", err)
}
goals, facts := 0, 0
for _, a := range atoms {
switch a.Type {
case TypeGoal:
goals++
case TypeFact:
facts++
}
}
if goals > ExtractionTypeQuota {
t.Errorf("goal atoms = %d, want <= %d", goals, ExtractionTypeQuota)
}
if facts > ExtractionTypeQuota {
t.Errorf("fact atoms = %d, want <= %d", facts, ExtractionTypeQuota)
}
if len(atoms) > ExtractionRunCap {
t.Errorf("total atoms = %d, want <= %d", len(atoms), ExtractionRunCap)
}
}

// TestApplyExtractionQuotasSmallBatchPassthrough pins that a batch within
// both caps passes through untouched, in original order.
func TestApplyExtractionQuotasSmallBatchPassthrough(t *testing.T) {
atoms := []MemoryAtom{
{Text: "a", Type: TypeFact, Confidence: 0.1},
{Text: "b", Type: TypeFact, Confidence: 0.9},
}
got := applyExtractionQuotas(atoms)
if len(got) != 2 || got[0].Text != "a" || got[1].Text != "b" {
t.Errorf("small batch must pass through in order, got %+v", got)
}
}

// TestExtractorTypeQuotaKeepsHighestConfidence pins that quota trimming is
// quality-ranked: when over quota, the highest-confidence atoms survive.
func TestExtractorTypeQuotaKeepsHighestConfidence(t *testing.T) {
resp := `[` + strings.Join([]string{
`{"text":"low confidence goal","type":"goal","confidence":0.3}`,
`{"text":"high confidence goal","type":"goal","confidence":0.95}`,
`{"text":"mid confidence goal","type":"goal","confidence":0.6}`,
`{"text":"another high goal","type":"goal","confidence":0.9}`,
}, ",") + `]`
llm := newMockLLM(resp)
ex := NewExtractor(llm, DefaultConfig())
atoms, err := ex.Extract(context.Background(), "goals")
if err != nil {
t.Fatalf("Extract failed: %v", err)
}
if len(atoms) != ExtractionTypeQuota {
t.Fatalf("expected %d atoms after quota, got %d", ExtractionTypeQuota, len(atoms))
}
kept := map[string]bool{}
for _, a := range atoms {
kept[a.Text] = true
}
if !kept["high confidence goal"] || !kept["another high goal"] {
t.Errorf("quota must keep the highest-confidence atoms, kept %v", kept)
}
if kept["low confidence goal"] {
t.Errorf("quota must drop the lowest-confidence atom first, kept %v", kept)
}
}
1 change: 1 addition & 0 deletions internal/memory/extended/recall.go
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,7 @@ func (r *Recall) Query(ctx context.Context, query string, recent []string, state
log.Printf("extended memory: recall query failed: %v", err)
return "", err
}
res = filterExpiredAtoms(res, r.cfg.EphemeralTTLDays)
if len(res) == 0 {
return "", nil
}
Expand Down
Loading
Loading