Skip to content

Commit da507ca

Browse files
committed
fix(sdk): semantic memory stores assistant text, not raw JSON output dumps
run() recorded `Task outcome: {"type":"lastMessage","value":[...]}` into semantic memory on every completed run — 437 identical, information-free facts accumulated in the dev database alone. recall() then injected those blobs into later prompts, crowding out real facts. extractOutcomeText now stores the last assistant text (truncated) for lastMessage/allMessages outputs, the JSON value for structuredOutput, and nothing for errors. E2e mocks made hermetic: SemanticMemoryStore is replaced with an in-memory fake and generateRepoMap suppressed, so tests no longer read or write <cwd>/.levelcode/memory.db (the mock prompt-matcher also strips the params-JSON block the runtime appends after the user message, whose quotes hijacked the quote-echo rule). sdk suite: 440 pass / 0 fail (9 e2e failures fixed), e2e wall time ~4x faster.
1 parent 66196c2 commit da507ca

2 files changed

Lines changed: 94 additions & 4 deletions

File tree

‎sdk/e2e/utils/e2e-mocks.ts‎

Lines changed: 51 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,15 @@
11
import { models } from '@levelcode/common/old-constants'
2+
import * as semanticMemoryModule from '@levelcode/common/memory/semantic-memory'
23
import { promptSuccess } from '@levelcode/common/util/error'
34
import { spyOn } from 'bun:test'
45
import z from 'zod/v4'
56

67
import { LevelCodeClient } from '../../src/client'
78
import * as databaseModule from '../../src/impl/database'
89
import * as llmModule from '../../src/impl/llm'
10+
import * as repoMapModule from '../../src/tools/repo-map'
11+
12+
import type { SemanticMemoryStore } from '@levelcode/common/memory/semantic-memory'
913

1014
import type { AgentTemplate } from '@levelcode/common/types/agent-template'
1115
import type {
@@ -98,8 +102,20 @@ function extractLatestUserMessage(text: string): string | null {
98102
return matches[matches.length - 1]?.[1] ?? null
99103
}
100104

105+
/**
106+
* The runtime appends run params (repo map, etc.) to the user message as a
107+
* pretty-printed JSON block: `\n\n{\n "repoMap": ...`. Those quotes would
108+
* hijack the quote-echo rule below, so strip everything from the params
109+
* block onward — the tests match on what the user actually typed.
110+
*/
111+
function stripParamsJsonBlock(text: string): string {
112+
const paramsStart = text.search(/\n\n\{\n "/)
113+
return paramsStart === -1 ? text : text.slice(0, paramsStart)
114+
}
115+
101116
function getPromptText(latestUserText: string, allText: string): string {
102-
return extractLatestUserMessage(allText) ?? latestUserText
117+
const extracted = extractLatestUserMessage(allText) ?? latestUserText
118+
return stripParamsJsonBlock(extracted)
103119
}
104120

105121
function splitTextIntoChunks(text: string): string[] {
@@ -412,5 +428,39 @@ export function setupE2eMocks(): void {
412428
promptAiSdkStructuredMock as typeof llmModule.promptAiSdkStructured,
413429
)
414430

431+
// Hermetic semantic memory: run() persists `Task outcome` facts into
432+
// `<cwd>/.levelcode/memory.db` and recalls them into later prompts. Those
433+
// JSON blobs break the deterministic prompt-matching below (the quote-echo
434+
// rule answers with the first quoted string, e.g. "type") and leak between
435+
// test runs. Replace the store with an in-process fake: no disk, no cross-
436+
// run pollution, recall stays empty.
437+
spyOn(semanticMemoryModule, 'SemanticMemoryStore').mockImplementation(
438+
function mockStore(this: unknown, _cwd?: string) {
439+
const facts: Array<{ fact: string; metadata: Record<string, unknown> }> =
440+
[]
441+
const store = {
442+
remember: (fact: string, metadata: Record<string, unknown> = {}) => {
443+
facts.push({ fact, metadata })
444+
return {
445+
id: `mock-mem-${facts.length}`,
446+
fact,
447+
metadata,
448+
timestamp: Date.now(),
449+
}
450+
},
451+
recall: () => [],
452+
forget: () => false,
453+
searchByTag: () => [],
454+
}
455+
return store
456+
} as unknown as typeof SemanticMemoryStore,
457+
)
458+
459+
// Hermetic repo map: run() scans the current working directory (the sdk
460+
// source tree during tests) and appends the result to every prompt —
461+
// slow, environment-dependent, and its quotes also trip the quote-echo
462+
// rule. E2e tests get no repo map.
463+
spyOn(repoMapModule, 'generateRepoMap').mockImplementation(async () => '')
464+
415465
spyOn(LevelCodeClient.prototype, 'checkConnection').mockResolvedValue(true)
416466
}

‎sdk/src/run.ts‎

Lines changed: 43 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -49,6 +49,7 @@ import type { CustomToolDefinition } from './custom-tool'
4949
import type { RunState } from './run-state'
5050
import type { FileFilter } from './tools/read-files'
5151
import type { ServerAction } from '@levelcode/common/actions'
52+
import type { AgentOutput } from '@levelcode/common/types/session-state'
5253
import type { AgentDefinition } from '@levelcode/common/templates/initial-agents-dir/types/agent-definition'
5354
import type {
5455
PublishedToolName,
@@ -728,6 +729,47 @@ function requireCwd(cwd: string | undefined, toolName: string): string {
728729
return cwd
729730
}
730731

732+
const MAX_OUTCOME_MEMORY_CHARS = 500
733+
734+
/**
735+
* Human-readable summary of a run's output for semantic memory. Prefers the
736+
* assistant's actual words over raw JSON: remembering
737+
* `{"type":"lastMessage","value":[]}`-style dumps pollutes recall with
738+
* hundreds of identical, information-free facts that crowd out real ones.
739+
*/
740+
function extractOutcomeText(output: AgentOutput): string {
741+
if (output.type === 'error') {
742+
// Failed runs are not facts worth recalling.
743+
return ''
744+
}
745+
if (output.type === 'lastMessage' || output.type === 'allMessages') {
746+
const messages = Array.isArray(output.value) ? output.value : []
747+
for (let i = messages.length - 1; i >= 0; i--) {
748+
const message = messages[i]
749+
if (message?.role !== 'assistant') continue
750+
const content = message.content
751+
if (typeof content === 'string') {
752+
return content.slice(0, MAX_OUTCOME_MEMORY_CHARS)
753+
}
754+
if (Array.isArray(content)) {
755+
const text = content
756+
.filter((part) => part?.type === 'text')
757+
.map((part) => (part as { text?: string }).text ?? '')
758+
.join(' ')
759+
.trim()
760+
if (text) {
761+
return text.slice(0, MAX_OUTCOME_MEMORY_CHARS)
762+
}
763+
}
764+
}
765+
// No assistant text at all (e.g. a tool-only last turn) — nothing to keep.
766+
return ''
767+
}
768+
// structuredOutput: the value is the machine-checkable result.
769+
const json = JSON.stringify(output.value)
770+
return json === undefined ? '' : json.slice(0, MAX_OUTCOME_MEMORY_CHARS)
771+
}
772+
731773
async function readFiles({
732774
filePaths,
733775
override,
@@ -1209,9 +1251,7 @@ async function handlePromptResponse({
12091251
} catch { /* tracing cleanup non-fatal */ }
12101252
try {
12111253
if (middleware?.semanticMemory && output) {
1212-
const outText = typeof output === 'object' && 'message' in output
1213-
? String((output as any).message ?? '').slice(0, 500)
1214-
: JSON.stringify(output).slice(0, 500)
1254+
const outText = extractOutcomeText(output)
12151255
if (outText) {
12161256
middleware.semanticMemory.remember(`Task outcome: ${outText}`, {
12171257
tags: ['outcome'],

0 commit comments

Comments
 (0)