From 2d94244e39adfd1f5dfdfdeee8766abd1326f3a0 Mon Sep 17 00:00:00 2001 From: Yoshiharu Uematsu Date: Sat, 3 Oct 2026 19:36:49 +0900 Subject: [PATCH] fix(search): limit same_entity recency marker to diff/doc rows MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit markRecency now takes each row's type and only compares diff and doc rows. Issue/PR/comment/review rows get no marker and do not enter the comparison. issue とコメントは同一 entity の版ではないため、recency は diff/doc 行に限定する。 ツール説明文も同様に更新し、issue + コメント群で marker が付かないテストを追加。 Refs #269 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01XiAGywnJ6R9ibkNrxg9fry --- src/aggregate.test.ts | 36 +++++++++++++++++++++++++++--------- src/aggregate.ts | 6 +++++- src/mcp.ts | 11 +++++++---- 3 files changed, 39 insertions(+), 14 deletions(-) diff --git a/src/aggregate.test.ts b/src/aggregate.test.ts index 1158f52..95e1695 100644 --- a/src/aggregate.test.ts +++ b/src/aggregate.test.ts @@ -245,28 +245,46 @@ describe("duplicate rate regression (issue #189 measurements, 2026-08-01)", () = }); describe("markRecency (issue #267)", () => { + it("marks nothing for an issue and its comments (issue #269)", () => { + const m = markRecency([ + { id: "issue", type: "issue", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "c1", type: "issue_comment", updatedAt: "2026-09-20T00:00:00Z" }, + { id: "c2", type: "issue_comment", updatedAt: "2026-09-29T00:00:00Z" }, + ]); + expect(m.size).toBe(0); + }); + it("compares only diff/doc rows in a mixed group (issue #269)", () => { + const m = markRecency([ + { id: "doc", type: "doc", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "diff", type: "diff", updatedAt: "2026-09-20T00:00:00Z" }, + { id: "other", type: "issue", updatedAt: "2026-09-29T00:00:00Z" }, + ]); + expect(m.get("doc")).toBe("superseded"); + expect(m.get("diff")).toBe("latest"); + expect(m.has("other")).toBe(false); + }); it("marks older rows superseded and newest latest", () => { const m = markRecency([ - { id: "old", updatedAt: "2026-09-15T00:00:00Z" }, - { id: "new", updatedAt: "2026-09-29T00:00:00Z" }, - { id: "mid", updatedAt: "2026-09-23T00:00:00Z" }, + { id: "old", type: "diff", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "new", type: "diff", updatedAt: "2026-09-29T00:00:00Z" }, + { id: "mid", type: "diff", updatedAt: "2026-09-23T00:00:00Z" }, ]); expect(m.get("old")).toBe("superseded"); expect(m.get("mid")).toBe("superseded"); expect(m.get("new")).toBe("latest"); }); it("marks nothing for a single row or equal timestamps", () => { - expect(markRecency([{ id: "a", updatedAt: "2026-09-15T00:00:00Z" }]).size).toBe(0); + expect(markRecency([{ id: "a", type: "diff", updatedAt: "2026-09-15T00:00:00Z" }]).size).toBe(0); expect(markRecency([ - { id: "a", updatedAt: "2026-09-15T00:00:00Z" }, - { id: "b", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "a", type: "diff", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "b", type: "diff", updatedAt: "2026-09-15T00:00:00Z" }, ]).size).toBe(0); }); it("leaves unparsable timestamps unmarked", () => { const m = markRecency([ - { id: "a", updatedAt: "" }, - { id: "b", updatedAt: "2026-09-15T00:00:00Z" }, - { id: "c", updatedAt: "2026-09-16T00:00:00Z" }, + { id: "a", type: "diff", updatedAt: "" }, + { id: "b", type: "diff", updatedAt: "2026-09-15T00:00:00Z" }, + { id: "c", type: "diff", updatedAt: "2026-09-16T00:00:00Z" }, ]); expect(m.has("a")).toBe(false); expect(m.get("c")).toBe("latest"); diff --git a/src/aggregate.ts b/src/aggregate.ts index 24e749f..e50da2d 100644 --- a/src/aggregate.ts +++ b/src/aggregate.ts @@ -137,10 +137,14 @@ export type Recency = "latest" | "superseded"; * "latest" for the newest. Returns an empty map when the group holds fewer * than two distinct timestamps — the pool proves nothing about recency then. * Rows with an unparsable timestamp are left unmarked. + * Only `diff` and `doc` rows take part (issue #269): an issue's comments are + * separate events, not versions of the issue, so other types are never + * marked and never enter the comparison. */ -export function markRecency(rows: Array<{ id: string; updatedAt: string | null | undefined }>): Map { +export function markRecency(rows: Array<{ id: string; type: string; updatedAt: string | null | undefined }>): Map { const out = new Map(); const times = rows + .filter((r) => r.type === "diff" || r.type === "doc") .map((r) => ({ id: r.id, t: r.updatedAt ? Date.parse(r.updatedAt) : NaN })) .filter((r) => !Number.isNaN(r.t)); if (new Set(times.map((r) => r.t)).size < 2) return out; diff --git a/src/mcp.ts b/src/mcp.ts index dfbe2fb..53d83a6 100644 --- a/src/mcp.ts +++ b/src/mcp.ts @@ -338,8 +338,8 @@ export function createRagMcpServer(env: Env): McpServer { "Results are aggregated per underlying entity: a file's doc row and its commit diffs are one result, " + "an issue or PR and its comments / reviews are one result. top_k therefore counts distinct entities, " + "and a result that absorbed others carries same_entity { count, others[] } with links to them. " + - "When those rows hold more than one distinct updated_at, the representative and each others entry carry recency: " + - "\"superseded\" = a newer row of the same entity is in the pool, \"latest\" = the newest row in the pool. " + + "Among diff and doc rows only, when those rows hold more than one distinct updated_at, each such row (representative or others entry) carries recency: " + + "\"superseded\" = a newer diff/doc row of the same entity is in the pool, \"latest\" = the newest such row in the pool. Issues, PRs, comments and reviews never carry recency. " + "Absent recency is not a claim of being latest, and recency never affects ranking or scores.\n" + "Every result row — and every same_entity.others entry — carries vector_id, the handle mode 4 takes. " + "It is a handle for reaching a row you just found, not a durable identifier: the id scheme has been " + @@ -943,8 +943,11 @@ export function createRagMcpServer(env: Env): McpServer { // dropped) so the caller can still reach every version / comment. const folded = collapsedInto.get(f.vectorId) ?? []; const recency = markRecency([ - { id: f.vectorId, updatedAt: r.updatedAt }, - ...folded.map((o) => ({ id: o.vectorId, updatedAt: resolveRow(payload.get(o.vectorId)).updatedAt })), + { id: f.vectorId, type: r.type, updatedAt: r.updatedAt }, + ...folded.map((o) => { + const or = resolveRow(payload.get(o.vectorId)); + return { id: o.vectorId, type: or.type, updatedAt: or.updatedAt }; + }), ]); const sameEntity = folded.length > 0