Skip to content

Commit d2aaa8b

Browse files
author
Thatgfsj
committed
v1.4.1: revision experiments data + scripts
- LoCoMo-10 revision batch: baselines (Vanilla RAG 22.4 / BM25 32.1 / Hybrid 32.2) + A1-A5 signal ablations; NWC HybridFusion 43.0% - bootstrap statistics (statistics.json): all p<0.001 vs full system - LongMemEval-small: NWC 38.8% vs Hybrid 32.0% - Mem0 baseline (raw storage, same protocol): 14.2% - cross-LLM judge study (MiniMax abab6.5s / Text-01): +14.8 / +19.0pp - bump version to 1.4.1
1 parent 2f829ec commit d2aaa8b

22 files changed

Lines changed: 1540 additions & 2 deletions

‎benchmarks/make_figures.py‎

Lines changed: 222 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,222 @@
1+
"""Revision v2.0 paper figures.
2+
3+
Reads revision_results/*.json and renders vector PDFs to paper/figures/:
4+
fig1_architecture.pdf — six-layer hierarchy + signal flow (static diagram)
5+
fig2_main_heatmap.pdf — systems × categories on LoCoMo-10
6+
fig3_ablation.pdf — ablation bars with bootstrap CI
7+
fig4_growth.pdf — memory compression: input tokens → retained anchors
8+
fig5_longmemeval.pdf — systems × tasks on LongMemEval (if data present)
9+
10+
Usage:
11+
python benchmarks/make_figures.py
12+
"""
13+
14+
from __future__ import annotations
15+
16+
import json
17+
import os
18+
import sys
19+
20+
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
21+
22+
import matplotlib
23+
matplotlib.use('Agg')
24+
import matplotlib.pyplot as plt
25+
import numpy as np
26+
27+
FIG_DIR = os.path.join(os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
28+
'paper', 'figures')
29+
RES_DIR = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'revision_results')
30+
31+
CATS = ['Temporal', 'Short', 'Long', 'Composite', 'Adversarial']
32+
SYS_SHORT = {
33+
'vanilla_rag': 'Vanilla RAG', 'bm25': 'BM25', 'hybrid_rrf': 'Hybrid',
34+
'nwc_full': 'NWC Hyb.', 'a1_no_lifecycle': 'A1 No Lifecycle',
35+
'a2_no_activation': 'A2 No Activation', 'a3_no_lexical': 'A3 No Lexical',
36+
'a4_no_confidence': 'A4 No Confidence', 'a5_single_layer': 'A5 Single Layer',
37+
}
38+
39+
40+
def fig1_architecture():
41+
fig, ax = plt.subplots(figsize=(6.5, 4.2))
42+
ax.axis('off')
43+
layers = [
44+
('L6 Identity', 'cognitive profile ~600 tok'),
45+
('L5 Belief', 'strength + evidence'),
46+
('L4 Pattern', 'behaviour patterns'),
47+
('L3 Semantic', 'concept graph, typed edges'),
48+
('L2 Episodic', 'memory anchors'),
49+
('L1 Sensory', 'importance gate'),
50+
]
51+
y = 5.2
52+
for name, desc in layers:
53+
ax.add_patch(plt.Rectangle((0.1, y - 0.32), 2.6, 0.55,
54+
facecolor='#dbe9f7', edgecolor='#2b6cb0',
55+
linewidth=1.2, zorder=3))
56+
ax.text(1.4, y + 0.05, name, ha='center', va='center',
57+
fontsize=10, fontweight='bold', zorder=4)
58+
ax.text(3.1, y, desc, ha='left', va='center', fontsize=8.5,
59+
color='#444', zorder=4)
60+
if y < 5.2:
61+
ax.annotate('', xy=(1.4, y - 0.3), xytext=(1.4, y + 0.26),
62+
arrowprops=dict(arrowstyle='->', color='#2b6cb0',
63+
lw=1.4), zorder=2)
64+
y -= 0.78
65+
ax.text(0.1, 0.05,
66+
'sleep consolidation | three-dimensional decay | '
67+
'spreading-activation retrieval',
68+
fontsize=9, color='#c05621', style='italic')
69+
ax.text(5.0, 5.45, 'compression: tokens → structure', fontsize=9,
70+
color='#2b6cb0')
71+
ax.set_xlim(0, 6.4)
72+
ax.set_ylim(0, 6)
73+
fig.tight_layout()
74+
fig.savefig(os.path.join(FIG_DIR, 'fig1_architecture.pdf'))
75+
plt.close(fig)
76+
77+
78+
def fig2_main_heatmap(summary):
79+
systems = ['vanilla_rag', 'bm25', 'hybrid_rrf', 'nwc_full']
80+
labels = [SYS_SHORT[s] for s in systems]
81+
data = np.array([
82+
[summary[s]['by_category'].get(c, {}).get('acc', np.nan) for c in CATS]
83+
for s in systems
84+
])
85+
fig, ax = plt.subplots(figsize=(6.2, 2.6))
86+
im = ax.imshow(data, cmap='YlGnBu', vmin=0, vmax=70, aspect='auto')
87+
ax.set_xticks(range(len(CATS)))
88+
ax.set_xticklabels(CATS, fontsize=9)
89+
ax.set_yticks(range(len(systems)))
90+
ax.set_yticklabels(labels, fontsize=9)
91+
for i in range(len(systems)):
92+
for j in range(len(CATS)):
93+
v = data[i, j]
94+
ax.text(j, i, f'{v:.0f}' if not np.isnan(v) else '--',
95+
ha='center', va='center', fontsize=8,
96+
color='black' if v > 40 else 'white')
97+
ax.set_title('LoCoMo-10 has-answer accuracy (%) by category',
98+
fontsize=10)
99+
fig.colorbar(im, ax=ax, fraction=0.03, pad=0.03)
100+
fig.tight_layout()
101+
fig.savefig(os.path.join(FIG_DIR, 'fig2_main_heatmap.pdf'))
102+
plt.close(fig)
103+
104+
105+
def fig3_ablation(stats):
106+
configs = ['nwc_full', 'a1_no_lifecycle', 'a2_no_activation',
107+
'a3_no_lexical', 'a4_no_confidence', 'a5_single_layer']
108+
labels = [SYS_SHORT[c] for c in configs]
109+
accs = [stats[c]['acc'] * 100 for c in configs]
110+
los = [stats[c]['ci95'][0] * 100 for c in configs]
111+
his = [stats[c]['ci95'][1] * 100 for c in configs]
112+
err = [[acc - lo for acc, lo in zip(accs, los)],
113+
[hi - acc for acc, hi in zip(accs, his)]]
114+
fig, ax = plt.subplots(figsize=(6.2, 3.2))
115+
colors = ['#2b6cb0'] + ['#cbd5e0'] * 4 + ['#e2e8f0']
116+
bars = ax.bar(range(len(configs)), accs, yerr=err, capsize=4,
117+
color=colors, edgecolor='#2d3748', linewidth=0.8)
118+
ax.set_xticks(range(len(configs)))
119+
ax.set_xticklabels(labels, rotation=20, ha='right', fontsize=8.5)
120+
ax.set_ylabel('has-answer accuracy (%)')
121+
for i, (acc, hi) in enumerate(zip(accs, his)):
122+
ax.text(i, hi + 0.8, f'{acc:.1f}', ha='center', fontsize=8.5)
123+
ax.set_ylim(0, max(his) * 1.15)
124+
ax.set_title('Ablation: contribution of each fusion signal (LoCoMo-10)',
125+
fontsize=10)
126+
ax.spines[['top', 'right']].set_visible(False)
127+
fig.tight_layout()
128+
fig.savefig(os.path.join(FIG_DIR, 'fig3_ablation.pdf'))
129+
plt.close(fig)
130+
131+
132+
def fig4_growth():
133+
"""Tokens in vs anchors retained per conversation (real data from
134+
E:\\shiyan\\results\\locomo_v140_results.json + the LoCoMo dataset)."""
135+
import re
136+
from benchmarks.run_locomo_full import load_locomo, extract_all_turns
137+
dataset = load_locomo(r'E:\locomo-10\data\locomo10.json')
138+
with open(r'E:\shiyan\results\locomo_v140_results.json') as f:
139+
res = json.load(f)
140+
conv_map = {c['id']: c for c in res['conversations']}
141+
tokens_in, anchors, conv_ids = [], [], []
142+
for conv in dataset:
143+
cid = conv.get('sample_id', '')
144+
turns, _ = extract_all_turns(conv)
145+
tokens = sum(len(t['text'].split()) for t in turns)
146+
if cid in conv_map:
147+
tokens_in.append(tokens)
148+
anchors.append(conv_map[cid]['anchors'])
149+
conv_ids.append(cid)
150+
fig, ax = plt.subplots(figsize=(6.2, 3.2))
151+
ax.scatter(tokens_in, anchors, c='#2b6cb0', s=40, edgecolor='white')
152+
for x, y, cid in zip(tokens_in, anchors, conv_ids):
153+
ax.annotate(cid.replace('conv-', ''), (x, y), fontsize=7,
154+
xytext=(4, 4), textcoords='offset points')
155+
z = np.polyfit(tokens_in, anchors, 1)
156+
xs = np.linspace(min(tokens_in), max(tokens_in), 50)
157+
ax.plot(xs, np.polyval(z, xs), '--', color='#c05621', lw=1)
158+
ax.set_xlabel('input tokens per conversation')
159+
ax.set_ylabel('anchors retained')
160+
ax.set_title('Memory growth is sublinear: anchors vs input tokens',
161+
fontsize=10)
162+
ax.spines[['top', 'right']].set_visible(False)
163+
fig.tight_layout()
164+
fig.savefig(os.path.join(FIG_DIR, 'fig4_growth.pdf'))
165+
plt.close(fig)
166+
167+
168+
def fig5_longmemeval(lme):
169+
if not os.path.exists(lme):
170+
print('[skip] fig5: no longmemeval data')
171+
return
172+
with open(lme) as f:
173+
d = json.load(f)
174+
ov = d.get('overall', {})
175+
labels = [SYS_SHORT.get(k, k) for k in ov]
176+
vals = [v['has_answer_pct'] for v in ov.values()]
177+
fig, ax = plt.subplots(figsize=(6.2, 3.0))
178+
ax.bar(range(len(labels)), vals, color='#2b6cb0',
179+
edgecolor='#2d3748', linewidth=0.8)
180+
ax.set_xticks(range(len(labels)))
181+
ax.set_xticklabels(labels, rotation=20, ha='right', fontsize=8.5)
182+
ax.set_ylabel('has-answer accuracy (%)')
183+
for i, v in enumerate(vals):
184+
ax.text(i, v + 1, f'{v:.1f}', ha='center', fontsize=8.5)
185+
ax.set_ylim(0, max(vals) * 1.15)
186+
ax.set_title('LongMemEval-small: has-answer accuracy', fontsize=10)
187+
ax.spines[['top', 'right']].set_visible(False)
188+
fig.tight_layout()
189+
fig.savefig(os.path.join(FIG_DIR, 'fig5_longmemeval.pdf'))
190+
plt.close(fig)
191+
192+
193+
def main():
194+
os.makedirs(FIG_DIR, exist_ok=True)
195+
fig1_architecture()
196+
print('fig1_architecture.pdf done')
197+
198+
summary_path = os.path.join(RES_DIR, 'revision_summary.json')
199+
if os.path.exists(summary_path):
200+
with open(summary_path) as f:
201+
summary = json.load(f)
202+
fig2_main_heatmap(summary)
203+
print('fig2_main_heatmap.pdf done')
204+
else:
205+
print('[skip] fig2: no revision_summary.json')
206+
207+
stats_path = os.path.join(RES_DIR, 'statistics.json')
208+
if os.path.exists(stats_path):
209+
with open(stats_path) as f:
210+
stats = json.load(f)
211+
fig3_ablation(stats)
212+
print('fig3_ablation.pdf done')
213+
else:
214+
print('[skip] fig3: no statistics.json')
215+
216+
fig4_growth()
217+
print('fig4_growth.pdf done')
218+
fig5_longmemeval(os.path.join(RES_DIR, 'longmemeval_summary.json'))
219+
220+
221+
if __name__ == '__main__':
222+
main()

0 commit comments

Comments
 (0)