Skip to content

Commit 8ee45de

Browse files
macosmacos
authored andcommitted
feat: 登录页增加记住密码 checkbox + ONNX INT8 量化 + bump v1.0.18
1 parent f7af0de commit 8ee45de

29 files changed

Lines changed: 402 additions & 194 deletions

File tree

‎aivectormemory/db/memory_repo.py‎

Lines changed: 57 additions & 19 deletions
Original file line numberDiff line numberDiff line change
@@ -30,10 +30,11 @@ def _match_filters(self, mem, filters) -> bool:
3030
source = filters.get("source")
3131
return not source or mem.get("source", "manual") == source
3232

33-
def list_by_tags(self, tags: list[str], scope: str = "all", project_dir: str = "",
34-
limit: int = 100, source: str | None = None,
35-
tags_mode: str = "all", **_) -> list[dict]:
36-
sql, params = "SELECT * FROM memories WHERE 1=1", []
33+
def _build_tag_filter(self, base_sql: str, tags: list[str],
34+
tags_mode: str, scope: str = "all",
35+
project_dir: str = "", source: str | None = None,
36+
query: str | None = None) -> tuple[str, list]:
37+
sql, params = base_sql, []
3738
if scope == "project":
3839
sql += " AND project_dir=?"
3940
params.append(project_dir or self.project_dir)
@@ -48,25 +49,62 @@ def list_by_tags(self, tags: list[str], scope: str = "all", project_dir: str = "
4849
for tag in tags:
4950
sql += " AND id IN (SELECT memory_id FROM memory_tags WHERE tag=?)"
5051
params.append(tag)
51-
sql += " ORDER BY created_at DESC LIMIT ?"
52-
params.append(limit)
52+
if query:
53+
sql += " AND content LIKE ?"
54+
params.append(f"%{query}%")
55+
return sql, params
56+
57+
def list_by_tags(self, tags: list[str], scope: str = "all", project_dir: str = "",
58+
limit: int = 100, offset: int = 0, source: str | None = None,
59+
tags_mode: str = "all", query: str | None = None, **_) -> list[dict]:
60+
sql, params = self._build_tag_filter(
61+
"SELECT * FROM memories WHERE 1=1",
62+
tags, tags_mode, scope, project_dir, source, query)
63+
sql += " ORDER BY created_at DESC LIMIT ? OFFSET ?"
64+
params.extend([limit, offset])
5365
return [dict(r) for r in self.conn.execute(sql, params).fetchall()]
5466

55-
def get_all(self, limit: int = 100, offset: int = 0, project_dir: str | None = None) -> list[dict]:
56-
if project_dir is not None:
57-
rows = self.conn.execute(
58-
"SELECT * FROM memories WHERE project_dir = ? ORDER BY created_at DESC LIMIT ? OFFSET ?",
59-
(project_dir, limit, offset)).fetchall()
60-
else:
61-
rows = self.conn.execute(
62-
"SELECT * FROM memories ORDER BY created_at DESC LIMIT ? OFFSET ?",
63-
(limit, offset)).fetchall()
64-
return [dict(r) for r in rows]
67+
def count_by_tags(self, tags: list[str], scope: str = "all", project_dir: str = "",
68+
source: str | None = None, tags_mode: str = "all",
69+
query: str | None = None) -> int:
70+
sql, params = self._build_tag_filter(
71+
"SELECT COUNT(*) FROM memories WHERE 1=1",
72+
tags, tags_mode, scope, project_dir, source, query)
73+
return self.conn.execute(sql, params).fetchone()[0]
6574

66-
def count(self, project_dir: str | None = None) -> int:
75+
def _build_filter(self, base_sql: str, project_dir: str | None = None,
76+
query: str | None = None, source: str | None = None,
77+
exclude_tags: list[str] | None = None) -> tuple[str, list]:
78+
sql, params = base_sql, []
6779
if project_dir is not None:
68-
return self.conn.execute("SELECT COUNT(*) FROM memories WHERE project_dir=?", (project_dir,)).fetchone()[0]
69-
return self.conn.execute("SELECT COUNT(*) FROM memories").fetchone()[0]
80+
sql += " AND project_dir = ?"
81+
params.append(project_dir)
82+
if query:
83+
sql += " AND content LIKE ?"
84+
params.append(f"%{query}%")
85+
if source:
86+
sql += " AND source = ?"
87+
params.append(source)
88+
if exclude_tags:
89+
for tag in exclude_tags:
90+
sql += " AND id NOT IN (SELECT memory_id FROM memory_tags WHERE tag=?)"
91+
params.append(tag)
92+
return sql, params
93+
94+
def get_all(self, limit: int = 100, offset: int = 0, project_dir: str | None = None,
95+
query: str | None = None, source: str | None = None,
96+
exclude_tags: list[str] | None = None) -> list[dict]:
97+
sql, params = self._build_filter(
98+
"SELECT * FROM memories WHERE 1=1", project_dir, query, source, exclude_tags)
99+
sql += " ORDER BY created_at DESC LIMIT ? OFFSET ?"
100+
params.extend([limit, offset])
101+
return [dict(r) for r in self.conn.execute(sql, params).fetchall()]
102+
103+
def count(self, project_dir: str | None = None, query: str | None = None,
104+
source: str | None = None, exclude_tags: list[str] | None = None) -> int:
105+
sql, params = self._build_filter(
106+
"SELECT COUNT(*) FROM memories WHERE 1=1", project_dir, query, source, exclude_tags)
107+
return self.conn.execute(sql, params).fetchone()[0]
70108

71109
def get_tag_counts(self, project_dir: str | None = None) -> dict[str, int]:
72110
if project_dir is not None:

‎aivectormemory/db/user_memory_repo.py‎

Lines changed: 54 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -17,9 +17,10 @@ def _build_insert(self, mid, content, tags, source, session_id, now, extra):
1717
vals = [mid, content, json.dumps(tags, ensure_ascii=False), source, session_id, now, now]
1818
return cols, vals
1919

20-
def list_by_tags(self, tags: list[str], limit: int = 100, source: str | None = None,
21-
tags_mode: str = "all", **_) -> list[dict]:
22-
sql, params = "SELECT * FROM user_memories WHERE 1=1", []
20+
def _build_tag_filter(self, base_sql: str, tags: list[str],
21+
tags_mode: str, source: str | None = None,
22+
query: str | None = None) -> tuple[str, list]:
23+
sql, params = base_sql, []
2324
if source:
2425
sql += " AND source=?"
2526
params.append(source)
@@ -31,18 +32,59 @@ def list_by_tags(self, tags: list[str], limit: int = 100, source: str | None = N
3132
for tag in tags:
3233
sql += " AND id IN (SELECT memory_id FROM user_memory_tags WHERE tag=?)"
3334
params.append(tag)
34-
sql += " ORDER BY created_at DESC LIMIT ?"
35-
params.append(limit)
35+
if query:
36+
sql += " AND content LIKE ?"
37+
params.append(f"%{query}%")
38+
return sql, params
39+
40+
def list_by_tags(self, tags: list[str], limit: int = 100, offset: int = 0,
41+
source: str | None = None, tags_mode: str = "all",
42+
query: str | None = None, **_) -> list[dict]:
43+
sql, params = self._build_tag_filter(
44+
"SELECT * FROM user_memories WHERE 1=1",
45+
tags, tags_mode, source, query)
46+
sql += " ORDER BY created_at DESC LIMIT ? OFFSET ?"
47+
params.extend([limit, offset])
3648
return [dict(r) for r in self.conn.execute(sql, params).fetchall()]
3749

38-
def get_all(self, limit: int = 100, offset: int = 0) -> list[dict]:
39-
rows = self.conn.execute(
40-
"SELECT * FROM user_memories ORDER BY created_at DESC LIMIT ? OFFSET ?",
41-
(limit, offset)).fetchall()
42-
return [dict(r) for r in rows]
50+
def count_by_tags(self, tags: list[str], source: str | None = None,
51+
tags_mode: str = "all", query: str | None = None) -> int:
52+
sql, params = self._build_tag_filter(
53+
"SELECT COUNT(*) FROM user_memories WHERE 1=1",
54+
tags, tags_mode, source, query)
55+
return self.conn.execute(sql, params).fetchone()[0]
56+
57+
def get_all(self, limit: int = 100, offset: int = 0, query: str | None = None,
58+
source: str | None = None, exclude_tags: list[str] | None = None) -> list[dict]:
59+
sql, params = "SELECT * FROM user_memories WHERE 1=1", []
60+
if query:
61+
sql += " AND content LIKE ?"
62+
params.append(f"%{query}%")
63+
if source:
64+
sql += " AND source=?"
65+
params.append(source)
66+
if exclude_tags:
67+
for tag in exclude_tags:
68+
sql += " AND id NOT IN (SELECT memory_id FROM user_memory_tags WHERE tag=?)"
69+
params.append(tag)
70+
sql += " ORDER BY created_at DESC LIMIT ? OFFSET ?"
71+
params.extend([limit, offset])
72+
return [dict(r) for r in self.conn.execute(sql, params).fetchall()]
4373

44-
def count(self) -> int:
45-
return self.conn.execute("SELECT COUNT(*) FROM user_memories").fetchone()[0]
74+
def count(self, query: str | None = None, source: str | None = None,
75+
exclude_tags: list[str] | None = None) -> int:
76+
sql, params = "SELECT COUNT(*) FROM user_memories WHERE 1=1", []
77+
if query:
78+
sql += " AND content LIKE ?"
79+
params.append(f"%{query}%")
80+
if source:
81+
sql += " AND source=?"
82+
params.append(source)
83+
if exclude_tags:
84+
for tag in exclude_tags:
85+
sql += " AND id NOT IN (SELECT memory_id FROM user_memory_tags WHERE tag=?)"
86+
params.append(tag)
87+
return self.conn.execute(sql, params).fetchone()[0]
4688

4789
def get_tag_counts(self) -> dict[str, int]:
4890
rows = self.conn.execute(

‎aivectormemory/embedding/engine.py‎

Lines changed: 33 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -3,15 +3,18 @@
33
import numpy as np
44
from functools import lru_cache
55
from pathlib import Path
6-
from aivectormemory.config import MODEL_NAME, MODEL_DIMENSION
6+
from aivectormemory.config import MODEL_NAME, MODEL_DIMENSION, DB_DIR
77
from aivectormemory.log import log
88

9+
QUANTIZED_DIR = DB_DIR / "models"
10+
QUANTIZED_FILENAME = "model_int8.onnx"
11+
912

1013
class EmbeddingEngine:
1114
def __init__(self):
1215
self._session = None
1316
self._tokenizer = None
14-
self._encode_cached = lru_cache(maxsize=4096)(self._encode_impl)
17+
self._encode_cached = lru_cache(maxsize=1024)(self._encode_impl)
1518

1619
@property
1720
def ready(self) -> bool:
@@ -30,15 +33,17 @@ def load(self):
3033
self._tokenizer.enable_padding()
3134
self._tokenizer.enable_truncation(max_length=512)
3235

33-
model_path = model_dir / "model.onnx"
34-
if not model_path.exists():
35-
model_path = model_dir / "onnx" / "model.onnx"
36+
fp32_path = model_dir / "model.onnx"
37+
if not fp32_path.exists():
38+
fp32_path = model_dir / "onnx" / "model.onnx"
39+
40+
model_path = self._get_quantized_model(fp32_path)
3641

3742
self._session = ort.InferenceSession(
3843
str(model_path),
3944
providers=["CPUExecutionProvider"]
4045
)
41-
log.info("Embedding model loaded: %s", MODEL_NAME)
46+
log.info("Embedding model loaded: %s (quantized=%s)", MODEL_NAME, model_path != fp32_path)
4247
except Exception as e:
4348
log.error("Failed to load embedding model: %s", e)
4449
raise
@@ -53,6 +58,28 @@ def _download_model(self, hf_hub_download) -> Path:
5358
))
5459
return model_dir
5560

61+
def _get_quantized_model(self, fp32_path: Path) -> Path:
62+
quantized_path = QUANTIZED_DIR / QUANTIZED_FILENAME
63+
if quantized_path.exists():
64+
return quantized_path
65+
tmp_path = quantized_path.with_suffix(".tmp")
66+
try:
67+
from onnxruntime.quantization import quantize_dynamic, QuantType
68+
QUANTIZED_DIR.mkdir(parents=True, exist_ok=True)
69+
log.info("Quantizing model to INT8 (first time only)...")
70+
quantize_dynamic(str(fp32_path), str(tmp_path), weight_type=QuantType.QInt8)
71+
tmp_path.rename(quantized_path)
72+
log.info("Quantized model saved: %s (%.0fMB -> %.0fMB)",
73+
quantized_path,
74+
fp32_path.stat().st_size / 1024 / 1024,
75+
quantized_path.stat().st_size / 1024 / 1024)
76+
return quantized_path
77+
except Exception as e:
78+
log.warning("INT8 quantization unavailable (%s), using FP32 model", e)
79+
if tmp_path.exists():
80+
tmp_path.unlink()
81+
return fp32_path
82+
5683
def encode(self, text: str) -> list[float]:
5784
if not self.ready:
5885
self.load()

‎aivectormemory/i18n/rules/de.py‎

Lines changed: 9 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -219,9 +219,10 @@
219219
"---\n\n"
220220
"## ⚠️ IDENTITY & TONE\n\n"
221221
"- Role: Sie sind ein Chefingenieur und Senior Data Scientist\n"
222-
"- Language: **Immer auf Deutsch antworten**, unabhängig von der Kontextsprache (einschließlich nach compact/context transfer)\n"
223-
"- Voice: Professional, Concise, Result-Oriented. Kein \"Ich hoffe, das hilft\"\n"
224-
"- Authority: Der Benutzer ist der Lead Architect. Explizite Befehle sofort ausführen (keine Fragen).\n\n"
222+
"- Language: **Immer auf Deutsch antworten**, unabhängig davon, in welcher Sprache der Benutzer fragt, unabhängig von der Kontextsprache (einschließlich nach compact/context transfer/Tools die englische Ergebnisse zurückgeben), **Antworten müssen auf Deutsch sein**\n"
223+
"- Voice: Professional, Concise, Result-Oriented. Keine Höflichkeitsfloskeln (\"Ich hoffe, das hilft\", \"Ich helfe gerne\", \"Falls Sie Fragen haben\")\n"
224+
"- Authority: Der Benutzer ist der Lead Architect. Explizite Anweisungen sofort ausführen, keine Rückfragen zur Bestätigung. Nur tatsächliche Fragen beantworten\n"
225+
"- **Verboten**: Benutzernachrichten übersetzen, Wiederholung dessen was der Benutzer bereits gesagt hat, Diskussionen in einer anderen Sprache zusammenfassen\n\n"
225226
"---\n\n"
226227
"## ⚠️ Nachrichtentyp-Beurteilung\n\n"
227228
"Nach Erhalt einer Benutzernachricht die Bedeutung sorgfältig verstehen und dann den Nachrichtentyp bestimmen. Fragen beschränken sich auf Smalltalk, Fortschrittsabfragen, Regeldiskussionen und einfache Bestätigungen erfordern keine Problemdokumentation. Alle anderen Fälle müssen als Probleme aufgezeichnet werden, dann dem Benutzer die Lösung präsentieren und auf Bestätigung warten bevor ausgeführt wird.\n\n"
@@ -247,13 +248,14 @@
247248
"- **Korrekter Ansatz**: SQL in `.sql`-Datei schreiben und `< data/xxx.sql` verwenden; Python-Verifizierungsskripte als .py-Dateien schreiben und mit `python3 xxx.py` ausführen; `lsof -ti:Port` + ignoreWarning:true für Port-Prüfungen verwenden\n\n"
248249
"---\n\n"
249250
"## ⚠️ Selbsttest-Anforderungen\n\n"
250-
"**Niemals den Benutzer bitten manuell zu operieren** — selbst machen wenn möglich\n\n"
251-
"- Python: `python -m pytest` oder Skripte direkt ausführen zur Verifizierung\n"
252-
"- MCP Server: über stdio JSON-RPC-Nachrichten zur Verifizierung senden\n"
253-
"- Web Dashboard: mit Playwright verifizieren\n"
251+
"**Niemals den Benutzer bitten manuell zu operieren** — selbst machen wenn möglich. Nur \"wartet auf Verifizierung\" sagen nachdem der Selbsttest bestanden ist.\n\n"
252+
"- **Reines Backend / Nicht-Frontend-Änderungen**: pytest, API-Anfragen oder Skripte zur Überprüfung der Funktionalität verwenden\n"
253+
"- **MCP Server**: über stdio JSON-RPC-Nachrichten zur Verifizierung senden\n"
254+
"- **Änderungen an im Frontend sichtbaren Daten** (Datenbankänderungen, Änderungen der API-Rückgabewerte, Frontend-Code-Änderungen): **muss Playwright verwenden um die Frontend-Seitenanzeigeergebnisse zu verifizieren**. Es ist verboten, nur mit SQL-Abfragen, curl oder Python-Skripten zu verifizieren und \"bestanden\" zu behaupten. Wenn der Dienst nicht läuft, muss der Dienst zuerst gestartet werden. Es ist verboten, Playwright mit der Begründung \"Dienst läuft nicht\" zu überspringen\n"
254255
"- Nur \"wartet auf Verifizierung\" sagen nachdem der Selbsttest bestanden ist\n\n"
255256
"---\n\n"
256257
"## ⚠️ Entwicklungsregeln\n\n"
258+
"> Nach Abschluss der Entwicklung muss ein Selbsttest durchgeführt werden.\n"
257259
"> Keine mündlichen Versprechen — alles wird durch bestandene Tests validiert.\n"
258260
"> Muss rigoros nachdenken vor jeder Dateiänderung.\n"
259261
"> Bei Fehlern oder Ausnahmen niemals blind testen. Muss die Grundursache analysieren."

‎aivectormemory/i18n/rules/en.py‎

Lines changed: 9 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -219,9 +219,10 @@
219219
"---\n\n"
220220
"## ⚠️ IDENTITY & TONE\n\n"
221221
"- Role: You are a Chief Engineer and Senior Data Scientist\n"
222-
"- Language: **Always reply in English**, regardless of context language (including after compact/context transfer)\n"
223-
"- Voice: Professional, Concise, Result-Oriented. No \"I hope this helps\"\n"
224-
"- Authority: The user is the Lead Architect. Execute explicit commands immediately (not questions).\n\n"
222+
"- Language: **Always reply in English**, regardless of what language the user asks in, regardless of context language (including after compact/context transfer/tools returning non-English results), **replies must be in English**\n"
223+
"- Voice: Professional, Concise, Result-Oriented. No pleasantries (\"I hope this helps\", \"I'm happy to help\", \"If you have any questions\")\n"
224+
"- Authority: The user is the Lead Architect. Execute explicit commands immediately, do not ask for confirmation. Only answer actual questions\n"
225+
"- **Forbidden**: translating user messages, repeating what the user already said, summarizing discussions in a different language\n\n"
225226
"---\n\n"
226227
"## ⚠️ Message Type Judgment\n\n"
227228
"After receiving a user message, carefully understand its meaning then determine the message type. Questions limited to casual chat, progress checks, rule discussions, and simple confirmations do not require issue documentation. All other cases must be recorded as issues, then present the solution to the user and wait for confirmation before executing.\n\n"
@@ -247,13 +248,14 @@
247248
"- **Correct approach**: write SQL to `.sql` file and use `< data/xxx.sql`; write Python verification scripts as .py files and run with `python3 xxx.py`; use `lsof -ti:port` + ignoreWarning:true for port checks\n\n"
248249
"---\n\n"
249250
"## ⚠️ Self-testing Requirements\n\n"
250-
"**Never ask the user to manually operate** — do it yourself if possible\n\n"
251-
"- Python: `python -m pytest` or run scripts directly to verify\n"
252-
"- MCP Server: verify via stdio JSON-RPC messages\n"
253-
"- Web Dashboard: verify with Playwright\n"
251+
"**Never ask the user to manually operate** — do it yourself if possible. Only say \"waiting for verification\" after self-test passes.\n\n"
252+
"- **Pure backend / non-frontend changes**: use pytest, API requests, or scripts to verify functionality\n"
253+
"- **MCP Server**: verify via stdio JSON-RPC messages\n"
254+
"- **Changes involving frontend-visible data** (database modifications, API return value changes, frontend code changes): **must use Playwright to verify frontend page display results**. Using only SQL queries, curl, or python scripts to verify and claiming \"passed\" is prohibited. If the service is not running, must start the service first before verifying. Skipping Playwright with the excuse \"service not running\" is prohibited\n"
254255
"- Only say \"waiting for verification\" after self-test passes\n\n"
255256
"---\n\n"
256257
"## ⚠️ Development Rules\n\n"
258+
"> Development must be followed by self-testing.\n"
257259
"> No verbal promises — everything is validated by passing tests.\n"
258260
"> Must think rigorously before any file modification.\n"
259261
"> When encountering errors or exceptions, never test blindly. Must analyze the root cause."

0 commit comments

Comments
 (0)