1- """Level 5 enricher: local-LLM narrative summary (Ollama over local HTTP).
2-
3- The summary is grounded in compact facts — counts, top terms, timeline,
4- and open-thread titles — never raw transcripts unless ``--include-text``
5- is set. Tokens stream to the progress sink as they arrive so the CLI can
6- render the summary live. LiteRT-LM and llama.cpp remain fetch-only in
7- this MVP; requesting them raises a clear configuration error.
1+ """Level 5 enricher: local-LLM narrative summary.
2+
3+ Two runtimes are wired: Ollama over local HTTP, and LiteRT-LM in-process
4+ (e.g. a Gemma ``.litertlm`` artifact loaded from the model cache). The
5+ summary is grounded in compact facts — counts, top terms, timeline, and
6+ open-thread titles — never raw transcripts unless ``--include-text`` is
7+ set. Tokens stream to the progress sink as they arrive so the CLI can
8+ render the summary live. llama.cpp remains fetch-only.
89"""
910
1011from __future__ import annotations
1112
13+ import contextlib
1214import json
1315import os
1416import typing as t
2123 from agentgrep .insights .model import InsightsReport
2224
2325_DEFAULT_ENDPOINT = "http://127.0.0.1:11434"
24- _DEFAULT_MODEL = "llama3.2"
26+ _DEFAULT_OLLAMA_MODEL = "llama3.2"
27+ _DEFAULT_LITERT_MODEL = "gemma-4-e2b"
28+ _LITERT_MAX_TOKENS = 2048
2529_MAX_FACT_TERMS = 12
2630_MAX_FACT_THREADS = 8
2731
@@ -58,48 +62,63 @@ def _build_prompt(report: InsightsReport, *, include_text: bool) -> str:
5862
5963
6064def build_llm (ctx : EnricherContext ) -> InsightsEnrichment :
61- """Stream a grounded summary from a local Ollama model."""
62- backend = ctx .request .llm_backend
63- if backend not in ("ollama" , "auto" ):
64- message = f"local LLM backend { backend !r} is fetch-only in this build; use --backend ollama"
65- raise BackendConfigurationError (message , level = "llm" )
66-
65+ """Stream a grounded summary from the selected local-LLM runtime."""
66+ prompt = _build_prompt (ctx .report , include_text = ctx .request .include_text )
67+ if ctx .backend == "litert-lm" :
68+ return _run_litert (ctx , prompt )
69+ if ctx .backend == "ollama" :
70+ return _run_ollama (ctx , prompt )
71+ message = f"local LLM backend { ctx .backend !r} is fetch-only in this build"
72+ raise BackendConfigurationError (message , level = "llm" )
73+
74+
75+ def _emit_delta (ctx : EnricherContext , backend : str , model : str , accumulated : str , text : str ) -> str :
76+ """Emit a streamed delta to the progress sink; return new accumulated text.
77+
78+ Handles runtimes that yield either cumulative or incremental chunks by
79+ treating a chunk that extends the accumulated text as cumulative.
80+ """
81+ if not text :
82+ return accumulated
83+ if text .startswith (accumulated ):
84+ delta = text [len (accumulated ) :]
85+ new_accumulated = text
86+ else :
87+ delta = text
88+ new_accumulated = accumulated + text
89+ if delta and ctx .progress is not None :
90+ ctx .progress .llm_chunk (
91+ backend = backend ,
92+ model = model ,
93+ delta = delta ,
94+ char_count = len (new_accumulated ),
95+ )
96+ return new_accumulated
97+
98+
99+ def _run_ollama (ctx : EnricherContext , prompt : str ) -> InsightsEnrichment :
100+ """Stream a summary from a local Ollama model over HTTP."""
67101 httpx = ctx .modules ["httpx" ]
68- model = ctx .request .model or _DEFAULT_MODEL
102+ model = ctx .request .model or _DEFAULT_OLLAMA_MODEL
69103 endpoint = _endpoint ()
70- prompt = _build_prompt (ctx .report , include_text = ctx .request .include_text )
71104 payload = {
72105 "model" : model ,
73106 "messages" : [{"role" : "user" , "content" : prompt }],
74107 "stream" : True ,
75108 }
76-
77109 if ctx .progress is not None :
78110 ctx .progress .phase ("summarize" , detail = f"ollama:{ model } " )
79111
80- summary_parts : list [ str ] = []
112+ accumulated = ""
81113 try :
82- with httpx .stream (
83- "POST" ,
84- f"{ endpoint } /api/chat" ,
85- json = payload ,
86- timeout = 120.0 ,
87- ) as response :
114+ with httpx .stream ("POST" , f"{ endpoint } /api/chat" , json = payload , timeout = 120.0 ) as response :
88115 response .raise_for_status ()
89116 for line in response .iter_lines ():
90117 if not line :
91118 continue
92119 event = json .loads (line )
93120 content = event .get ("message" , {}).get ("content" , "" )
94- if content :
95- summary_parts .append (content )
96- if ctx .progress is not None :
97- ctx .progress .llm_chunk (
98- backend = "ollama" ,
99- model = model ,
100- delta = content ,
101- char_count = sum (len (part ) for part in summary_parts ),
102- )
121+ accumulated = _emit_delta (ctx , "ollama" , model , accumulated , accumulated + content )
103122 if event .get ("done" ):
104123 break
105124 except Exception as exc :
@@ -113,12 +132,93 @@ def build_llm(ctx: EnricherContext) -> InsightsEnrichment:
113132 failed = f"Ollama summary failed: { exc } "
114133 raise BackendRuntimeError (failed , level = "llm" ) from exc
115134
116- summary = "" .join (summary_parts ).strip ()
135+ return _summary_enrichment (
136+ accumulated .strip (), backend = "ollama" , model = model , endpoint = endpoint
137+ )
138+
139+
140+ def _run_litert (ctx : EnricherContext , prompt : str ) -> InsightsEnrichment :
141+ """Stream a summary from an in-process LiteRT-LM model artifact."""
142+ from agentgrep .insights import models as models_mod
143+
144+ litert_lm = ctx .modules ["litert_lm" ]
145+ model_id = ctx .request .model or _DEFAULT_LITERT_MODEL
146+ spec = models_mod .resolve_llm_model (model_id , "litert-lm" )
147+ if spec is None or spec .artifact_filename is None :
148+ message = f"no curated LiteRT-LM model { model_id !r} "
149+ raise BackendConfigurationError (message , level = "llm" )
150+
151+ if not models_mod .is_installed (spec , ctx .model_cache ):
152+ if not ctx .policy .allow_download :
153+ message = f"LiteRT-LM model { spec .model_id !r} is not provisioned"
154+ install = (
155+ f"agentgrep insights models install { spec .model_id } "
156+ f"--level llm --backend litert-lm --yes"
157+ )
158+ raise BackendConfigurationError (message , level = "llm" , setup_command = install )
159+ models_mod .install_model (
160+ spec ,
161+ model_cache = ctx .model_cache ,
162+ progress = ctx .progress ,
163+ import_module = ctx .import_module ,
164+ )
165+
166+ model_path = models_mod .model_cache_path (spec , ctx .model_cache ) / spec .artifact_filename
167+ if ctx .progress is not None :
168+ ctx .progress .phase ("summarize" , detail = f"litert-lm:{ spec .model_id } " )
169+
170+ # The LiteRT-LM C++ runtime logs model metadata to stderr at INFO; quiet it
171+ # so the streamed summary is the only thing the user sees.
172+ with contextlib .suppress (Exception ):
173+ litert_lm .set_min_log_severity (litert_lm .LogSeverity .ERROR )
174+
175+ accumulated = ""
176+ try :
177+ engine = litert_lm .Engine (
178+ str (model_path ),
179+ backend = litert_lm .Backend .CPU ,
180+ max_num_tokens = _LITERT_MAX_TOKENS ,
181+ )
182+ try :
183+ conversation = engine .create_conversation ()
184+ for chunk in conversation .send_message_async (prompt ):
185+ accumulated = _emit_delta (
186+ ctx , "litert-lm" , spec .model_id , accumulated , _litert_chunk_text (chunk )
187+ )
188+ finally :
189+ engine .close ()
190+ except Exception as exc :
191+ failed = f"LiteRT-LM summary failed: { exc } "
192+ raise BackendRuntimeError (failed , level = "llm" ) from exc
193+
194+ return _summary_enrichment (
195+ accumulated .strip (), backend = "litert-lm" , model = spec .model_id , endpoint = str (model_path )
196+ )
197+
198+
199+ def _litert_chunk_text (chunk : t .Any ) -> str :
200+ """Extract response text from a LiteRT-LM conversation chunk."""
201+ content = chunk .get ("content" ) if isinstance (chunk , dict ) else None
202+ if isinstance (content , str ):
203+ return content
204+ if isinstance (content , list ):
205+ return "" .join (part .get ("text" , "" ) for part in content if isinstance (part , dict ))
206+ return ""
207+
208+
209+ def _summary_enrichment (
210+ summary : str ,
211+ * ,
212+ backend : str ,
213+ model : str ,
214+ endpoint : str ,
215+ ) -> InsightsEnrichment :
216+ """Build the enrichment payload for a generated summary."""
117217 return InsightsEnrichment (
118218 level = "llm" ,
119- backend = "ollama" ,
219+ backend = backend ,
120220 status = "ok" ,
121- message = f"summarized via ollama :{ model } " ,
221+ message = f"summarized via { backend } :{ model } " ,
122222 data = {"summary" : summary , "model" : model , "endpoint" : endpoint },
123- provenance = {"backend" : "ollama" , "model" : model , "endpoint" : endpoint },
223+ provenance = {"backend" : backend , "model" : model , "endpoint" : endpoint },
124224 )
0 commit comments