Coverage for src/local_deep_research/advanced_search_system/knowledge/followup_context_manager.py: 97%
110 statements
« prev ^ index » next coverage.py v7.16.0, created at 2026-10-02 13:53 +0000
« prev ^ index » next coverage.py v7.16.0, created at 2026-10-02 13:53 +0000
1"""
2Follow-up Context Manager
4Manages and processes past research context for follow-up questions.
5This is a standalone class with no abstract base, to avoid implementing
6many unused abstract methods.
7"""
9from typing import Dict, List, Any, Optional
10from loguru import logger
12from langchain_core.language_models.chat_models import BaseChatModel
14from ...utilities.json_utils import get_llm_response_text
15from ...utilities.llm_utils import invoke_llm_sync
18class FollowUpContextHandler:
19 """
20 Manages past research context for follow-up research.
22 This class handles:
23 1. Loading and structuring past research data
24 2. Summarizing findings for follow-up context
25 3. Extracting relevant information for new searches
26 4. Building comprehensive context for strategies
27 """
29 def __init__(
30 self, model: BaseChatModel, settings_snapshot: Optional[Dict] = None
31 ):
32 """
33 Initialize the context manager.
35 Args:
36 model: Language model for processing context
37 settings_snapshot: Optional settings snapshot
38 """
39 self.model = model
40 self.settings_snapshot = settings_snapshot or {}
41 self.past_research_cache = {}
43 def build_context(
44 self, research_data: Dict[str, Any], follow_up_query: str
45 ) -> Dict[str, Any]:
46 """
47 Build comprehensive context from past research.
49 Args:
50 research_data: Past research data including findings, sources, etc.
51 follow_up_query: The follow-up question being asked
53 Returns:
54 Structured context dictionary for follow-up research
55 """
56 logger.info(f"Building context for follow-up: {follow_up_query}")
58 # Extract all components
59 return {
60 "parent_research_id": research_data.get("research_id", ""),
61 "original_query": research_data.get("query", ""),
62 "follow_up_query": follow_up_query,
63 "past_findings": self._extract_findings(research_data),
64 "past_sources": self._extract_sources(research_data),
65 "key_entities": self._extract_entities(research_data),
66 "summary": self._create_summary(research_data, follow_up_query),
67 "report_content": research_data.get("report_content", ""),
68 "formatted_findings": research_data.get("formatted_findings", ""),
69 "all_links_of_system": research_data.get("all_links_of_system", []),
70 "metadata": self._extract_metadata(research_data),
71 }
73 def _extract_findings(self, research_data: Dict) -> str:
74 """
75 Extract and format findings from past research.
77 Args:
78 research_data: Past research data
80 Returns:
81 Formatted findings string
82 """
83 findings_parts = []
85 # Check various possible locations for findings
86 if formatted := research_data.get("formatted_findings"):
87 findings_parts.append(formatted)
89 if report := research_data.get("report_content"):
90 # Take first part of report if no formatted findings
91 if not findings_parts:
92 findings_parts.append(report[:2000])
94 if not findings_parts:
95 # Multi-turn chat supplies its condensed prior findings under the
96 # "past_findings" key (it has no formatted_findings/report_content
97 # of its own). Honor it before declaring nothing available —
98 # otherwise the chat-built summary is silently dropped and the
99 # follow-up prompt sees "No previous findings available".
100 if past_findings := research_data.get("past_findings"):
101 return past_findings
102 return "No previous findings available"
104 return "\n\n".join(findings_parts)
106 def _extract_sources(self, research_data: Dict) -> List[Dict]:
107 """
108 Extract and structure sources from past research.
110 Args:
111 research_data: Past research data
113 Returns:
114 List of source dictionaries
115 """
116 sources = []
117 seen_urls = set()
119 # Check all possible source fields
120 for field in ["resources", "all_links_of_system", "past_links"]:
121 if field_sources := research_data.get(field, []):
122 for source in field_sources:
123 url = source.get("url", "")
124 # Avoid duplicates by URL
125 if url and url not in seen_urls:
126 sources.append(source)
127 seen_urls.add(url)
128 elif not url:
129 # Include sources without URLs (shouldn't happen but be safe)
130 sources.append(source)
132 return sources
134 def _extract_entities(self, research_data: Dict) -> List[str]:
135 """
136 Extract key entities from past research.
138 Args:
139 research_data: Past research data
141 Returns:
142 List of key entities
143 """
144 findings = self._extract_findings(research_data)
146 if not findings or not self.model:
147 return []
149 prompt = f"""
150Extract key entities (names, places, organizations, concepts) from these research findings:
152{findings[:2000]}
154Return up to 10 most important entities, one per line.
155"""
157 try:
158 response = invoke_llm_sync(self.model, prompt)
159 entities = [
160 line.strip()
161 for line in get_llm_response_text(response).split("\n")
162 if line.strip()
163 ]
164 return entities[:10]
165 except Exception:
166 logger.warning("Failed to extract entities")
167 return []
169 def _create_summary(self, research_data: Dict, follow_up_query: str) -> str:
170 """
171 Create a targeted summary of past research relevant to the follow-up question.
172 This is used internally for building context.
174 Args:
175 research_data: Past research data
176 follow_up_query: The follow-up question
178 Returns:
179 Targeted summary for context building
180 """
181 findings = self._extract_findings(research_data)
182 original_query = research_data.get("query", "")
184 # For internal context, create a brief targeted summary
185 return self._generate_summary(
186 findings=findings,
187 query=follow_up_query,
188 original_query=original_query,
189 max_sentences=5,
190 purpose="context",
191 )
193 def _extract_metadata(self, research_data: Dict) -> Dict:
194 """
195 Extract metadata from past research.
197 Args:
198 research_data: Past research data
200 Returns:
201 Metadata dictionary
202 """
203 return {
204 "strategy": research_data.get("strategy", ""),
205 "mode": research_data.get("mode", ""),
206 "created_at": research_data.get("created_at", ""),
207 "research_meta": research_data.get("research_meta", {}),
208 }
210 def summarize_for_followup(
211 self, findings: str, query: str, max_length: int = 1000
212 ) -> str:
213 """
214 Create a concise summary of findings for external use (e.g., in prompts).
215 This creates a length-constrained summary suitable for inclusion in LLM prompts.
217 Args:
218 findings: Past research findings
219 query: Follow-up query
220 max_length: Maximum length of summary in characters
222 Returns:
223 Concise summary constrained to max_length
224 """
225 # Use the shared summary generation with specific parameters for external use
226 return self._generate_summary(
227 findings=findings,
228 query=query,
229 original_query=None,
230 max_sentences=max_length
231 // 100, # Approximate sentences based on length
232 purpose="prompt",
233 max_length=max_length,
234 )
236 def _generate_summary(
237 self,
238 findings: str,
239 query: str,
240 original_query: Optional[str] = None,
241 max_sentences: int = 5,
242 purpose: str = "context",
243 max_length: Optional[int] = None,
244 ) -> str:
245 """
246 Shared summary generation logic.
248 Args:
249 findings: Research findings to summarize
250 query: Follow-up query
251 original_query: Original research query (optional)
252 max_sentences: Maximum number of sentences
253 purpose: Purpose of summary ("context" or "prompt")
254 max_length: Maximum character length (optional)
256 Returns:
257 Generated summary
258 """
259 if not findings:
260 return ""
262 # If findings are already short enough, return as-is
263 if max_length and len(findings) <= max_length:
264 return findings
266 if not self.model:
267 # Fallback without model
268 if max_length:
269 return findings[:max_length] + "..."
270 return findings[:500] + "..."
272 # Build prompt based on purpose
273 if purpose == "context" and original_query:
274 prompt = f"""
275Create a brief summary of previous research findings that are relevant to this follow-up question:
277Original research question: "{original_query}"
278Follow-up question: "{query}"
280Previous findings:
281{findings[:3000]}
283Provide a {max_sentences}-sentence summary focusing on aspects relevant to the follow-up question.
284"""
285 else:
286 prompt = f"""
287Summarize these research findings in relation to the follow-up question:
289Follow-up question: "{query}"
291Findings:
292{findings[:4000]}
294Create a summary of {max_sentences} sentences that captures the most relevant information.
295"""
297 try:
298 response = invoke_llm_sync(self.model, prompt)
299 summary = get_llm_response_text(response)
301 # Apply length constraint if specified
302 if max_length and len(summary) > max_length: 302 ↛ 303line 302 didn't jump to line 303 because the condition on line 302 was never true
303 summary = summary[:max_length] + "..."
305 return summary
306 except Exception:
307 logger.warning("Summary generation failed")
308 # Fallback to truncation
309 if max_length: 309 ↛ 311line 309 didn't jump to line 311 because the condition on line 309 was always true
310 return findings[:max_length] + "..."
311 return findings[:500] + "..."
313 def identify_gaps(
314 self, research_data: Dict, follow_up_query: str
315 ) -> List[str]:
316 """
317 Identify information gaps that the follow-up should address.
319 Args:
320 research_data: Past research data
321 follow_up_query: Follow-up question
323 Returns:
324 List of identified gaps
325 """
326 findings = self._extract_findings(research_data)
328 if not findings or not self.model:
329 return []
331 prompt = f"""
332Based on the previous research and the follow-up question, identify information gaps:
334Previous research findings:
335{findings[:2000]}
337Follow-up question: "{follow_up_query}"
339What specific information is missing or needs clarification? List up to 5 gaps, one per line.
340"""
342 try:
343 response = invoke_llm_sync(self.model, prompt)
344 gaps = [
345 line.strip()
346 for line in get_llm_response_text(response).split("\n")
347 if line.strip()
348 ]
349 return gaps[:5]
350 except Exception:
351 logger.warning("Failed to identify gaps")
352 return []
354 def format_for_settings_snapshot(
355 self, context: Dict[str, Any]
356 ) -> Dict[str, Any]:
357 """
358 Format context for inclusion in settings snapshot.
359 Only includes essential metadata, not actual content.
361 Args:
362 context: Full context dictionary
364 Returns:
365 Minimal metadata for settings snapshot
366 """
367 # Only include minimal metadata in settings snapshot
368 # Settings snapshot should be for settings, not data
369 return {
370 "followup_metadata": {
371 "parent_research_id": context.get("parent_research_id"),
372 "is_followup": True,
373 "has_context": bool(context.get("past_findings")),
374 }
375 }
377 def get_relevant_context_for_llm(
378 self, context: Dict[str, Any], max_tokens: int = 2000
379 ) -> str:
380 """
381 Get a concise version of context for LLM prompts.
383 Args:
384 context: Full context dictionary
385 max_tokens: Approximate maximum tokens
387 Returns:
388 Concise context string
389 """
390 parts = []
392 # Add original and follow-up queries
393 parts.append(f"Original research: {context.get('original_query', '')}")
394 parts.append(
395 f"Follow-up question: {context.get('follow_up_query', '')}"
396 )
398 # Add summary
399 if summary := context.get("summary"):
400 parts.append(f"\nPrevious findings summary:\n{summary}")
402 # Add key entities
403 if entities := context.get("key_entities"):
404 parts.append(f"\nKey entities: {', '.join(entities[:5])}")
406 # Add source count
407 if sources := context.get("past_sources"):
408 parts.append(f"\nAvailable sources: {len(sources)}")
410 result = "\n".join(parts)
412 # Truncate if needed (rough approximation: 4 chars per token)
413 max_chars = max_tokens * 4
414 if len(result) > max_chars:
415 result = result[:max_chars] + "..."
417 return result