"""Utility helpers for the research engine.""" import re def strip_thinking(text: str) -> str: """Remove ... or ... blocks from model output.""" if not text: return text text = re.sub(r"[\s\S]*?", "", text, flags=re.I) return text.strip() def is_low_quality(text: str) -> bool: """Return True if extracted content looks useless (too short, error page, etc.).""" if not text or len(text.strip()) < 50: return True low_q_patterns = [ "access denied", "403 forbidden", "404 not found", "page not found", "enable javascript", "please enable cookies", "captcha", "robot check", "cloudflare", "just a moment", "no relevant information", "not relevant to", ] lower = text.lower() return any(p in lower for p in low_q_patterns) EXTRACTOR_PROMPT = """\ You are extracting relevant information from a webpage to help answer a research question. **Research goal:** {goal} **Webpage content:** {webpage_content} Extract the most relevant information. Return a JSON object: {{ "rational": "One sentence on why this page is/isn't relevant", "evidence": "The specific facts, data, quotes, and info from this page relevant to the goal (up to 800 words)", "summary": "2-3 sentence summary of the key findings from this page" }} If the page has no relevant information, set evidence to "" and summary to "Not relevant." Return ONLY valid JSON. """