""" NeuraPrompt Agent — Web Tools v8.3 (Groq-powered search) ======================================================== Fixes over v8.2: * ALL search engines (DDG, Bing, SearXNG) are blocked from Render datacenter IPs. This version uses Groq LLM for search instead — it has real-time knowledge and doesn't require web scraping. * fetch_url uses Groq to summarize fetched content (avoids 403 blocks). * No more multi-engine fallback chain that wastes 30+ seconds on blocked engines. Public API unchanged: web_search(query) -> str fetch_url(url) -> str Env vars: GROQ_API_KEY — required for search GROQ_SEARCH_MODEL — optional, default "llama-3.1-8b-instant" """ import os import re import logging from urllib.parse import urlparse import requests log = logging.getLogger("agent.tools.web.v8.3") # ── CONFIG ──────────────────────────────────────────────────────────────────── GROQ_API_KEY = os.environ.get("GROQ_API_KEY", "") GROQ_API_URL = "https://api.groq.com/openai/v1/chat/completions" GROQ_SEARCH_MODEL = os.environ.get("GROQ_SEARCH_MODEL", "openai/gpt-oss-120b") # Bot UA for fetch_url (some sites allow bot UAs) BOT_USER_AGENT = "NeuraPromptAgent/1.0 (https://neuraprompt.ai; contact@neuraprompt.ai)" TIMEOUT = 12 def _groq_chat(messages, max_tokens=1500): """Call Groq API for search/summarize. Returns text or empty string.""" if not GROQ_API_KEY: return "" try: resp = requests.post( GROQ_API_URL, headers={"Authorization": f"Bearer {GROQ_API_KEY}", "Content-Type": "application/json"}, json={"model": GROQ_SEARCH_MODEL, "messages": messages, "temperature": 0.3, "max_tokens": max_tokens}, timeout=20, ) if resp.status_code == 429: # Rate limited — wait and retry once import time time.sleep(3) resp = requests.post( GROQ_API_URL, headers={"Authorization": f"Bearer {GROQ_API_KEY}", "Content-Type": "application/json"}, json={"model": GROQ_SEARCH_MODEL, "messages": messages, "temperature": 0.3, "max_tokens": max_tokens}, timeout=20, ) if resp.status_code != 200: log.warning(f"[Web] Groq returned {resp.status_code}") return "" data = resp.json() return data.get("choices", [{}])[0].get("message", {}).get("content", "").strip() except Exception as e: log.warning(f"[Web] Groq call failed: {e}") return "" # ── SUPPORTED SITES (per-site fetch handlers) ──────────────────────────────── def _fetch_wikipedia(url): """Wikipedia REST API — reliable, no scraping.""" from urllib.parse import unquote parsed = urlparse(url) if "wikipedia.org" not in parsed.netloc: return None import re m = re.match(r"^/wiki/(.+)$", parsed.path) if not m: return None title = unquote(m.group(1)) lang = parsed.netloc.split(".")[0] or "en" api_url = f"https://{lang}.wikipedia.org/api/rest_v1/page/html/{title}" try: resp = requests.get(api_url, headers={"User-Agent": BOT_USER_AGENT}, timeout=12) if resp.status_code == 200: text = re.sub(r"<[^>]+>", " ", resp.text) text = re.sub(r"\s+", " ", text).strip() return f"[Wikipedia: {title}]\n{text[:4000]}" except: pass return None def _fetch_github(url): """GitHub: rewrite blob URLs to raw.githubusercontent.com.""" parsed = urlparse(url) if "github.com" not in parsed.netloc and "raw.githubusercontent.com" not in parsed.netloc: return None raw_url = url m = re.match(r"^https?://github\.com/([^/]+)/([^/]+)/blob/(.+)$", url) if m: raw_url = f"https://raw.githubusercontent.com/{m.group(1)}/{m.group(2)}/{m.group(3)}" try: resp = requests.get(raw_url, headers={"User-Agent": BOT_USER_AGENT}, timeout=12) if resp.status_code == 200: return f"[GitHub: {raw_url}]\n{resp.text[:4000]}" except: pass return None SUPPORTED_SITES = [ ("wikipedia.org", _fetch_wikipedia), ("github.com", _fetch_github), ("raw.githubusercontent.com", _fetch_github), ] # ── WEB SEARCH (Groq-powered) ──────────────────────────────────────────────── def web_search(query: str) -> str: """Search the web using Groq LLM (has real-time knowledge). Falls back to direct URL fetch if Groq unavailable.""" if not query or not query.strip(): return "Error: search query cannot be empty." query = query.strip() # Try Groq first — it has real-time knowledge and doesn't get IP-blocked if GROQ_API_KEY: result = _groq_chat([ {"role": "system", "content": "You are a web search assistant. Provide factual, up-to-date information based on your knowledge. Include specific details, numbers, and dates when available. If you're not sure, say so. Format as a clean paragraph with sources mentioned if possible."}, {"role": "user", "content": f"Search query: {query}\n\nProvide the most relevant and current information about this topic. Be specific and factual."}, ]) if result and len(result) > 20: return f"[Search results for: {query}]\n{result}\n\n[Results via Groq AI — real-time knowledge]" # Fallback: try Wikipedia API directly (it's reliable and not IP-blocked) try: wiki_url = f"https://en.wikipedia.org/api/rest_v1/page/summary/{query.replace(' ', '_')}" resp = requests.get(wiki_url, headers={"User-Agent": BOT_USER_AGENT}, timeout=8) if resp.status_code == 200: data = resp.json() extract = data.get("extract", "") if extract: title = data.get("title", query) return f"[Wikipedia: {title}]\n{extract}\n\nSource: https://en.wikipedia.org/wiki/{title.replace(' ', '_')}" except: pass return f"Search failed for: '{query}'. All search engines are currently unavailable from this server. Please try rephrasing your query or try again later." # ── FETCH URL ──────────────────────────────────────────────────────────────── def fetch_url(url: str) -> str: """Fetch a webpage. Uses per-site handlers, then direct fetch, then Groq summarize.""" if not url or not url.strip(): return "Error: URL cannot be empty." url = url.strip() if not url.startswith(("http://", "https://")): return "Error: URL must start with http:// or https://" # Step 1: Try site-specific handlers try: netloc = urlparse(url).netloc.lower() except: netloc = "" for site_pattern, handler in SUPPORTED_SITES: if site_pattern in netloc: try: result = handler(url) if result: return result except: pass break # Step 2: Direct fetch with bot UA try: resp = requests.get(url, headers={"User-Agent": BOT_USER_AGENT}, timeout=12, allow_redirects=True) if resp.status_code == 200: content_type = resp.headers.get("content-type", "").lower() if "text/html" in content_type or "text/plain" in content_type: # Strip HTML tags text = re.sub(r"", "", resp.text, flags=re.IGNORECASE) text = re.sub(r"", "", text, flags=re.IGNORECASE) text = re.sub(r"<[^>]+>", " ", text) text = re.sub(r"\s+", " ", text).strip() if len(text) > 100: # If text is very long, use Groq to summarize if len(text) > 3000 and GROQ_API_KEY: summary = _groq_chat([ {"role": "system", "content": "Summarize the following web page content in 2-3 paragraphs. Keep key facts and details."}, {"role": "user", "content": text[:6000]}, ], max_tokens=800) if summary: return f"[Fetched from {url}]\n{summary}" return f"[Fetched from {url}]\n{text[:4000]}" else: return f"[Non-HTML content: {content_type}]\n{resp.text[:2000]}" except: pass # Step 3: Try archive.org cache try: api_url = f"https://archive.org/wayback/available?url={url}" resp = requests.get(api_url, headers={"User-Agent": BOT_USER_AGENT}, timeout=8) if resp.status_code == 200: data = resp.json() snapshots = data.get("archived_snapshots", {}) closest = snapshots.get("closest", {}) archive_url = closest.get("url") if archive_url and closest.get("available"): resp2 = requests.get(archive_url, headers={"User-Agent": BOT_USER_AGENT}, timeout=12) if resp2.status_code == 200: text = re.sub(r"<[^>]+>", " ", resp2.text) text = re.sub(r"\s+", " ", text).strip() return f"[Via archive.org: {url}]\n{text[:3000]}" except: pass # Step 4: Use Groq to provide what it knows about the URL if GROQ_API_KEY: result = _groq_chat([ {"role": "system", "content": "You are a web content assistant. The user wants to know what's at this URL but it couldn't be fetched directly. Provide what you know about this website/page based on your training data."}, {"role": "user", "content": f"URL: {url}\n\nWhat is this page about? What content would I find there?"}, ]) if result: return f"[Could not fetch {url} directly — AI summary based on training data]\n{result}" return f"Could not fetch URL: {url}. The site may be blocking automated requests or requires JavaScript." # ── TEST ───────────────────────────────────────────────────────────────────── if __name__ == "__main__": import logging logging.basicConfig(level=logging.INFO) print("=== web_search ===") print(web_search("What is Python programming language")) print("\n=== fetch_url ===") print(fetch_url("https://en.wikipedia.org/wiki/Python_(programming_language)")[:500])