import urllib.request import urllib.parse import json import re import asyncio from typing import Any, Dict, Optional from .utils import ToolError import config API_URL = "https://en.wikipedia.org/w/api.php" def strip_html(text: str) -> str: """Removes HTML tags from a string using regex to provide clean text to the LLM.""" return re.sub(r'<[^>]*>', '', text) def _make_request(params: Dict[str, Any]) -> Dict[str, Any]: """ Synchronous helper to make the API request using urllib. urllib is used instead of httpx to avoid TLS/HTTP fingerprinting that triggers 403 Forbidden responses from Wikipedia. """ query_string = urllib.parse.urlencode(params) url = f"{API_URL}?{query_string}" headers = { "User-Agent": config.USER_AGENT } req = urllib.request.Request(url, headers=headers) with urllib.request.urlopen(req) as response: return json.loads(response.read().decode('utf-8')) async def _search(title: str, limit: int = 5, fallback: bool = False) -> str: """ Searches Wikipedia for a title. fallback=True: Used when a direct page request fails, providing 'Article not found' header. """ params = { "action": "query", "list": "search", "srsearch": title, "format": "json", "srlimit": limit } # Run synchronous urllib call in a thread to avoid blocking the event loop data = await asyncio.to_thread(_make_request, params) search_results = data.get("query", {}).get("search", []) if not search_results: return f"No Wikipedia results found for '{title}'." if fallback: header = f"Article not found. Here are some similar results for '{title}':" else: header = f"Search results for '{title}':" lines = [header] for i, res in enumerate(search_results, 1): title_res = res.get("title") snippet = strip_html(res.get("snippet", "")) # Brackets around title help the AI identify the exact string for subsequent calls lines.append(f"{i}. [{title_res}] - Snippet: {snippet}") return "\n".join(lines) async def _fetch_content(title: str, section_index: int) -> Optional[str]: """ Retrieves raw Wikitext for a specific section. section_index=0 returns the lead section. """ params = { "action": "query", "prop": "revisions", "rvprop": "content", "rvsection": section_index, "titles": title, "redirects": 1, "format": "json" } data = await asyncio.to_thread(_make_request, params) pages = data.get("query", {}).get("pages", {}) if not pages: return None page_id = next(iter(pages)) page = pages[page_id] if "missing" in page: return None revisions = page.get("revisions", []) if not revisions: return None return revisions[0].get("*") async def _fetch_toc(title: str) -> Optional[str]: """ Retrieves the Table of Contents data and formats it hierarchically. """ params = { "action": "parse", "page": title, "prop": "tocdata", "format": "json", "redirects": 1 } data = await asyncio.to_thread(_make_request, params) parse_data = data.get("parse") if not parse_data: return None toc_data = parse_data.get("tocdata", {}) sections = toc_data.get("sections", []) if not sections: return "No table of contents found for this page." lines = [f"Table of Contents for \"{parse_data.get('title', title)}\":"] for s in sections: level = s.get("tocLevel", 1) index = s.get("index") line = s.get("line") indent = " " * (level - 1) lines.append(f"{indent}[{index}] {line}") return "\n".join(lines) async def handle(args: Dict[str, Any]) -> Dict[str, Any]: """ Main handler for the browse_wikipedia tool. Supports modes: summary, toc, section, and search. """ title = args.get("title") if not title: raise ToolError("Missing required parameter: 'title'") mode = args.get("mode", "summary") section_index = args.get("section_index") search_limit = args.get("search_limit", 5) if mode == "search": result = await _search(title, search_limit, fallback=False) elif mode == "summary": # Lead section + ToC is the default 'summary' to guide the AI's next steps content = await _fetch_content(title, 0) if content is None: result = await _search(title, search_limit, fallback=True) else: toc = await _fetch_toc(title) result = f"{content}\n\n---\n\n{toc}" elif mode == "toc": toc = await _fetch_toc(title) if toc is None: result = await _search(title, search_limit, fallback=True) else: result = toc elif mode == "section": if section_index is None: raise ToolError("Missing required parameter 'section_index' for mode='section'") content = await _fetch_content(title, int(section_index)) if content is None: result = await _search(title, search_limit, fallback=True) else: result = content else: raise ToolError(f"Invalid mode '{mode}'. Supported modes: summary, toc, section, search") return {"text": result}