MooseCP/tools/wikipedia.py
2026-07-25 02:40:17 -07:00

174 lines
5.5 KiB
Python

import urllib.request
import urllib.parse
import json
import re
import asyncio
from typing import Any, Dict, Optional
from .utils import ToolError
import config
API_URL = "https://en.wikipedia.org/w/api.php"
def strip_html(text: str) -> str:
"""Removes HTML tags from a string using regex to provide clean text to the LLM."""
return re.sub(r'<[^>]*>', '', text)
def _make_request(params: Dict[str, Any]) -> Dict[str, Any]:
"""
Synchronous helper to make the API request using urllib.
urllib is used instead of httpx to avoid TLS/HTTP fingerprinting
that triggers 403 Forbidden responses from Wikipedia.
"""
query_string = urllib.parse.urlencode(params)
url = f"{API_URL}?{query_string}"
headers = {
"User-Agent": config.USER_AGENT
}
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req) as response:
return json.loads(response.read().decode('utf-8'))
async def _search(title: str, limit: int = 5, fallback: bool = False) -> str:
"""
Searches Wikipedia for a title.
fallback=True: Used when a direct page request fails, providing 'Article not found' header.
"""
params = {
"action": "query",
"list": "search",
"srsearch": title,
"format": "json",
"srlimit": limit
}
# Run synchronous urllib call in a thread to avoid blocking the event loop
data = await asyncio.to_thread(_make_request, params)
search_results = data.get("query", {}).get("search", [])
if not search_results:
return f"No Wikipedia results found for '{title}'."
if fallback:
header = f"Article not found. Please select one of the following similar articles to proceed with '{title}':"
else:
header = f"Search results for '{title}':"
lines = [header]
for i, res in enumerate(search_results, 1):
title_res = res.get("title")
snippet = strip_html(res.get("snippet", ""))
# Brackets around title help the AI identify the exact string for subsequent calls
lines.append(f"{i}. [{title_res}] - Snippet: {snippet}")
if fallback:
lines.append("\n[Tip: Use the title in brackets [ ] to explore the selected page.]")
return "\n".join(lines)
async def _fetch_content(title: str, section_index: int) -> Optional[str]:
"""
Retrieves raw Wikitext for a specific section.
section_index=0 returns the lead section.
"""
params = {
"action": "query",
"prop": "revisions",
"rvprop": "content",
"rvsection": section_index,
"titles": title,
"redirects": 1,
"format": "json"
}
data = await asyncio.to_thread(_make_request, params)
pages = data.get("query", {}).get("pages", {})
if not pages:
return None
page_id = next(iter(pages))
page = pages[page_id]
if "missing" in page:
return None
revisions = page.get("revisions", [])
if not revisions:
return None
return revisions[0].get("*")
async def _fetch_toc(title: str) -> Optional[str]:
"""
Retrieves the Table of Contents data and formats it hierarchically.
"""
params = {
"action": "parse",
"page": title,
"prop": "tocdata",
"format": "json",
"redirects": 1
}
data = await asyncio.to_thread(_make_request, params)
parse_data = data.get("parse")
if not parse_data:
return None
toc_data = parse_data.get("tocdata", {})
sections = toc_data.get("sections", [])
if not sections:
return "No table of contents found for this page."
lines = [f"Table of Contents for \"{parse_data.get('title', title)}\":"]
for s in sections:
level = s.get("tocLevel", 1)
index = s.get("index")
line = s.get("line")
indent = " " * (level - 1)
lines.append(f"{indent}[{index}] {line}")
return "\n".join(lines)
async def handle(args: Dict[str, Any]) -> Dict[str, Any]:
"""
Main handler for the browse_wikipedia tool.
Supports modes: summary, toc, section, and search.
"""
title = args.get("title")
if not title:
raise ToolError("Missing required parameter: 'title'")
mode = args.get("mode", "summary")
section_index = args.get("section_index")
search_limit = args.get("search_limit", 5)
if mode == "search":
result = await _search(title, search_limit, fallback=False)
elif mode == "summary":
# Lead section + ToC is the default 'summary' to guide the AI's next steps
content = await _fetch_content(title, 0)
if content is None:
result = await _search(title, search_limit, fallback=True)
else:
toc = await _fetch_toc(title)
result = f"{content}\n\n---\n\n{toc}"
elif mode == "toc":
toc = await _fetch_toc(title)
if toc is None:
result = await _search(title, search_limit, fallback=True)
else:
result = toc
elif mode == "section":
if section_index is None:
raise ToolError("Missing required parameter 'section_index' for mode='section'")
content = await _fetch_content(title, int(section_index))
if content is None:
result = await _search(title, search_limit, fallback=True)
else:
result = content
else:
raise ToolError(f"Invalid mode '{mode}'. Supported modes: summary, toc, section, search")
return {"text": result}