From 2b43592151947bb8a36339fa2e29abf0be9031db Mon Sep 17 00:00:00 2001 From: Vincent Date: Sat, 22 Aug 2026 09:42:25 +0900 Subject: [PATCH] MCP server that fetches multiple URLs in parallel and extracts the content relevant to a query. -Clean and deduplicate URLs -Clean extracted text -Semantic selection to keep only text relevant to the query -Limit number of extracted characters per URL and also limit the total characters extracted to avoid returning too much text Arguments: urls: List of URLs to process. query: User question or information wanted Semantic ranking is performed against this query. Optional Arguments: max_chars_per_url: Maximum characters returned for each URL. Default set to 6000 max_total_chars: Maximum characters returned across all URLs. Default set to 30000 top_k_chunks: Maximum number of relevant chunks per URL. Default set to 6 min_relevance_score: Minimum semantic similarity score. Default set to 0.25 Returns: List of relevant web content results. Below an example of info return for one URL: ======================================== === SOURCE URL: https://www.marketsandmarkets.com/Market-Reports/3d-scanner-market-119952472.html === STATUS: success CONTENT: [Semantic Retrieval] Query: Extract information about companies, products, pricing, key features, target customers, market trends, growth, and competitive information related to 3D scanners. Chunks considered: 109 Chunks selected: 6 Relevance scores: [0.759, 0.757, 0.749, 0.71, 0.7, 0.696] Content characters: 3046 Truncated: False --- RELEVANT CONTENT --- Chunk 1 Chunk 2 Chunk 3 Chunk 4 Chunk 5 Chunk 6 ======================================== --- .gitignore | 12 + mcp_server - simple.py | 51 +++ mcp_server.py | 826 +++++++++++++++++++++++++++++++++++++++++ requirements.txt | 6 + 4 files changed, 895 insertions(+) create mode 100644 .gitignore create mode 100644 mcp_server - simple.py create mode 100644 mcp_server.py create mode 100644 requirements.txt diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..4fe6041 --- /dev/null +++ b/.gitignore @@ -0,0 +1,12 @@ +# Ignore all __pycache__ directories +**/__pycache__/ + +# Ignore compiled Python files +*.py[cod] + +venv/ +log/ +output/ + + +bin/__pycache__/ \ No newline at end of file diff --git a/mcp_server - simple.py b/mcp_server - simple.py new file mode 100644 index 0000000..9fb75f8 --- /dev/null +++ b/mcp_server - simple.py @@ -0,0 +1,51 @@ +# mcp_server.py +import requests +import json +from bs4 import BeautifulSoup +from fastmcp import FastMCP +from concurrent.futures import ThreadPoolExecutor + +mcp = FastMCP("BatchWebContentExtraction") + +def scrape_single_url(url: str) -> dict: + """Helper function to scrape and clean a single URL.""" + try: + headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"} + response = requests.get(url, headers=headers, timeout=10) + response.raise_for_status() + + soup = BeautifulSoup(response.text, 'html.parser') + for script in soup(["script", "style", "nav", "footer", "header", "aside"]): + script.decompose() + + raw_text = soup.get_text(separator=' ') + lines = (line.strip() for line in raw_text.splitlines()) + chunks = (phrase.strip() for line in lines for phrase in line.split(" ")) + clean_text = '\n'.join(chunk for chunk in chunks if chunk) + + return {"url": url, "status": "success", "content": clean_text[:5000]} + + except Exception as e: + return {"url": url, "status": "failed", "content": str(e)} + +@mcp.tool() +def batch_web_content_extraction(urls: list[str]) -> str: + """ + Fetches text content from a list of multiple URLs simultaneously in parallel. + Returns a unified JSON string of results. + """ + if not urls: + return json.dumps({"error": "No URLs provided"}) + + results = [] + with ThreadPoolExecutor(max_workers=min(len(urls), 5)) as executor: + futures_map = {executor.submit(scrape_single_url, url): url for url in urls} + for future in futures_map: + results.append(future.result()) + + return json.dumps(results) + +if __name__ == "__main__": + # Use standard streamable-http protocol on port 8000 + mcp.run(transport="streamable-http", host="0.0.0.0", port=8000) + diff --git a/mcp_server.py b/mcp_server.py new file mode 100644 index 0000000..308f3c4 --- /dev/null +++ b/mcp_server.py @@ -0,0 +1,826 @@ +# mcp_server.py +import json + +import requests +#from bs4 import BeautifulSoup +from fastmcp import FastMCP +from concurrent.futures import ThreadPoolExecutor, as_completed +import trafilatura +import requests +from sentence_transformers import SentenceTransformer +import numpy as np +import torch +from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode + + +mcp = FastMCP("BatchWebContentExtraction") + + +import re +from urllib.parse import ( + urlparse, + urlunparse, + parse_qsl, + urlencode +) + +TRACKING_PARAMS = { + "utm_source", + "utm_medium", + "utm_campaign", + "utm_term", + "utm_content", + "utm_id", + "gclid", + "fbclid", + "msclkid", + "dclid", + "srsltid", + "mc_cid", + "mc_eid", +} + + +def clean_url(url: str) -> str: + """Normalize URLs and remove Markdown/tracking wrappers.""" + + if not url: + return "" + + url = url.strip() + + # -------------------------------------------------- + # 1. Extract URL from Markdown link + # + # [https://example.com/page](https://example.com/page) + # ↓ + # https://example.com/page + # -------------------------------------------------- + + markdown_match = re.match( + r"^\[.*?\]\((https?://[^)]+)\)$", + url + ) + + if markdown_match: + url = markdown_match.group(1) + + # -------------------------------------------------- + # 2. Remove accidental surrounding quotes + # -------------------------------------------------- + + url = url.strip("\"'") + + # -------------------------------------------------- + # 3. Parse URL + # -------------------------------------------------- + + parsed = urlparse(url) + + if parsed.scheme not in ("http", "https"): + return "" + + # -------------------------------------------------- + # 4. Remove tracking parameters + # -------------------------------------------------- + + query_params = [ + (key, value) + for key, value in parse_qsl( + parsed.query, + keep_blank_values=True + ) + if key.lower() not in TRACKING_PARAMS + ] + + # -------------------------------------------------- + # 5. Rebuild clean URL + # -------------------------------------------------- + + cleaned = urlunparse(( + parsed.scheme.lower(), + parsed.netloc.lower(), + parsed.path.rstrip("/") or "/", + parsed.params, + urlencode(query_params), + "" + )) + + return cleaned + + +# ============================================================ +# Configuration +# ============================================================ + +MAX_WORKERS = 5 + +# Maximum characters considered from each URL +MAX_SOURCE_CHARS = 50000 + +# Maximum characters returned from each URL +MAX_CHARS_PER_URL = 6000 + +# Maximum characters returned by the whole tool +MAX_TOTAL_CHARS = 30000 + +# Number of relevant chunks to keep from each URL +TOP_K_CHUNKS = 6 + +# Minimum semantic similarity score +MIN_RELEVANCE_SCORE = 0.25 + +# Chunk size +CHUNK_SIZE = 600 + +# Slight overlap between chunks +CHUNK_OVERLAP = 100 + + +# ============================================================ +# Load embedding model ONCE +# ============================================================ + +device = "cuda" if torch.cuda.is_available() else "cpu" + +print(f"Embedding device: {device}") + +if device == "cuda": + print(f"GPU: {torch.cuda.get_device_name(0)}") + +embedding_model = SentenceTransformer( + "all-MiniLM-L6-v2", + device=device +) + +print("Embedding model loaded.") + +# ============================================================ +# Text cleaning +# ============================================================ + +def clean_extracted_text(text: str) -> str: + """ + Performs lightweight cleanup after Trafilatura extraction. + """ + + if not text: + return "" + + text = text.replace("\r\n", "\n") + text = text.replace("\r", "\n") + + lines = [] + + for line in text.split("\n"): + + line = " ".join(line.split()) + + if line: + lines.append(line) + + # Remove duplicate consecutive lines + cleaned = [] + + previous = None + + for line in lines: + + if line != previous: + cleaned.append(line) + + previous = line + + return "\n".join(cleaned) + + +# ============================================================ +# Limit text +# ============================================================ + +def limit_text(text: str, max_chars: int) -> tuple[str, bool]: + """ + Limits text without cutting a paragraph whenever possible. + """ + + if len(text) <= max_chars: + return text, False + + truncated = text[:max_chars] + + last_newline = truncated.rfind("\n") + + if last_newline > max_chars * 0.7: + truncated = truncated[:last_newline] + + return ( + truncated.rstrip() + "\n[Content truncated]", + True + ) + + +# ============================================================ +# Fetch and extract URL +# ============================================================ + +def scrape_single_url(url: str) -> dict: + """ + Fetches a URL and extracts the main textual content. + """ + + url = clean_url(url) + + try: + + headers = { + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 " + "(KHTML, like Gecko) " + "Chrome/131.0 Safari/537.36" + ) + } + + response = requests.get( + url, + headers=headers, + timeout=15 + ) + + response.raise_for_status() + + # ---------------------------------------------------- + # Extract main content + # ---------------------------------------------------- + + text = trafilatura.extract( + response.text, + include_links=False, + include_images=False, + include_tables=True, + favor_precision=True + ) + + if not text: + + return { + "url": url, + "status": "failed", + "content": "No meaningful content could be extracted." + } + + # ---------------------------------------------------- + # Clean + # ---------------------------------------------------- + + clean_text = clean_extracted_text(text) + + original_chars = len(clean_text) + + # Don't allow enormous documents into the embedding stage + clean_text, source_truncated = limit_text( + clean_text, + MAX_SOURCE_CHARS + ) + + return { + "url": url, + "status": "success", + "content": clean_text, + "original_chars": original_chars, + "source_truncated": source_truncated + } + + except requests.exceptions.Timeout: + + return { + "url": url, + "status": "failed", + "content": "Request timed out." + } + + except requests.exceptions.HTTPError as e: + + return { + "url": url, + "status": "failed", + "content": f"HTTP error: {e}" + } + + except Exception as e: + + return { + "url": url, + "status": "failed", + "content": str(e) + } + + +# ============================================================ +# Chunking +# ============================================================ + +def create_chunks( + text: str, + chunk_size: int = CHUNK_SIZE, + overlap: int = CHUNK_OVERLAP +) -> list[str]: + """ + Splits text into overlapping chunks. + + Tries to respect paragraph boundaries. + """ + + paragraphs = [ + p.strip() + for p in text.split("\n") + if p.strip() + ] + + chunks = [] + current = "" + + for paragraph in paragraphs: + + # If paragraph fits into current chunk + if len(current) + len(paragraph) + 1 <= chunk_size: + + if current: + current += "\n" + + current += paragraph + + else: + + if current: + chunks.append(current) + + # Handle paragraphs larger than chunk size + if len(paragraph) > chunk_size: + + start = 0 + + while start < len(paragraph): + + end = start + chunk_size + + piece = paragraph[start:end] + + chunks.append(piece) + + start = end - overlap + + current = "" + + else: + + current = paragraph + + if current: + chunks.append(current) + + return chunks + + +# ============================================================ +# Semantic relevance +# ============================================================ + +def rank_chunks( + query: str, + chunks: list[str] +) -> list[tuple[str, float]]: + """ + Calculates semantic similarity between the query + and every chunk. + + Returns chunks sorted by relevance. + """ + + if not chunks: + return [] + + # Encode query + query_embedding = embedding_model.encode( + query, + normalize_embeddings=True, + device=device + ) + + # Encode chunks + chunk_embeddings = embedding_model.encode( + chunks, + normalize_embeddings=True, + batch_size=128, + show_progress_bar=False, + device=device + ) + + # Cosine similarity + scores = np.dot( + chunk_embeddings, + query_embedding + ) + + ranked = sorted( + zip(chunks, scores), + key=lambda x: x[1], + reverse=True + ) + + return ranked + + +# ============================================================ +# Select relevant content +# ============================================================ + +def select_relevant_content( + text: str, + query: str, + top_k: int = TOP_K_CHUNKS, + min_score: float = MIN_RELEVANCE_SCORE, + max_chars: int = MAX_CHARS_PER_URL +) -> dict: + + # -------------------------------------------------------- + # Create chunks + # -------------------------------------------------------- + + chunks = create_chunks(text) + + if not chunks: + return { + "content": "", + "chunks_considered": 0, + "chunks_selected": 0, + "scores": [], + "truncated": False + } + + # -------------------------------------------------------- + # Rank chunks by semantic similarity + # -------------------------------------------------------- + + ranked = rank_chunks( + query, + chunks + ) + + # -------------------------------------------------------- + # Select highest-scoring chunks + # -------------------------------------------------------- + + selected = [ + (chunk, float(score)) + for chunk, score in ranked + if score >= min_score + ][:top_k] + + # -------------------------------------------------------- + # Fallback: + # If nothing passes the threshold, keep the best chunk + # -------------------------------------------------------- + + if not selected and ranked: + + selected = [ + ( + ranked[0][0], + float(ranked[0][1]) + ) + ] + + # -------------------------------------------------------- + # IMPORTANT: + # Keep relevance order. + # + # The first chunk is the most relevant. + # -------------------------------------------------------- + + selected_chunks = [ + chunk + for chunk, score in selected + ] + + scores = [ + round(score, 3) + for chunk, score in selected + ] + + # -------------------------------------------------------- + # Build content + # -------------------------------------------------------- + + final_content = "\n\n".join( + selected_chunks + ) + + # -------------------------------------------------------- + # Apply character limit + # -------------------------------------------------------- + + final_content, truncated = limit_text( + final_content, + max_chars + ) + + # -------------------------------------------------------- + # Debug information + # -------------------------------------------------------- + + debug_header = ( + f"[Semantic Retrieval]\n" + f"Query: {query}\n" + f"Chunks considered: {len(chunks)}\n" + f"Chunks selected: {len(selected)}\n" + f"Relevance scores: {scores}\n" + f"Content characters: {len(final_content)}\n" + f"Truncated: {truncated}\n" + f"\n--- RELEVANT CONTENT ---\n" + ) + + return { + "content": debug_header + final_content, + "chunks_considered": len(chunks), + "chunks_selected": len(selected), + "scores": scores, + "truncated": truncated + } + +# ============================================================ +# MCP Tool +# ============================================================ + +@mcp.tool() +def batch_web_content_extraction( + urls: list[str], + query: str = "", + max_chars_per_url: int = MAX_CHARS_PER_URL, + max_total_chars: int = MAX_TOTAL_CHARS, + top_k_chunks: int = TOP_K_CHUNKS, + min_relevance_score: float = MIN_RELEVANCE_SCORE +) -> list[dict]: + """ + Fetches multiple URLs in parallel and extracts the + content relevant to a query. + + Args: + + urls: + URLs to process. + + query: + User question or information need. + Semantic ranking is performed against this query. + + max_chars_per_url: + Maximum characters returned for each URL. + + max_total_chars: + Maximum characters returned across all URLs. + + top_k_chunks: + Maximum number of relevant chunks per URL. + + min_relevance_score: + Minimum semantic similarity score. + + Returns: + List of relevant web content results. + """ + + if not urls: + + return [ + { + "status": "error", + "content": "No URLs provided." + } + ] + + # -------------------------------------------------------- + # Validate limits + # -------------------------------------------------------- + + max_chars_per_url = max( + 1000, + min(max_chars_per_url, 20000) + ) + + max_total_chars = max( + 5000, + min(max_total_chars, 100000) + ) + + top_k_chunks = max( + 1, + min(top_k_chunks, 20) + ) + + min_relevance_score = max( + 0.0, + min(min_relevance_score, 1.0) + ) + + # -------------------------------------------------------- + # Fetch URLs in parallel + # -------------------------------------------------------- + + # Clean and deduplicate URLs + cleaned_urls = [] + + for url in urls: + cleaned = clean_url(url) + + if cleaned and cleaned not in cleaned_urls: + cleaned_urls.append(cleaned) + + + scraped_results = [] + + with ThreadPoolExecutor( + max_workers=min(len(cleaned_urls), MAX_WORKERS) + ) as executor: + + futures = { + executor.submit( + scrape_single_url, + url + ): url + for url in cleaned_urls + } + + for future in as_completed(futures): + + try: + + result = future.result() + + except Exception as e: + + result = { + "url": futures[future], + "status": "failed", + "content": str(e) + } + + scraped_results.append(result) + + # -------------------------------------------------------- + # If no query supplied, behave like Phase 1 + # -------------------------------------------------------- + + if not query.strip(): + + final_results = [] + total_chars = 0 + + for result in scraped_results: + + if result.get("status") != "success": + final_results.append(result) + continue + + content = result.get("content", "") + + remaining = max_total_chars - total_chars + + if remaining <= 0: + + content = ( + "[Content omitted due to total size limit]" + ) + + else: + + content, truncated = limit_text( + content, + min(max_chars_per_url, remaining) + ) + + result["truncated"] = ( + result.get("source_truncated", False) + or truncated + ) + + total_chars += len(content) + + result["content"] = content + result["returned_chars"] = len(content) + + final_results.append(result) + + return final_results + + # -------------------------------------------------------- + # Semantic selection + # -------------------------------------------------------- + + final_results = [] + + for result in scraped_results: + + if result.get("status") != "success": + + final_results.append(result) + continue + + selection = select_relevant_content( + text=result["content"], + query=query, + top_k=top_k_chunks, + min_score=min_relevance_score, + max_chars=max_chars_per_url + ) + + result["content"] = selection["content"] + + result["chunks_considered"] = ( + selection["chunks_considered"] + ) + + result["chunks_selected"] = ( + selection["chunks_selected"] + ) + + result["relevance_scores"] = ( + selection["scores"] + ) + + result["truncated"] = ( + selection["truncated"] + ) + + result["returned_chars"] = len( + result["content"] + ) + + final_results.append(result) + + # -------------------------------------------------------- + # Apply global character budget + # -------------------------------------------------------- + + total_chars = 0 + + for result in final_results: + + if result.get("status") != "success": + continue + + content = result.get("content", "") + + remaining = max_total_chars - total_chars + + if remaining <= 0: + + result["content"] = ( + "[Content omitted due to total size limit]" + ) + + result["returned_chars"] = 0 + result["truncated"] = True + + elif len(content) > remaining: + + content, _ = limit_text( + content, + remaining + ) + + result["content"] = content + result["returned_chars"] = len(content) + result["truncated"] = True + + total_chars += len(content) + + else: + + total_chars += len(content) + + + + serialized = json.dumps( + final_results, + ensure_ascii=False + ) + + print("=" * 60) + print(f"MCP FINAL RESULTS:") + print(f"Number of URLs: {len(final_results)}") + print(f"Serialized characters: {len(serialized):,}") + print( + f"Content characters: " + f"{sum(len(r.get('content', '')) for r in final_results):,}" + ) + print("=" * 60) + + + return final_results + + +if __name__ == "__main__": + # Use standard streamable-http protocol on port 8000 + mcp.run(transport="streamable-http", host="0.0.0.0", port=8000) + diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..f59069b --- /dev/null +++ b/requirements.txt @@ -0,0 +1,6 @@ +requests +#beautifulsoup4 +trafilatura +sentence-transformers +fastmcp +torch