Files
vincent 2b43592151 MCP server that fetches multiple URLs in parallel and extracts the content relevant to a query.
-Clean and deduplicate URLs
-Clean extracted text
-Semantic selection to keep only text relevant to the query
-Limit number of extracted characters per URL and also limit the total characters extracted to avoid returning too much text

Arguments:
urls:
List of URLs to process.

query:
User question or information wanted
Semantic ranking is performed against this query.

Optional Arguments:
max_chars_per_url:
Maximum characters returned for each URL.
Default set to 6000

max_total_chars:
Maximum characters returned across all URLs.
Default set to 30000

top_k_chunks:
Maximum number of relevant chunks per URL.
Default set to 6

min_relevance_score:
Minimum semantic similarity score.
Default set to 0.25

Returns:
List of relevant web content results.
Below an example of info return for one URL:

========================================

=== SOURCE URL: https://www.marketsandmarkets.com/Market-Reports/3d-scanner-market-119952472.html ===
STATUS: success
CONTENT:
[Semantic Retrieval]
Query: Extract information about companies, products, pricing, key features, target customers, market trends, growth, and competitive information related to 3D scanners.
Chunks considered: 109
Chunks selected: 6
Relevance scores: [0.759, 0.757, 0.749, 0.71, 0.7, 0.696]
Content characters: 3046
Truncated: False

--- RELEVANT CONTENT ---
Chunk 1

Chunk 2

Chunk 3

Chunk 4

Chunk 5

Chunk 6

========================================
2026-08-22 09:42:25 +09:00

827 lines
19 KiB
Python

# mcp_server.py
import json
import requests
#from bs4 import BeautifulSoup
from fastmcp import FastMCP
from concurrent.futures import ThreadPoolExecutor, as_completed
import trafilatura
import requests
from sentence_transformers import SentenceTransformer
import numpy as np
import torch
from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode
mcp = FastMCP("BatchWebContentExtraction")
import re
from urllib.parse import (
urlparse,
urlunparse,
parse_qsl,
urlencode
)
TRACKING_PARAMS = {
"utm_source",
"utm_medium",
"utm_campaign",
"utm_term",
"utm_content",
"utm_id",
"gclid",
"fbclid",
"msclkid",
"dclid",
"srsltid",
"mc_cid",
"mc_eid",
}
def clean_url(url: str) -> str:
"""Normalize URLs and remove Markdown/tracking wrappers."""
if not url:
return ""
url = url.strip()
# --------------------------------------------------
# 1. Extract URL from Markdown link
#
# [https://example.com/page](https://example.com/page)
# ↓
# https://example.com/page
# --------------------------------------------------
markdown_match = re.match(
r"^\[.*?\]\((https?://[^)]+)\)$",
url
)
if markdown_match:
url = markdown_match.group(1)
# --------------------------------------------------
# 2. Remove accidental surrounding quotes
# --------------------------------------------------
url = url.strip("\"'")
# --------------------------------------------------
# 3. Parse URL
# --------------------------------------------------
parsed = urlparse(url)
if parsed.scheme not in ("http", "https"):
return ""
# --------------------------------------------------
# 4. Remove tracking parameters
# --------------------------------------------------
query_params = [
(key, value)
for key, value in parse_qsl(
parsed.query,
keep_blank_values=True
)
if key.lower() not in TRACKING_PARAMS
]
# --------------------------------------------------
# 5. Rebuild clean URL
# --------------------------------------------------
cleaned = urlunparse((
parsed.scheme.lower(),
parsed.netloc.lower(),
parsed.path.rstrip("/") or "/",
parsed.params,
urlencode(query_params),
""
))
return cleaned
# ============================================================
# Configuration
# ============================================================
MAX_WORKERS = 5
# Maximum characters considered from each URL
MAX_SOURCE_CHARS = 50000
# Maximum characters returned from each URL
MAX_CHARS_PER_URL = 6000
# Maximum characters returned by the whole tool
MAX_TOTAL_CHARS = 30000
# Number of relevant chunks to keep from each URL
TOP_K_CHUNKS = 6
# Minimum semantic similarity score
MIN_RELEVANCE_SCORE = 0.25
# Chunk size
CHUNK_SIZE = 600
# Slight overlap between chunks
CHUNK_OVERLAP = 100
# ============================================================
# Load embedding model ONCE
# ============================================================
device = "cuda" if torch.cuda.is_available() else "cpu"
print(f"Embedding device: {device}")
if device == "cuda":
print(f"GPU: {torch.cuda.get_device_name(0)}")
embedding_model = SentenceTransformer(
"all-MiniLM-L6-v2",
device=device
)
print("Embedding model loaded.")
# ============================================================
# Text cleaning
# ============================================================
def clean_extracted_text(text: str) -> str:
"""
Performs lightweight cleanup after Trafilatura extraction.
"""
if not text:
return ""
text = text.replace("\r\n", "\n")
text = text.replace("\r", "\n")
lines = []
for line in text.split("\n"):
line = " ".join(line.split())
if line:
lines.append(line)
# Remove duplicate consecutive lines
cleaned = []
previous = None
for line in lines:
if line != previous:
cleaned.append(line)
previous = line
return "\n".join(cleaned)
# ============================================================
# Limit text
# ============================================================
def limit_text(text: str, max_chars: int) -> tuple[str, bool]:
"""
Limits text without cutting a paragraph whenever possible.
"""
if len(text) <= max_chars:
return text, False
truncated = text[:max_chars]
last_newline = truncated.rfind("\n")
if last_newline > max_chars * 0.7:
truncated = truncated[:last_newline]
return (
truncated.rstrip() + "\n[Content truncated]",
True
)
# ============================================================
# Fetch and extract URL
# ============================================================
def scrape_single_url(url: str) -> dict:
"""
Fetches a URL and extracts the main textual content.
"""
url = clean_url(url)
try:
headers = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 "
"(KHTML, like Gecko) "
"Chrome/131.0 Safari/537.36"
)
}
response = requests.get(
url,
headers=headers,
timeout=15
)
response.raise_for_status()
# ----------------------------------------------------
# Extract main content
# ----------------------------------------------------
text = trafilatura.extract(
response.text,
include_links=False,
include_images=False,
include_tables=True,
favor_precision=True
)
if not text:
return {
"url": url,
"status": "failed",
"content": "No meaningful content could be extracted."
}
# ----------------------------------------------------
# Clean
# ----------------------------------------------------
clean_text = clean_extracted_text(text)
original_chars = len(clean_text)
# Don't allow enormous documents into the embedding stage
clean_text, source_truncated = limit_text(
clean_text,
MAX_SOURCE_CHARS
)
return {
"url": url,
"status": "success",
"content": clean_text,
"original_chars": original_chars,
"source_truncated": source_truncated
}
except requests.exceptions.Timeout:
return {
"url": url,
"status": "failed",
"content": "Request timed out."
}
except requests.exceptions.HTTPError as e:
return {
"url": url,
"status": "failed",
"content": f"HTTP error: {e}"
}
except Exception as e:
return {
"url": url,
"status": "failed",
"content": str(e)
}
# ============================================================
# Chunking
# ============================================================
def create_chunks(
text: str,
chunk_size: int = CHUNK_SIZE,
overlap: int = CHUNK_OVERLAP
) -> list[str]:
"""
Splits text into overlapping chunks.
Tries to respect paragraph boundaries.
"""
paragraphs = [
p.strip()
for p in text.split("\n")
if p.strip()
]
chunks = []
current = ""
for paragraph in paragraphs:
# If paragraph fits into current chunk
if len(current) + len(paragraph) + 1 <= chunk_size:
if current:
current += "\n"
current += paragraph
else:
if current:
chunks.append(current)
# Handle paragraphs larger than chunk size
if len(paragraph) > chunk_size:
start = 0
while start < len(paragraph):
end = start + chunk_size
piece = paragraph[start:end]
chunks.append(piece)
start = end - overlap
current = ""
else:
current = paragraph
if current:
chunks.append(current)
return chunks
# ============================================================
# Semantic relevance
# ============================================================
def rank_chunks(
query: str,
chunks: list[str]
) -> list[tuple[str, float]]:
"""
Calculates semantic similarity between the query
and every chunk.
Returns chunks sorted by relevance.
"""
if not chunks:
return []
# Encode query
query_embedding = embedding_model.encode(
query,
normalize_embeddings=True,
device=device
)
# Encode chunks
chunk_embeddings = embedding_model.encode(
chunks,
normalize_embeddings=True,
batch_size=128,
show_progress_bar=False,
device=device
)
# Cosine similarity
scores = np.dot(
chunk_embeddings,
query_embedding
)
ranked = sorted(
zip(chunks, scores),
key=lambda x: x[1],
reverse=True
)
return ranked
# ============================================================
# Select relevant content
# ============================================================
def select_relevant_content(
text: str,
query: str,
top_k: int = TOP_K_CHUNKS,
min_score: float = MIN_RELEVANCE_SCORE,
max_chars: int = MAX_CHARS_PER_URL
) -> dict:
# --------------------------------------------------------
# Create chunks
# --------------------------------------------------------
chunks = create_chunks(text)
if not chunks:
return {
"content": "",
"chunks_considered": 0,
"chunks_selected": 0,
"scores": [],
"truncated": False
}
# --------------------------------------------------------
# Rank chunks by semantic similarity
# --------------------------------------------------------
ranked = rank_chunks(
query,
chunks
)
# --------------------------------------------------------
# Select highest-scoring chunks
# --------------------------------------------------------
selected = [
(chunk, float(score))
for chunk, score in ranked
if score >= min_score
][:top_k]
# --------------------------------------------------------
# Fallback:
# If nothing passes the threshold, keep the best chunk
# --------------------------------------------------------
if not selected and ranked:
selected = [
(
ranked[0][0],
float(ranked[0][1])
)
]
# --------------------------------------------------------
# IMPORTANT:
# Keep relevance order.
#
# The first chunk is the most relevant.
# --------------------------------------------------------
selected_chunks = [
chunk
for chunk, score in selected
]
scores = [
round(score, 3)
for chunk, score in selected
]
# --------------------------------------------------------
# Build content
# --------------------------------------------------------
final_content = "\n\n".join(
selected_chunks
)
# --------------------------------------------------------
# Apply character limit
# --------------------------------------------------------
final_content, truncated = limit_text(
final_content,
max_chars
)
# --------------------------------------------------------
# Debug information
# --------------------------------------------------------
debug_header = (
f"[Semantic Retrieval]\n"
f"Query: {query}\n"
f"Chunks considered: {len(chunks)}\n"
f"Chunks selected: {len(selected)}\n"
f"Relevance scores: {scores}\n"
f"Content characters: {len(final_content)}\n"
f"Truncated: {truncated}\n"
f"\n--- RELEVANT CONTENT ---\n"
)
return {
"content": debug_header + final_content,
"chunks_considered": len(chunks),
"chunks_selected": len(selected),
"scores": scores,
"truncated": truncated
}
# ============================================================
# MCP Tool
# ============================================================
@mcp.tool()
def batch_web_content_extraction(
urls: list[str],
query: str = "",
max_chars_per_url: int = MAX_CHARS_PER_URL,
max_total_chars: int = MAX_TOTAL_CHARS,
top_k_chunks: int = TOP_K_CHUNKS,
min_relevance_score: float = MIN_RELEVANCE_SCORE
) -> list[dict]:
"""
Fetches multiple URLs in parallel and extracts the
content relevant to a query.
Args:
urls:
URLs to process.
query:
User question or information need.
Semantic ranking is performed against this query.
max_chars_per_url:
Maximum characters returned for each URL.
max_total_chars:
Maximum characters returned across all URLs.
top_k_chunks:
Maximum number of relevant chunks per URL.
min_relevance_score:
Minimum semantic similarity score.
Returns:
List of relevant web content results.
"""
if not urls:
return [
{
"status": "error",
"content": "No URLs provided."
}
]
# --------------------------------------------------------
# Validate limits
# --------------------------------------------------------
max_chars_per_url = max(
1000,
min(max_chars_per_url, 20000)
)
max_total_chars = max(
5000,
min(max_total_chars, 100000)
)
top_k_chunks = max(
1,
min(top_k_chunks, 20)
)
min_relevance_score = max(
0.0,
min(min_relevance_score, 1.0)
)
# --------------------------------------------------------
# Fetch URLs in parallel
# --------------------------------------------------------
# Clean and deduplicate URLs
cleaned_urls = []
for url in urls:
cleaned = clean_url(url)
if cleaned and cleaned not in cleaned_urls:
cleaned_urls.append(cleaned)
scraped_results = []
with ThreadPoolExecutor(
max_workers=min(len(cleaned_urls), MAX_WORKERS)
) as executor:
futures = {
executor.submit(
scrape_single_url,
url
): url
for url in cleaned_urls
}
for future in as_completed(futures):
try:
result = future.result()
except Exception as e:
result = {
"url": futures[future],
"status": "failed",
"content": str(e)
}
scraped_results.append(result)
# --------------------------------------------------------
# If no query supplied, behave like Phase 1
# --------------------------------------------------------
if not query.strip():
final_results = []
total_chars = 0
for result in scraped_results:
if result.get("status") != "success":
final_results.append(result)
continue
content = result.get("content", "")
remaining = max_total_chars - total_chars
if remaining <= 0:
content = (
"[Content omitted due to total size limit]"
)
else:
content, truncated = limit_text(
content,
min(max_chars_per_url, remaining)
)
result["truncated"] = (
result.get("source_truncated", False)
or truncated
)
total_chars += len(content)
result["content"] = content
result["returned_chars"] = len(content)
final_results.append(result)
return final_results
# --------------------------------------------------------
# Semantic selection
# --------------------------------------------------------
final_results = []
for result in scraped_results:
if result.get("status") != "success":
final_results.append(result)
continue
selection = select_relevant_content(
text=result["content"],
query=query,
top_k=top_k_chunks,
min_score=min_relevance_score,
max_chars=max_chars_per_url
)
result["content"] = selection["content"]
result["chunks_considered"] = (
selection["chunks_considered"]
)
result["chunks_selected"] = (
selection["chunks_selected"]
)
result["relevance_scores"] = (
selection["scores"]
)
result["truncated"] = (
selection["truncated"]
)
result["returned_chars"] = len(
result["content"]
)
final_results.append(result)
# --------------------------------------------------------
# Apply global character budget
# --------------------------------------------------------
total_chars = 0
for result in final_results:
if result.get("status") != "success":
continue
content = result.get("content", "")
remaining = max_total_chars - total_chars
if remaining <= 0:
result["content"] = (
"[Content omitted due to total size limit]"
)
result["returned_chars"] = 0
result["truncated"] = True
elif len(content) > remaining:
content, _ = limit_text(
content,
remaining
)
result["content"] = content
result["returned_chars"] = len(content)
result["truncated"] = True
total_chars += len(content)
else:
total_chars += len(content)
serialized = json.dumps(
final_results,
ensure_ascii=False
)
print("=" * 60)
print(f"MCP FINAL RESULTS:")
print(f"Number of URLs: {len(final_results)}")
print(f"Serialized characters: {len(serialized):,}")
print(
f"Content characters: "
f"{sum(len(r.get('content', '')) for r in final_results):,}"
)
print("=" * 60)
return final_results
if __name__ == "__main__":
# Use standard streamable-http protocol on port 8000
mcp.run(transport="streamable-http", host="0.0.0.0", port=8000)