MCP server that fetches multiple URLs in parallel and extracts the content relevant to a query.
-Clean and deduplicate URLs -Clean extracted text -Semantic selection to keep only text relevant to the query -Limit number of extracted characters per URL and also limit the total characters extracted to avoid returning too much text Arguments: urls: List of URLs to process. query: User question or information wanted Semantic ranking is performed against this query. Optional Arguments: max_chars_per_url: Maximum characters returned for each URL. Default set to 6000 max_total_chars: Maximum characters returned across all URLs. Default set to 30000 top_k_chunks: Maximum number of relevant chunks per URL. Default set to 6 min_relevance_score: Minimum semantic similarity score. Default set to 0.25 Returns: List of relevant web content results. Below an example of info return for one URL: ======================================== === SOURCE URL: https://www.marketsandmarkets.com/Market-Reports/3d-scanner-market-119952472.html === STATUS: success CONTENT: [Semantic Retrieval] Query: Extract information about companies, products, pricing, key features, target customers, market trends, growth, and competitive information related to 3D scanners. Chunks considered: 109 Chunks selected: 6 Relevance scores: [0.759, 0.757, 0.749, 0.71, 0.7, 0.696] Content characters: 3046 Truncated: False --- RELEVANT CONTENT --- Chunk 1 Chunk 2 Chunk 3 Chunk 4 Chunk 5 Chunk 6 ========================================
This commit is contained in:
+12
@@ -0,0 +1,12 @@
|
||||
# Ignore all __pycache__ directories
|
||||
**/__pycache__/
|
||||
|
||||
# Ignore compiled Python files
|
||||
*.py[cod]
|
||||
|
||||
venv/
|
||||
log/
|
||||
output/
|
||||
|
||||
|
||||
bin/__pycache__/
|
||||
@@ -0,0 +1,51 @@
|
||||
# mcp_server.py
|
||||
import requests
|
||||
import json
|
||||
from bs4 import BeautifulSoup
|
||||
from fastmcp import FastMCP
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
|
||||
mcp = FastMCP("BatchWebContentExtraction")
|
||||
|
||||
def scrape_single_url(url: str) -> dict:
|
||||
"""Helper function to scrape and clean a single URL."""
|
||||
try:
|
||||
headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
|
||||
response = requests.get(url, headers=headers, timeout=10)
|
||||
response.raise_for_status()
|
||||
|
||||
soup = BeautifulSoup(response.text, 'html.parser')
|
||||
for script in soup(["script", "style", "nav", "footer", "header", "aside"]):
|
||||
script.decompose()
|
||||
|
||||
raw_text = soup.get_text(separator=' ')
|
||||
lines = (line.strip() for line in raw_text.splitlines())
|
||||
chunks = (phrase.strip() for line in lines for phrase in line.split(" "))
|
||||
clean_text = '\n'.join(chunk for chunk in chunks if chunk)
|
||||
|
||||
return {"url": url, "status": "success", "content": clean_text[:5000]}
|
||||
|
||||
except Exception as e:
|
||||
return {"url": url, "status": "failed", "content": str(e)}
|
||||
|
||||
@mcp.tool()
|
||||
def batch_web_content_extraction(urls: list[str]) -> str:
|
||||
"""
|
||||
Fetches text content from a list of multiple URLs simultaneously in parallel.
|
||||
Returns a unified JSON string of results.
|
||||
"""
|
||||
if not urls:
|
||||
return json.dumps({"error": "No URLs provided"})
|
||||
|
||||
results = []
|
||||
with ThreadPoolExecutor(max_workers=min(len(urls), 5)) as executor:
|
||||
futures_map = {executor.submit(scrape_single_url, url): url for url in urls}
|
||||
for future in futures_map:
|
||||
results.append(future.result())
|
||||
|
||||
return json.dumps(results)
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Use standard streamable-http protocol on port 8000
|
||||
mcp.run(transport="streamable-http", host="0.0.0.0", port=8000)
|
||||
|
||||
+826
@@ -0,0 +1,826 @@
|
||||
# mcp_server.py
|
||||
import json
|
||||
|
||||
import requests
|
||||
#from bs4 import BeautifulSoup
|
||||
from fastmcp import FastMCP
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
import trafilatura
|
||||
import requests
|
||||
from sentence_transformers import SentenceTransformer
|
||||
import numpy as np
|
||||
import torch
|
||||
from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode
|
||||
|
||||
|
||||
mcp = FastMCP("BatchWebContentExtraction")
|
||||
|
||||
|
||||
import re
|
||||
from urllib.parse import (
|
||||
urlparse,
|
||||
urlunparse,
|
||||
parse_qsl,
|
||||
urlencode
|
||||
)
|
||||
|
||||
TRACKING_PARAMS = {
|
||||
"utm_source",
|
||||
"utm_medium",
|
||||
"utm_campaign",
|
||||
"utm_term",
|
||||
"utm_content",
|
||||
"utm_id",
|
||||
"gclid",
|
||||
"fbclid",
|
||||
"msclkid",
|
||||
"dclid",
|
||||
"srsltid",
|
||||
"mc_cid",
|
||||
"mc_eid",
|
||||
}
|
||||
|
||||
|
||||
def clean_url(url: str) -> str:
|
||||
"""Normalize URLs and remove Markdown/tracking wrappers."""
|
||||
|
||||
if not url:
|
||||
return ""
|
||||
|
||||
url = url.strip()
|
||||
|
||||
# --------------------------------------------------
|
||||
# 1. Extract URL from Markdown link
|
||||
#
|
||||
# [https://example.com/page](https://example.com/page)
|
||||
# ↓
|
||||
# https://example.com/page
|
||||
# --------------------------------------------------
|
||||
|
||||
markdown_match = re.match(
|
||||
r"^\[.*?\]\((https?://[^)]+)\)$",
|
||||
url
|
||||
)
|
||||
|
||||
if markdown_match:
|
||||
url = markdown_match.group(1)
|
||||
|
||||
# --------------------------------------------------
|
||||
# 2. Remove accidental surrounding quotes
|
||||
# --------------------------------------------------
|
||||
|
||||
url = url.strip("\"'")
|
||||
|
||||
# --------------------------------------------------
|
||||
# 3. Parse URL
|
||||
# --------------------------------------------------
|
||||
|
||||
parsed = urlparse(url)
|
||||
|
||||
if parsed.scheme not in ("http", "https"):
|
||||
return ""
|
||||
|
||||
# --------------------------------------------------
|
||||
# 4. Remove tracking parameters
|
||||
# --------------------------------------------------
|
||||
|
||||
query_params = [
|
||||
(key, value)
|
||||
for key, value in parse_qsl(
|
||||
parsed.query,
|
||||
keep_blank_values=True
|
||||
)
|
||||
if key.lower() not in TRACKING_PARAMS
|
||||
]
|
||||
|
||||
# --------------------------------------------------
|
||||
# 5. Rebuild clean URL
|
||||
# --------------------------------------------------
|
||||
|
||||
cleaned = urlunparse((
|
||||
parsed.scheme.lower(),
|
||||
parsed.netloc.lower(),
|
||||
parsed.path.rstrip("/") or "/",
|
||||
parsed.params,
|
||||
urlencode(query_params),
|
||||
""
|
||||
))
|
||||
|
||||
return cleaned
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Configuration
|
||||
# ============================================================
|
||||
|
||||
MAX_WORKERS = 5
|
||||
|
||||
# Maximum characters considered from each URL
|
||||
MAX_SOURCE_CHARS = 50000
|
||||
|
||||
# Maximum characters returned from each URL
|
||||
MAX_CHARS_PER_URL = 6000
|
||||
|
||||
# Maximum characters returned by the whole tool
|
||||
MAX_TOTAL_CHARS = 30000
|
||||
|
||||
# Number of relevant chunks to keep from each URL
|
||||
TOP_K_CHUNKS = 6
|
||||
|
||||
# Minimum semantic similarity score
|
||||
MIN_RELEVANCE_SCORE = 0.25
|
||||
|
||||
# Chunk size
|
||||
CHUNK_SIZE = 600
|
||||
|
||||
# Slight overlap between chunks
|
||||
CHUNK_OVERLAP = 100
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Load embedding model ONCE
|
||||
# ============================================================
|
||||
|
||||
device = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
print(f"Embedding device: {device}")
|
||||
|
||||
if device == "cuda":
|
||||
print(f"GPU: {torch.cuda.get_device_name(0)}")
|
||||
|
||||
embedding_model = SentenceTransformer(
|
||||
"all-MiniLM-L6-v2",
|
||||
device=device
|
||||
)
|
||||
|
||||
print("Embedding model loaded.")
|
||||
|
||||
# ============================================================
|
||||
# Text cleaning
|
||||
# ============================================================
|
||||
|
||||
def clean_extracted_text(text: str) -> str:
|
||||
"""
|
||||
Performs lightweight cleanup after Trafilatura extraction.
|
||||
"""
|
||||
|
||||
if not text:
|
||||
return ""
|
||||
|
||||
text = text.replace("\r\n", "\n")
|
||||
text = text.replace("\r", "\n")
|
||||
|
||||
lines = []
|
||||
|
||||
for line in text.split("\n"):
|
||||
|
||||
line = " ".join(line.split())
|
||||
|
||||
if line:
|
||||
lines.append(line)
|
||||
|
||||
# Remove duplicate consecutive lines
|
||||
cleaned = []
|
||||
|
||||
previous = None
|
||||
|
||||
for line in lines:
|
||||
|
||||
if line != previous:
|
||||
cleaned.append(line)
|
||||
|
||||
previous = line
|
||||
|
||||
return "\n".join(cleaned)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Limit text
|
||||
# ============================================================
|
||||
|
||||
def limit_text(text: str, max_chars: int) -> tuple[str, bool]:
|
||||
"""
|
||||
Limits text without cutting a paragraph whenever possible.
|
||||
"""
|
||||
|
||||
if len(text) <= max_chars:
|
||||
return text, False
|
||||
|
||||
truncated = text[:max_chars]
|
||||
|
||||
last_newline = truncated.rfind("\n")
|
||||
|
||||
if last_newline > max_chars * 0.7:
|
||||
truncated = truncated[:last_newline]
|
||||
|
||||
return (
|
||||
truncated.rstrip() + "\n[Content truncated]",
|
||||
True
|
||||
)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Fetch and extract URL
|
||||
# ============================================================
|
||||
|
||||
def scrape_single_url(url: str) -> dict:
|
||||
"""
|
||||
Fetches a URL and extracts the main textual content.
|
||||
"""
|
||||
|
||||
url = clean_url(url)
|
||||
|
||||
try:
|
||||
|
||||
headers = {
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 "
|
||||
"(KHTML, like Gecko) "
|
||||
"Chrome/131.0 Safari/537.36"
|
||||
)
|
||||
}
|
||||
|
||||
response = requests.get(
|
||||
url,
|
||||
headers=headers,
|
||||
timeout=15
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
|
||||
# ----------------------------------------------------
|
||||
# Extract main content
|
||||
# ----------------------------------------------------
|
||||
|
||||
text = trafilatura.extract(
|
||||
response.text,
|
||||
include_links=False,
|
||||
include_images=False,
|
||||
include_tables=True,
|
||||
favor_precision=True
|
||||
)
|
||||
|
||||
if not text:
|
||||
|
||||
return {
|
||||
"url": url,
|
||||
"status": "failed",
|
||||
"content": "No meaningful content could be extracted."
|
||||
}
|
||||
|
||||
# ----------------------------------------------------
|
||||
# Clean
|
||||
# ----------------------------------------------------
|
||||
|
||||
clean_text = clean_extracted_text(text)
|
||||
|
||||
original_chars = len(clean_text)
|
||||
|
||||
# Don't allow enormous documents into the embedding stage
|
||||
clean_text, source_truncated = limit_text(
|
||||
clean_text,
|
||||
MAX_SOURCE_CHARS
|
||||
)
|
||||
|
||||
return {
|
||||
"url": url,
|
||||
"status": "success",
|
||||
"content": clean_text,
|
||||
"original_chars": original_chars,
|
||||
"source_truncated": source_truncated
|
||||
}
|
||||
|
||||
except requests.exceptions.Timeout:
|
||||
|
||||
return {
|
||||
"url": url,
|
||||
"status": "failed",
|
||||
"content": "Request timed out."
|
||||
}
|
||||
|
||||
except requests.exceptions.HTTPError as e:
|
||||
|
||||
return {
|
||||
"url": url,
|
||||
"status": "failed",
|
||||
"content": f"HTTP error: {e}"
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
|
||||
return {
|
||||
"url": url,
|
||||
"status": "failed",
|
||||
"content": str(e)
|
||||
}
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Chunking
|
||||
# ============================================================
|
||||
|
||||
def create_chunks(
|
||||
text: str,
|
||||
chunk_size: int = CHUNK_SIZE,
|
||||
overlap: int = CHUNK_OVERLAP
|
||||
) -> list[str]:
|
||||
"""
|
||||
Splits text into overlapping chunks.
|
||||
|
||||
Tries to respect paragraph boundaries.
|
||||
"""
|
||||
|
||||
paragraphs = [
|
||||
p.strip()
|
||||
for p in text.split("\n")
|
||||
if p.strip()
|
||||
]
|
||||
|
||||
chunks = []
|
||||
current = ""
|
||||
|
||||
for paragraph in paragraphs:
|
||||
|
||||
# If paragraph fits into current chunk
|
||||
if len(current) + len(paragraph) + 1 <= chunk_size:
|
||||
|
||||
if current:
|
||||
current += "\n"
|
||||
|
||||
current += paragraph
|
||||
|
||||
else:
|
||||
|
||||
if current:
|
||||
chunks.append(current)
|
||||
|
||||
# Handle paragraphs larger than chunk size
|
||||
if len(paragraph) > chunk_size:
|
||||
|
||||
start = 0
|
||||
|
||||
while start < len(paragraph):
|
||||
|
||||
end = start + chunk_size
|
||||
|
||||
piece = paragraph[start:end]
|
||||
|
||||
chunks.append(piece)
|
||||
|
||||
start = end - overlap
|
||||
|
||||
current = ""
|
||||
|
||||
else:
|
||||
|
||||
current = paragraph
|
||||
|
||||
if current:
|
||||
chunks.append(current)
|
||||
|
||||
return chunks
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Semantic relevance
|
||||
# ============================================================
|
||||
|
||||
def rank_chunks(
|
||||
query: str,
|
||||
chunks: list[str]
|
||||
) -> list[tuple[str, float]]:
|
||||
"""
|
||||
Calculates semantic similarity between the query
|
||||
and every chunk.
|
||||
|
||||
Returns chunks sorted by relevance.
|
||||
"""
|
||||
|
||||
if not chunks:
|
||||
return []
|
||||
|
||||
# Encode query
|
||||
query_embedding = embedding_model.encode(
|
||||
query,
|
||||
normalize_embeddings=True,
|
||||
device=device
|
||||
)
|
||||
|
||||
# Encode chunks
|
||||
chunk_embeddings = embedding_model.encode(
|
||||
chunks,
|
||||
normalize_embeddings=True,
|
||||
batch_size=128,
|
||||
show_progress_bar=False,
|
||||
device=device
|
||||
)
|
||||
|
||||
# Cosine similarity
|
||||
scores = np.dot(
|
||||
chunk_embeddings,
|
||||
query_embedding
|
||||
)
|
||||
|
||||
ranked = sorted(
|
||||
zip(chunks, scores),
|
||||
key=lambda x: x[1],
|
||||
reverse=True
|
||||
)
|
||||
|
||||
return ranked
|
||||
|
||||
|
||||
# ============================================================
|
||||
# Select relevant content
|
||||
# ============================================================
|
||||
|
||||
def select_relevant_content(
|
||||
text: str,
|
||||
query: str,
|
||||
top_k: int = TOP_K_CHUNKS,
|
||||
min_score: float = MIN_RELEVANCE_SCORE,
|
||||
max_chars: int = MAX_CHARS_PER_URL
|
||||
) -> dict:
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Create chunks
|
||||
# --------------------------------------------------------
|
||||
|
||||
chunks = create_chunks(text)
|
||||
|
||||
if not chunks:
|
||||
return {
|
||||
"content": "",
|
||||
"chunks_considered": 0,
|
||||
"chunks_selected": 0,
|
||||
"scores": [],
|
||||
"truncated": False
|
||||
}
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Rank chunks by semantic similarity
|
||||
# --------------------------------------------------------
|
||||
|
||||
ranked = rank_chunks(
|
||||
query,
|
||||
chunks
|
||||
)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Select highest-scoring chunks
|
||||
# --------------------------------------------------------
|
||||
|
||||
selected = [
|
||||
(chunk, float(score))
|
||||
for chunk, score in ranked
|
||||
if score >= min_score
|
||||
][:top_k]
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Fallback:
|
||||
# If nothing passes the threshold, keep the best chunk
|
||||
# --------------------------------------------------------
|
||||
|
||||
if not selected and ranked:
|
||||
|
||||
selected = [
|
||||
(
|
||||
ranked[0][0],
|
||||
float(ranked[0][1])
|
||||
)
|
||||
]
|
||||
|
||||
# --------------------------------------------------------
|
||||
# IMPORTANT:
|
||||
# Keep relevance order.
|
||||
#
|
||||
# The first chunk is the most relevant.
|
||||
# --------------------------------------------------------
|
||||
|
||||
selected_chunks = [
|
||||
chunk
|
||||
for chunk, score in selected
|
||||
]
|
||||
|
||||
scores = [
|
||||
round(score, 3)
|
||||
for chunk, score in selected
|
||||
]
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Build content
|
||||
# --------------------------------------------------------
|
||||
|
||||
final_content = "\n\n".join(
|
||||
selected_chunks
|
||||
)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Apply character limit
|
||||
# --------------------------------------------------------
|
||||
|
||||
final_content, truncated = limit_text(
|
||||
final_content,
|
||||
max_chars
|
||||
)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Debug information
|
||||
# --------------------------------------------------------
|
||||
|
||||
debug_header = (
|
||||
f"[Semantic Retrieval]\n"
|
||||
f"Query: {query}\n"
|
||||
f"Chunks considered: {len(chunks)}\n"
|
||||
f"Chunks selected: {len(selected)}\n"
|
||||
f"Relevance scores: {scores}\n"
|
||||
f"Content characters: {len(final_content)}\n"
|
||||
f"Truncated: {truncated}\n"
|
||||
f"\n--- RELEVANT CONTENT ---\n"
|
||||
)
|
||||
|
||||
return {
|
||||
"content": debug_header + final_content,
|
||||
"chunks_considered": len(chunks),
|
||||
"chunks_selected": len(selected),
|
||||
"scores": scores,
|
||||
"truncated": truncated
|
||||
}
|
||||
|
||||
# ============================================================
|
||||
# MCP Tool
|
||||
# ============================================================
|
||||
|
||||
@mcp.tool()
|
||||
def batch_web_content_extraction(
|
||||
urls: list[str],
|
||||
query: str = "",
|
||||
max_chars_per_url: int = MAX_CHARS_PER_URL,
|
||||
max_total_chars: int = MAX_TOTAL_CHARS,
|
||||
top_k_chunks: int = TOP_K_CHUNKS,
|
||||
min_relevance_score: float = MIN_RELEVANCE_SCORE
|
||||
) -> list[dict]:
|
||||
"""
|
||||
Fetches multiple URLs in parallel and extracts the
|
||||
content relevant to a query.
|
||||
|
||||
Args:
|
||||
|
||||
urls:
|
||||
URLs to process.
|
||||
|
||||
query:
|
||||
User question or information need.
|
||||
Semantic ranking is performed against this query.
|
||||
|
||||
max_chars_per_url:
|
||||
Maximum characters returned for each URL.
|
||||
|
||||
max_total_chars:
|
||||
Maximum characters returned across all URLs.
|
||||
|
||||
top_k_chunks:
|
||||
Maximum number of relevant chunks per URL.
|
||||
|
||||
min_relevance_score:
|
||||
Minimum semantic similarity score.
|
||||
|
||||
Returns:
|
||||
List of relevant web content results.
|
||||
"""
|
||||
|
||||
if not urls:
|
||||
|
||||
return [
|
||||
{
|
||||
"status": "error",
|
||||
"content": "No URLs provided."
|
||||
}
|
||||
]
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Validate limits
|
||||
# --------------------------------------------------------
|
||||
|
||||
max_chars_per_url = max(
|
||||
1000,
|
||||
min(max_chars_per_url, 20000)
|
||||
)
|
||||
|
||||
max_total_chars = max(
|
||||
5000,
|
||||
min(max_total_chars, 100000)
|
||||
)
|
||||
|
||||
top_k_chunks = max(
|
||||
1,
|
||||
min(top_k_chunks, 20)
|
||||
)
|
||||
|
||||
min_relevance_score = max(
|
||||
0.0,
|
||||
min(min_relevance_score, 1.0)
|
||||
)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Fetch URLs in parallel
|
||||
# --------------------------------------------------------
|
||||
|
||||
# Clean and deduplicate URLs
|
||||
cleaned_urls = []
|
||||
|
||||
for url in urls:
|
||||
cleaned = clean_url(url)
|
||||
|
||||
if cleaned and cleaned not in cleaned_urls:
|
||||
cleaned_urls.append(cleaned)
|
||||
|
||||
|
||||
scraped_results = []
|
||||
|
||||
with ThreadPoolExecutor(
|
||||
max_workers=min(len(cleaned_urls), MAX_WORKERS)
|
||||
) as executor:
|
||||
|
||||
futures = {
|
||||
executor.submit(
|
||||
scrape_single_url,
|
||||
url
|
||||
): url
|
||||
for url in cleaned_urls
|
||||
}
|
||||
|
||||
for future in as_completed(futures):
|
||||
|
||||
try:
|
||||
|
||||
result = future.result()
|
||||
|
||||
except Exception as e:
|
||||
|
||||
result = {
|
||||
"url": futures[future],
|
||||
"status": "failed",
|
||||
"content": str(e)
|
||||
}
|
||||
|
||||
scraped_results.append(result)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# If no query supplied, behave like Phase 1
|
||||
# --------------------------------------------------------
|
||||
|
||||
if not query.strip():
|
||||
|
||||
final_results = []
|
||||
total_chars = 0
|
||||
|
||||
for result in scraped_results:
|
||||
|
||||
if result.get("status") != "success":
|
||||
final_results.append(result)
|
||||
continue
|
||||
|
||||
content = result.get("content", "")
|
||||
|
||||
remaining = max_total_chars - total_chars
|
||||
|
||||
if remaining <= 0:
|
||||
|
||||
content = (
|
||||
"[Content omitted due to total size limit]"
|
||||
)
|
||||
|
||||
else:
|
||||
|
||||
content, truncated = limit_text(
|
||||
content,
|
||||
min(max_chars_per_url, remaining)
|
||||
)
|
||||
|
||||
result["truncated"] = (
|
||||
result.get("source_truncated", False)
|
||||
or truncated
|
||||
)
|
||||
|
||||
total_chars += len(content)
|
||||
|
||||
result["content"] = content
|
||||
result["returned_chars"] = len(content)
|
||||
|
||||
final_results.append(result)
|
||||
|
||||
return final_results
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Semantic selection
|
||||
# --------------------------------------------------------
|
||||
|
||||
final_results = []
|
||||
|
||||
for result in scraped_results:
|
||||
|
||||
if result.get("status") != "success":
|
||||
|
||||
final_results.append(result)
|
||||
continue
|
||||
|
||||
selection = select_relevant_content(
|
||||
text=result["content"],
|
||||
query=query,
|
||||
top_k=top_k_chunks,
|
||||
min_score=min_relevance_score,
|
||||
max_chars=max_chars_per_url
|
||||
)
|
||||
|
||||
result["content"] = selection["content"]
|
||||
|
||||
result["chunks_considered"] = (
|
||||
selection["chunks_considered"]
|
||||
)
|
||||
|
||||
result["chunks_selected"] = (
|
||||
selection["chunks_selected"]
|
||||
)
|
||||
|
||||
result["relevance_scores"] = (
|
||||
selection["scores"]
|
||||
)
|
||||
|
||||
result["truncated"] = (
|
||||
selection["truncated"]
|
||||
)
|
||||
|
||||
result["returned_chars"] = len(
|
||||
result["content"]
|
||||
)
|
||||
|
||||
final_results.append(result)
|
||||
|
||||
# --------------------------------------------------------
|
||||
# Apply global character budget
|
||||
# --------------------------------------------------------
|
||||
|
||||
total_chars = 0
|
||||
|
||||
for result in final_results:
|
||||
|
||||
if result.get("status") != "success":
|
||||
continue
|
||||
|
||||
content = result.get("content", "")
|
||||
|
||||
remaining = max_total_chars - total_chars
|
||||
|
||||
if remaining <= 0:
|
||||
|
||||
result["content"] = (
|
||||
"[Content omitted due to total size limit]"
|
||||
)
|
||||
|
||||
result["returned_chars"] = 0
|
||||
result["truncated"] = True
|
||||
|
||||
elif len(content) > remaining:
|
||||
|
||||
content, _ = limit_text(
|
||||
content,
|
||||
remaining
|
||||
)
|
||||
|
||||
result["content"] = content
|
||||
result["returned_chars"] = len(content)
|
||||
result["truncated"] = True
|
||||
|
||||
total_chars += len(content)
|
||||
|
||||
else:
|
||||
|
||||
total_chars += len(content)
|
||||
|
||||
|
||||
|
||||
serialized = json.dumps(
|
||||
final_results,
|
||||
ensure_ascii=False
|
||||
)
|
||||
|
||||
print("=" * 60)
|
||||
print(f"MCP FINAL RESULTS:")
|
||||
print(f"Number of URLs: {len(final_results)}")
|
||||
print(f"Serialized characters: {len(serialized):,}")
|
||||
print(
|
||||
f"Content characters: "
|
||||
f"{sum(len(r.get('content', '')) for r in final_results):,}"
|
||||
)
|
||||
print("=" * 60)
|
||||
|
||||
|
||||
return final_results
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Use standard streamable-http protocol on port 8000
|
||||
mcp.run(transport="streamable-http", host="0.0.0.0", port=8000)
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
requests
|
||||
#beautifulsoup4
|
||||
trafilatura
|
||||
sentence-transformers
|
||||
fastmcp
|
||||
torch
|
||||
Reference in New Issue
Block a user