From f8b2e93cc12043222a73ebc89c202fcdf730ac4f Mon Sep 17 00:00:00 2001 From: Vincent Date: Sat, 22 Aug 2026 10:21:10 +0900 Subject: [PATCH] Add readme file --- README.md | 130 ++++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 130 insertions(+) create mode 100644 README.md diff --git a/README.md b/README.md new file mode 100644 index 0000000..3e23543 --- /dev/null +++ b/README.md @@ -0,0 +1,130 @@ +# Go through the following steps to use this MCP serer: +1) Create Python Virtual Env from your root directory +python -m venv venv + +2) Activate Virtual Env +a) Windows: venv\Scripts\activate.bat +b) Unix: source venv/bin/activate + +3) Installing Dependencies +pip install -r requirements.txt + +4) Start MCP Sever +python -m mcp_server + + +5) In your project where you want to use this Server, implemt a tool like the following: + +@tool("FastMCP Batch Web Content extraction Tool") + def fastmcp_batch_web_content_extraction_tool(urls: list[str], query: str) -> str: + """ + Connects directly to the FastMCP HTTP server to scrape and extract text from + a list of multiple URLs simultaneously in parallel. + """ + async def call_fast_mcp(): + # Initialize client using FastMCP's precise matching transport protocol + transport = StreamableHttpTransport("http://localhost:8000/mcp/") + + async with Client(transport) as client: + # call_tool abstracts away deep JSON-RPC structures + result = await client.call_tool("batch_web_content_extraction", {"urls": urls, "query": query}) + return result + + try: + # Spin up clean execution loop for this worker thread + mcp_response = asyncio.run(call_fast_mcp()) + + # FastMCP Client response objects expose string contents naturally via .content + # Ensure we safely extract the raw text string from the MCP content blocks + if hasattr(mcp_response, "content") and mcp_response.content: + # The text is inside the first content block object + raw_content = mcp_response.content[0].text + else: + # Fallback to string casting if it's already a plain string + raw_content = str(mcp_response) + + # Parse JSON and pretty-format the string back to the CrewAI Agent + try: + parsed_data = json.loads(raw_content) + formatted_output = [] + for item in parsed_data: + formatted_output.append(f"=== SOURCE URL: {item['url']} ===") + formatted_output.append(f"STATUS: {item['status']}") + formatted_output.append(f"CONTENT:\n{item['content']}\n") + formatted_output.append("=" * 40 + "\n") + + logger.info(f"""FastMCP Batch Web Content extraction tool used.\nURLs:\n{"\n".join(urls)}\nQuery: {query}\nResult:\n{"\n".join(formatted_output)}""") + return "\n".join(formatted_output) + except json.JSONDecodeError: + logger.info(f"""FastMCP Batch Web Content extraction tool used. Error in JSON formatting so raw data returned:\nURLs:\n{"\n".join(urls)}\nQuery: {query}\nResult:\n{raw_content}""") + return raw_content + + except Exception as e: + logger.info(f"""FastMCP Batch Web Content extraction tool used. Error executing the tool and no data returned.""") + return f"Error executing Parallel FastMCP tool over Streamable HTTP: {str(e)}" + + +# Info about this Server + +MCP server that fetches multiple URLs in parallel and extracts the content relevant to a query. +-Clean and deduplicate URLs +-Clean extracted text +-Semantic selection to keep only text relevant to the query +-Limit number of extracted characters per URL and also limit the total characters extracted to avoid returning too much text + +Arguments: +urls: +List of URLs to process. + +query: +User question or information wanted +Semantic ranking is performed against this query. + +Optional Arguments: +max_chars_per_url: +Maximum characters returned for each URL. +Default set to 6000 + +max_total_chars: +Maximum characters returned across all URLs. +Default set to 30000 + +top_k_chunks: +Maximum number of relevant chunks per URL. +Default set to 6 + +min_relevance_score: +Minimum semantic similarity score. +Default set to 0.25 + +Returns: +List of relevant web content results. +Below an example of info return for one URL: + +======================================== + +=== SOURCE URL: https://www.marketsandmarkets.com/Market-Reports/3d-scanner-market-119952472.html === +STATUS: success +CONTENT: +[Semantic Retrieval] +Query: Extract information about companies, products, pricing, key features, target customers, market trends, growth, and competitive information related to 3D scanners. +Chunks considered: 109 +Chunks selected: 6 +Relevance scores: [0.759, 0.757, 0.749, 0.71, 0.7, 0.696] +Content characters: 3046 +Truncated: False + +--- RELEVANT CONTENT --- +Chunk 1 + +Chunk 2 + +Chunk 3 + +Chunk 4 + +Chunk 5 + +Chunk 6 + +======================================== \ No newline at end of file