From 7f78f87508e3d9bd968f774f993dafa0b5e74132 Mon Sep 17 00:00:00 2001 From: saidsurucu Date: Mon, 9 Mar 2026 22:17:27 +0300 Subject: [PATCH] feat: Add Sigorta Tahkim Komisyonu MCP module (3 tools) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add Insurance Arbitration Commission integration with Tavily search and direct PDF download for 64 quarterly journal issues (2010-2025). Tools: - search_sigorta_tahkim_decisions: Search via Tavily API - get_sigorta_tahkim_document_markdown: PDF download + paginated markdown - search_within_sigorta_tahkim_issue: Keyword search within individual decisions of a journal issue, with Turkish İ/I case folding support Total tools: 25 (was 22) --- mcp_server_main.py | 168 ++++++++++++- sigorta_tahkim_mcp_module/__init__.py | 21 ++ sigorta_tahkim_mcp_module/client.py | 340 ++++++++++++++++++++++++++ sigorta_tahkim_mcp_module/models.py | 59 +++++ 4 files changed, 587 insertions(+), 1 deletion(-) create mode 100644 sigorta_tahkim_mcp_module/__init__.py create mode 100644 sigorta_tahkim_mcp_module/client.py create mode 100644 sigorta_tahkim_mcp_module/models.py diff --git a/mcp_server_main.py b/mcp_server_main.py index 265f1f5..bb1c583 100644 --- a/mcp_server_main.py +++ b/mcp_server_main.py @@ -329,6 +329,12 @@ from bddk_mcp_module.models import ( BddkSearchRequest ) +# Sigorta Tahkim Module Imports +from sigorta_tahkim_mcp_module.client import SigortaTahkimApiClient +from sigorta_tahkim_mcp_module.models import ( + SigortaTahkimSearchRequest +) + # Create a placeholder app that will be properly initialized after tools are defined @@ -355,6 +361,7 @@ sayistay_client_instance = SayistayApiClient() sayistay_unified_client_instance = SayistayUnifiedClient() kvkk_client_instance = KvkkApiClient() bddk_client_instance = BddkApiClient() +sigorta_tahkim_client_instance = SigortaTahkimApiClient() # Health check client (singleton for reuse) _health_check_client: Optional[httpx.AsyncClient] = None @@ -1702,7 +1709,8 @@ def perform_cleanup(): globals().get('sayistay_client_instance'), globals().get('sayistay_unified_client_instance'), globals().get('kvkk_client_instance'), - globals().get('bddk_client_instance') + globals().get('bddk_client_instance'), + globals().get('sigorta_tahkim_client_instance') ] async def close_all_clients_async(): tasks = [] @@ -2104,6 +2112,164 @@ async def get_bddk_document_markdown( "error": str(e) } +# --- MCP Tools for Sigorta Tahkim Komisyonu (Insurance Arbitration Commission) --- +@app.tool( + description="Search Sigorta Tahkim Komisyonu (Insurance Arbitration Commission) decisions from Hakem Karar Dergisi journals (64 issues, 2010-2025). Covers insurance disputes: traffic, health, fire, DASK, life insurance.", + annotations={ + "readOnlyHint": True, + "openWorldHint": True, + "idempotentHint": True + } +) +async def search_sigorta_tahkim_decisions( + keywords: str = Field(..., description="Search keywords in Turkish (e.g., 'trafik sigortası', 'kasko', 'DASK')"), + page: int = Field(1, ge=1, description="Page number") +) -> dict: + """Search Sigorta Tahkim Komisyonu insurance arbitration decisions.""" + logger.info(f"Sigorta Tahkim search tool called with keywords: {keywords}, page: {page}") + + pageSize = 10 + + try: + search_request = SigortaTahkimSearchRequest( + keywords=keywords, + page=page, + pageSize=pageSize + ) + + result = await sigorta_tahkim_client_instance.search_decisions(search_request) + logger.info(f"Sigorta Tahkim search completed. Found {len(result.decisions)} results on page {page}") + + return { + "decisions": [ + { + "title": dec.title, + "document_id": dec.document_id, + "content": dec.content, + "url": dec.url + } + for dec in result.decisions + ], + "total_results": result.total_results, + "page": result.page, + "pageSize": result.pageSize + } + + except Exception as e: + logger.exception(f"Error searching Sigorta Tahkim decisions: {e}") + return { + "decisions": [], + "total_results": 0, + "page": page, + "pageSize": pageSize, + "error": str(e) + } + +@app.tool( + description="Retrieve full PDF content of a Sigorta Tahkim Komisyonu Hakem Karar Dergisi issue by number. Returns paginated Markdown. Issues 1-64 available (2010-2025).", + annotations={ + "readOnlyHint": True, + "openWorldHint": False, + "idempotentHint": True + } +) +async def get_sigorta_tahkim_document_markdown( + issue_number: str = Field(..., description="Journal issue number (1-64, e.g., '64')"), + page_number: int = Field(1, ge=1, description="Page number for paginated content") +) -> dict: + """Retrieve Sigorta Tahkim journal issue PDF as paginated Markdown.""" + logger.info(f"Sigorta Tahkim document retrieval for issue: {issue_number}, page: {page_number}") + + if not issue_number or not issue_number.strip(): + return { + "document_id": issue_number, + "markdown_content": "", + "page_number": page_number, + "total_pages": 0, + "source_url": "", + "error": "Issue number is required" + } + + try: + result = await sigorta_tahkim_client_instance.get_document_markdown(issue_number, page_number) + logger.info(f"Sigorta Tahkim document retrieved. Page {result.page_number}/{result.total_pages}") + + return { + "document_id": result.document_id, + "markdown_content": result.markdown_content, + "page_number": result.page_number, + "total_pages": result.total_pages, + "source_url": result.source_url + } + + except Exception as e: + logger.exception(f"Error retrieving Sigorta Tahkim document: {e}") + return { + "document_id": issue_number, + "markdown_content": "", + "page_number": page_number, + "total_pages": 0, + "source_url": "", + "error": str(e) + } + +@app.tool( + description="Search within a specific Sigorta Tahkim Komisyonu journal issue for keywords. Downloads the PDF, splits into individual decisions, and returns matching decisions with excerpts sorted by relevance.", + annotations={ + "readOnlyHint": True, + "openWorldHint": False, + "idempotentHint": True + } +) +async def search_within_sigorta_tahkim_issue( + issue_number: str = Field(..., description="Journal issue number (1-64, e.g., '64')"), + keyword: str = Field(..., description="Search keyword in Turkish (e.g., 'trafik kazası', 'tazminat')"), + max_results: int = Field(10, ge=1, le=25, description="Max matching decisions to return") +) -> dict: + """Search for keywords within a specific Sigorta Tahkim journal issue's decisions.""" + logger.info(f"Sigorta Tahkim search_within called: issue={issue_number}, keyword={keyword}") + + if not issue_number or not issue_number.strip(): + return {"issue_number": issue_number, "keyword": keyword, "matches": [], "error": "Issue number is required"} + if not keyword or not keyword.strip(): + return {"issue_number": issue_number, "keyword": keyword, "matches": [], "error": "Keyword is required"} + + try: + result = await sigorta_tahkim_client_instance.search_within_issue( + issue_number, keyword, max_results + ) + logger.info( + f"Sigorta Tahkim search_within completed: " + f"{result.matching_decisions}/{result.total_decisions} decisions match" + ) + + return { + "issue_number": result.issue_number, + "keyword": result.keyword, + "total_decisions": result.total_decisions, + "matching_decisions": result.matching_decisions, + "matches": [ + { + "decision_header": m.decision_header, + "relevance_score": m.relevance_score, + "excerpt": m.excerpt, + "body_length": m.body_length + } + for m in result.matches + ] + } + + except Exception as e: + logger.exception(f"Error in search_within Sigorta Tahkim: {e}") + return { + "issue_number": issue_number, + "keyword": keyword, + "total_decisions": 0, + "matching_decisions": 0, + "matches": [], + "error": str(e) + } + # --- ChatGPT Deep Research Compatible Tools --- def get_preview_text(markdown_content: str, skip_chars: int = 100, preview_chars: int = 200) -> str: diff --git a/sigorta_tahkim_mcp_module/__init__.py b/sigorta_tahkim_mcp_module/__init__.py new file mode 100644 index 0000000..9b367c0 --- /dev/null +++ b/sigorta_tahkim_mcp_module/__init__.py @@ -0,0 +1,21 @@ +# sigorta_tahkim_mcp_module/__init__.py + +from .client import SigortaTahkimApiClient +from .models import ( + SigortaTahkimSearchRequest, + SigortaTahkimDecisionSummary, + SigortaTahkimSearchResult, + SigortaTahkimDocumentMarkdown, + SigortaTahkimSearchWithinMatch, + SigortaTahkimSearchWithinResult +) + +__all__ = [ + "SigortaTahkimApiClient", + "SigortaTahkimSearchRequest", + "SigortaTahkimDecisionSummary", + "SigortaTahkimSearchResult", + "SigortaTahkimDocumentMarkdown", + "SigortaTahkimSearchWithinMatch", + "SigortaTahkimSearchWithinResult" +] diff --git a/sigorta_tahkim_mcp_module/client.py b/sigorta_tahkim_mcp_module/client.py new file mode 100644 index 0000000..984139f --- /dev/null +++ b/sigorta_tahkim_mcp_module/client.py @@ -0,0 +1,340 @@ +# sigorta_tahkim_mcp_module/client.py + +import httpx +from typing import Optional +import logging +import os +import re +import io +import math +from markitdown import MarkItDown + +from .models import ( + SigortaTahkimSearchRequest, + SigortaTahkimDecisionSummary, + SigortaTahkimSearchResult, + SigortaTahkimDocumentMarkdown, + SigortaTahkimSearchWithinMatch, + SigortaTahkimSearchWithinResult +) + +logger = logging.getLogger(__name__) +if not logger.hasHandlers(): + logging.basicConfig( + level=logging.INFO, + format='%(asctime)s - %(name)s - %(levelname)s - %(message)s' + ) + + +# Turkish-specific lowercase: İ→i, I→ı (Python's str.lower() doesn't handle these) +_TR_UPPER = str.maketrans("İIÇĞÖŞÜ", "iıçğöşü") + + +def _turkish_lower(text: str) -> str: + """Lowercase with Turkish İ/I handling.""" + return text.translate(_TR_UPPER).lower() + + +class SigortaTahkimApiClient: + """ + API client for searching and retrieving Sigorta Tahkim Komisyonu + (Insurance Arbitration Commission) decisions using Tavily Search API + for discovery and direct PDF download for content retrieval. + + The commission publishes quarterly PDF journals ("Hakem Karar Dergisi") + containing arbitration decisions. There are 64 issues spanning 2010-2025. + """ + + TAVILY_API_URL = "https://api.tavily.com/search" + BASE_URL = "https://www.sigortatahkim.org" + PDF_BASE_URL = "https://www.sigortatahkim.org/content/CmsFiles/" + DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000 + + def __init__(self, request_timeout: float = 60.0): + """Initialize the Sigorta Tahkim API client.""" + self.tavily_api_key = os.getenv("TAVILY_API_KEY") + if not self.tavily_api_key: + self.tavily_api_key = "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" + logger.info("Using fallback Tavily API token (development token)") + else: + logger.info("Using Tavily API key from environment variable") + + self.http_client = httpx.AsyncClient( + headers={ + "User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36" + }, + timeout=httpx.Timeout(request_timeout) + ) + self.markitdown = MarkItDown() + + async def close_client_session(self): + """Close the HTTP client session.""" + await self.http_client.aclose() + logger.info("SigortaTahkimApiClient: HTTP client session closed.") + + def _get_pdf_filename(self, issue_number: int) -> str: + """Get the PDF filename for a given journal issue number.""" + if issue_number == 4: + return "karardergisisayi4.pdf" + elif 57 <= issue_number <= 61: + return f"revizekd{issue_number}.pdf" + else: + return f"karardrgs{issue_number}.pdf" + + def _extract_issue_number(self, url: str) -> Optional[str]: + """Extract journal issue number from a sigortatahkim.org URL.""" + # Pattern: karardrgs{N}.pdf + match = re.search(r'karardrgs(\d+)\.pdf', url, re.IGNORECASE) + if match: + return match.group(1) + + # Pattern: revizekd{N}.pdf + match = re.search(r'revizekd(\d+)\.pdf', url, re.IGNORECASE) + if match: + return match.group(1) + + # Pattern: karardergisisayi{N}.pdf + match = re.search(r'karardergisisayi(\d+)\.pdf', url, re.IGNORECASE) + if match: + return match.group(1) + + # Pattern: sayı or sayi in URL path with number + match = re.search(r'say[ıi]\s*[-:]?\s*(\d+)', url, re.IGNORECASE) + if match: + return match.group(1) + + return None + + async def search_decisions( + self, + request: SigortaTahkimSearchRequest + ) -> SigortaTahkimSearchResult: + """ + Search for Sigorta Tahkim Komisyonu decisions using Tavily API. + + Args: + request: Search request parameters + + Returns: + SigortaTahkimSearchResult with matching decisions + """ + try: + headers = { + "Content-Type": "application/json", + "Authorization": f"Bearer {self.tavily_api_key}" + } + + payload = { + "query": request.keywords, + "country": "turkey", + "include_domains": ["sigortatahkim.org"], + "max_results": request.pageSize, + "search_depth": "advanced" + } + + if request.page > 1: + logger.warning(f"Tavily API doesn't support pagination. Page {request.page} requested.") + + response = await self.http_client.post( + self.TAVILY_API_URL, + json=payload, + headers=headers + ) + response.raise_for_status() + + data = response.json() + logger.info(f"Tavily returned {len(data.get('results', []))} results for Sigorta Tahkim") + + decisions = [] + for result in data.get("results", []): + url = result.get("url", "") + title = result.get("title", "").strip() + content = result.get("content", "")[:500] + + issue_num = self._extract_issue_number(url) + doc_id = issue_num if issue_num else url + + decision = SigortaTahkimDecisionSummary( + title=title, + document_id=doc_id, + content=content, + url=url + ) + decisions.append(decision) + + return SigortaTahkimSearchResult( + decisions=decisions, + total_results=len(data.get("results", [])), + page=request.page, + pageSize=request.pageSize + ) + + except httpx.HTTPStatusError as e: + logger.error(f"HTTP error searching Sigorta Tahkim decisions: {e}") + if e.response.status_code == 401: + raise Exception("Tavily API authentication failed. Check API key.") + raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}") + except Exception as e: + logger.error(f"Error searching Sigorta Tahkim decisions: {e}") + raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}") + + # Regex pattern to split decisions within a journal issue + DECISION_HEADER_PATTERN = re.compile( + r'(\d{2}\.\d{2}\.\d{4}\s+Tarih\s+ve\s+K-\d{4}/\d+\s+Sayılı\s+Hakem\s+Kararı)' + ) + # Minimum body length to distinguish real decisions from TOC entries + MIN_DECISION_BODY_LENGTH = 1000 + + async def _download_and_convert_pdf(self, issue_number: str) -> tuple[str, str]: + """ + Download a journal issue PDF and convert to markdown. + + Returns: + Tuple of (markdown_content, pdf_url) + """ + issue_num = int(issue_number) + filename = self._get_pdf_filename(issue_num) + pdf_url = f"{self.PDF_BASE_URL}{filename}" + + logger.info(f"Downloading Sigorta Tahkim PDF: {pdf_url}") + + response = await self.http_client.get(pdf_url, follow_redirects=True) + response.raise_for_status() + + pdf_stream = io.BytesIO(response.content) + result = self.markitdown.convert_stream(pdf_stream, file_extension=".pdf") + return result.text_content.strip(), pdf_url + + def _split_into_decisions(self, markdown_content: str) -> list[tuple[str, str]]: + """ + Split markdown content into individual decisions. + + Returns: + List of (header, body) tuples for decisions with substantial content. + """ + parts = self.DECISION_HEADER_PATTERN.split(markdown_content) + decisions = [] + for i in range(1, len(parts) - 1, 2): + header = parts[i].strip() + body = parts[i + 1].strip() if i + 1 < len(parts) else "" + if len(body) >= self.MIN_DECISION_BODY_LENGTH: + decisions.append((header, body)) + return decisions + + async def get_document_markdown( + self, + issue_number: str, + page_number: int = 1 + ) -> SigortaTahkimDocumentMarkdown: + """ + Retrieve a Sigorta Tahkim journal issue PDF and convert to Markdown. + + Args: + issue_number: Journal issue number (e.g., '64') + page_number: Page number for paginated content (1-indexed) + + Returns: + SigortaTahkimDocumentMarkdown with paginated content + """ + try: + markdown_content, pdf_url = await self._download_and_convert_pdf(issue_number) + + total_length = len(markdown_content) + total_pages = max(1, math.ceil(total_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)) + + start_idx = (page_number - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE + end_idx = start_idx + self.DOCUMENT_MARKDOWN_CHUNK_SIZE + page_content = markdown_content[start_idx:end_idx] + + return SigortaTahkimDocumentMarkdown( + document_id=issue_number, + markdown_content=page_content, + page_number=page_number, + total_pages=total_pages, + source_url=pdf_url + ) + + except ValueError: + raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').") + except httpx.HTTPStatusError as e: + logger.error(f"HTTP error fetching Sigorta Tahkim issue {issue_number}: {e}") + raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}") + except Exception as e: + logger.error(f"Error processing Sigorta Tahkim issue {issue_number}: {e}") + raise Exception(f"Failed to process journal issue {issue_number}: {str(e)}") + + async def search_within_issue( + self, + issue_number: str, + keyword: str, + max_results: int = 10 + ) -> SigortaTahkimSearchWithinResult: + """ + Search for a keyword within a specific journal issue's decisions. + + Downloads the PDF, splits into individual decisions, and returns + matching decisions sorted by relevance (match count). + + Args: + issue_number: Journal issue number (e.g., '64') + keyword: Search keyword or phrase in Turkish + max_results: Maximum matching decisions to return + + Returns: + SigortaTahkimSearchWithinResult with matching decisions + """ + try: + markdown_content, _ = await self._download_and_convert_pdf(issue_number) + decisions = self._split_into_decisions(markdown_content) + + logger.info( + f"Searching '{keyword}' within issue {issue_number}: " + f"{len(decisions)} decisions found" + ) + + keyword_lower = _turkish_lower(keyword) + matches = [] + + for header, body in decisions: + body_lower = _turkish_lower(body) + count = body_lower.count(keyword_lower) + if count == 0: + continue + + # Extract excerpt around the first match + first_pos = body_lower.find(keyword_lower) + excerpt_start = max(0, first_pos - 200) + excerpt_end = min(len(body), first_pos + len(keyword) + 200) + excerpt = body[excerpt_start:excerpt_end].strip() + if excerpt_start > 0: + excerpt = "..." + excerpt + if excerpt_end < len(body): + excerpt = excerpt + "..." + + matches.append(SigortaTahkimSearchWithinMatch( + decision_header=header, + relevance_score=count, + excerpt=excerpt, + body_length=len(body) + )) + + # Sort by relevance (highest match count first) + matches.sort(key=lambda m: m.relevance_score, reverse=True) + matches = matches[:max_results] + + return SigortaTahkimSearchWithinResult( + issue_number=issue_number, + keyword=keyword, + total_decisions=len(decisions), + matching_decisions=len(matches), + matches=matches + ) + + except ValueError: + raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').") + except httpx.HTTPStatusError as e: + logger.error(f"HTTP error in search_within issue {issue_number}: {e}") + raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}") + except Exception as e: + logger.error(f"Error in search_within issue {issue_number}: {e}") + raise Exception(f"Failed to search within issue {issue_number}: {str(e)}") diff --git a/sigorta_tahkim_mcp_module/models.py b/sigorta_tahkim_mcp_module/models.py new file mode 100644 index 0000000..ddfc7e6 --- /dev/null +++ b/sigorta_tahkim_mcp_module/models.py @@ -0,0 +1,59 @@ +# sigorta_tahkim_mcp_module/models.py + +from pydantic import BaseModel, Field +from typing import List + + +class SigortaTahkimSearchRequest(BaseModel): + """Request model for searching Sigorta Tahkim Komisyonu decisions via Tavily API.""" + keywords: str = Field(..., description="Search keywords in Turkish") + page: int = Field(1, ge=1, description="Page number (1-indexed)") + pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)") + + +class SigortaTahkimDecisionSummary(BaseModel): + """Summary of a Sigorta Tahkim decision from search results.""" + title: str = Field(..., description="Decision title or journal issue info") + document_id: str = Field(..., description="Journal issue number (e.g., '64')") + content: str = Field(..., description="Decision summary/excerpt") + url: str = Field("", description="Source URL") + + +class SigortaTahkimSearchResult(BaseModel): + """Response model for Sigorta Tahkim decision search results.""" + decisions: List[SigortaTahkimDecisionSummary] = Field( + default_factory=list, + description="List of matching decisions" + ) + total_results: int = Field(0, description="Total number of results") + page: int = Field(1, description="Current page number") + pageSize: int = Field(10, description="Results per page") + + +class SigortaTahkimDocumentMarkdown(BaseModel): + """Sigorta Tahkim journal issue converted to Markdown format.""" + document_id: str = Field(..., description="Journal issue number") + markdown_content: str = Field("", description="Document content in Markdown") + page_number: int = Field(1, description="Current page number") + total_pages: int = Field(1, description="Total number of pages") + source_url: str = Field("", description="PDF source URL") + + +class SigortaTahkimSearchWithinMatch(BaseModel): + """A single matching decision from search within a journal issue.""" + decision_header: str = Field(..., description="Decision header (date and K-number)") + relevance_score: int = Field(0, description="Number of keyword matches") + excerpt: str = Field("", description="Matching excerpt with context") + body_length: int = Field(0, description="Full decision body length in chars") + + +class SigortaTahkimSearchWithinResult(BaseModel): + """Response model for search within a journal issue.""" + issue_number: str = Field(..., description="Journal issue number searched") + keyword: str = Field("", description="Search keyword used") + total_decisions: int = Field(0, description="Total decisions in issue") + matching_decisions: int = Field(0, description="Number of matching decisions") + matches: List[SigortaTahkimSearchWithinMatch] = Field( + default_factory=list, + description="List of matching decisions sorted by relevance" + )