Fix linting issues with ruff --fix

Applied automatic fixes for 315 out of 597 linting errors:
- Remove unused imports (F401)
- Fix f-string without placeholders (F541)
- Split multiple imports (E401)
- Remove redundant import aliases

Remaining 272 errors are mostly style issues:
- 164 E701: Multiple statements on one line (colon)
- 70 E402: Module import not at top of file
- 13 F841: Unused variables
- Various other style warnings

Code functionality unchanged - all fixes are cosmetic improvements.
This commit is contained in:
saidsurucu
2025-09-20 01:13:30 +03:00
parent 6f94eca33c
commit 9b43070754
70 changed files with 1127 additions and 277 deletions
+10 -11
View File
@@ -8,7 +8,6 @@ and trying to reverse engineer the hash generation logic.
import asyncio
import json
import hashlib
import hmac
import base64
from fastmcp import Client
from mcp_server_main import app
@@ -152,7 +151,7 @@ async def test_hash_generation_comprehensive():
# Test with first decision
sample_decision = decisions[0]
print(f"\n📋 Sample decision for hash analysis:")
print("\n📋 Sample decision for hash analysis:")
for key, value in sample_decision.items():
print(f" {key}: {value}")
@@ -162,20 +161,20 @@ async def test_hash_generation_comprehensive():
all_hashes = {}
# Test different hash generation methods
print(f"\n🔨 Testing webpack-style hashing...")
print("\n🔨 Testing webpack-style hashing...")
webpack_hashes = test_webpack_style_hashing(sample_decision)
all_hashes.update(webpack_hashes)
print(f"🔨 Testing Angular routing hashes...")
print("🔨 Testing Angular routing hashes...")
angular_hashes = test_angular_routing_hashes(sample_decision)
all_hashes.update(angular_hashes)
print(f"🔨 Testing base64 encoding variants...")
print("🔨 Testing base64 encoding variants...")
b64_hashes = test_base64_encoding_variants(sample_decision)
all_hashes.update(b64_hashes)
# Check for matches
print(f"\n🎯 Checking for hash matches...")
print("\n🎯 Checking for hash matches...")
matches_found = []
partial_matches = []
@@ -191,13 +190,13 @@ async def test_hash_generation_comprehensive():
print(f" 🔍 Partial match (last 8): {hash_name} -> ...{hash_value[-16:]}")
if not matches_found and not partial_matches:
print(f" ❌ No matches found")
print(f"\n📝 Sample generated hashes (first 10):")
print(" ❌ No matches found")
print("\n📝 Sample generated hashes (first 10):")
for i, (hash_name, hash_value) in enumerate(list(all_hashes.items())[:10]):
print(f" {hash_name}: {hash_value}")
# Try combinations with other decisions
print(f"\n🔄 Testing hash combinations with multiple decisions...")
print("\n🔄 Testing hash combinations with multiple decisions...")
if len(decisions) > 1:
for i, decision in enumerate(decisions[1:3]): # Test 2 more
print(f"\n Testing decision {i+2}: {decision.get('kararNo')}")
@@ -209,7 +208,7 @@ async def test_hash_generation_comprehensive():
matches_found.append((f"decision_{i+2}_{hash_name}", hash_value))
# Try composite hashes (combining multiple fields)
print(f"\n🔗 Testing composite hash generation...")
print("\n🔗 Testing composite hash generation...")
composite_tests = [
f"{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararNo')}",
f"{sample_decision.get('kararNo')}_{sample_decision.get('kararTarihi')}",
@@ -224,7 +223,7 @@ async def test_hash_generation_comprehensive():
print(f" 🎉 COMPOSITE MATCH FOUND: test_{i} -> {composite_str[:50]}...")
matches_found.append((f"composite_{i}", composite_hash))
print(f"\n🎯 Hash analysis completed!")
print("\n🎯 Hash analysis completed!")
print(f" Total matches found: {len(matches_found)}")
print(f" Partial matches: {len(partial_matches)}")
+3 -3
View File
@@ -2,13 +2,13 @@
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
import httpx
from bs4 import BeautifulSoup, Tag
from typing import Dict, Any, List, Optional, Tuple
from bs4 import BeautifulSoup
from typing import List, Optional, Tuple
import logging
import html
import re
import io
from urllib.parse import urlencode, urljoin, quote
from urllib.parse import urljoin
from markitdown import MarkItDown
import math # For math.ceil for pagination
+2 -2
View File
@@ -3,12 +3,12 @@
import httpx
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Tuple
from typing import List, Optional, Tuple
import logging
import html
import re
import io
from urllib.parse import urlencode, urljoin, quote
from urllib.parse import urljoin
from markitdown import MarkItDown
import math # For math.ceil for pagination
-1
View File
@@ -2,7 +2,6 @@
# Unified client for both Norm Denetimi and Bireysel Başvuru
import logging
from typing import Optional
from urllib.parse import urlparse
from .models import (
+8 -11
View File
@@ -10,16 +10,13 @@ Usage:
"""
import os
import time
import logging
import json
from datetime import datetime, timedelta
from fastapi import FastAPI, Request, HTTPException, Query
from fastapi.responses import JSONResponse, HTMLResponse, Response
from fastapi.exception_handlers import http_exception_handler
from starlette.middleware import Middleware
from starlette.middleware.cors import CORSMiddleware
from starlette.middleware.base import BaseHTTPMiddleware
# Import the proper create_app function that includes all middleware
from mcp_server_main import create_app
@@ -505,14 +502,14 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
logger.info("User authenticated successfully via Clerk")
# Return success response
return HTMLResponse(f"""
return HTMLResponse("""
<html>
<head>
<title>MCP Connection Successful</title>
<style>
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
.success {{ color: #28a745; }}
.token {{ background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }}
body { font-family: Arial, sans-serif; text-align: center; padding: 50px; }
.success { color: #28a745; }
.token { background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }
</style>
</head>
<body>
@@ -525,13 +522,13 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
<p>You can now close this window and return to your MCP client.</p>
<script>
// Try to close the popup if opened as such
if (window.opener) {{
window.opener.postMessage({{
if (window.opener) {
window.opener.postMessage({
type: 'MCP_AUTH_SUCCESS',
token: 'use_clerk_jwt_token'
}}, '*');
}, '*');
setTimeout(() => window.close(), 3000);
}}
}
</script>
</body>
</html>
+1 -2
View File
@@ -1,13 +1,12 @@
# bddk_mcp_module/client.py
import httpx
from typing import List, Optional, Dict, Any
from typing import Optional
import logging
import os
import re
import io
import math
from urllib.parse import urlparse
from markitdown import MarkItDown
from .models import (
+1 -1
View File
@@ -1,7 +1,7 @@
# bddk_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import List, Optional
from typing import List
class BddkSearchRequest(BaseModel):
"""
+1 -2
View File
@@ -1,8 +1,7 @@
# bedesten_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import List, Optional, Dict, Any, Literal, Union
from datetime import datetime
from typing import List, Optional, Dict, Any, Literal
# Import compressed BirimAdiEnum for chamber filtering
from .enums import BirimAdiEnum
+1 -3
View File
@@ -1,11 +1,9 @@
# danistay_mcp_module/client.py
import httpx
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional
from typing import Dict, List, Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
+1 -2
View File
@@ -2,10 +2,9 @@
import httpx
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
from typing import Dict, Any, List, Optional
from typing import Dict, Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
+1 -3
View File
@@ -7,11 +7,9 @@ import os
from typing import List, Dict, Any, Optional
from datetime import datetime
from fastapi import FastAPI, HTTPException, Query, Depends, Body
from fastapi import FastAPI, HTTPException, Query
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import JSONResponse
from pydantic import BaseModel, Field
import json
# Import the main MCP app
from mcp_server_main import app as mcp_server
+3 -4
View File
@@ -10,13 +10,12 @@ from playwright.async_api import (
)
from bs4 import BeautifulSoup
import logging
from typing import Dict, Any, List, Optional
from typing import List, Optional
import urllib.parse
import base64 # Base64 için
import re
import html as html_parser
from markitdown import MarkItDown
import os
import math
import io
import random
@@ -907,7 +906,7 @@ class KikApiClient:
event_target_for_submit = self.FIELD_LOCATORS['search_button_id']
# Use human-like clicking for search button
search_button_selector = f"a[id='{event_target_for_submit}']"
logger.info(f"Performing human-like search button click...")
logger.info("Performing human-like search button click...")
try:
# Hide datepicker first to prevent interference
@@ -1181,7 +1180,7 @@ class KikApiClient:
try:
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=2000):
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
await current_main_page.wait_for_selector(f"div#detayPopUp:not(.in)", timeout=5000)
await current_main_page.wait_for_selector("div#detayPopUp:not(.in)", timeout=5000)
except: pass
return KikDocumentMarkdown(
-4
View File
@@ -1,13 +1,9 @@
# kik_mcp_module/client_v2.py
import httpx
import requests
import logging
import uuid
import base64
import ssl
from typing import Optional
from datetime import datetime
from .models_v2 import (
KikV2DecisionType, KikV2SearchPayload, KikV2SearchPayloadDk, KikV2SearchPayloadMk,
+1 -1
View File
@@ -1,5 +1,5 @@
# kik_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
from pydantic import BaseModel, Field, computed_field, ConfigDict
from typing import List, Optional
from enum import Enum
import base64 # Base64 encoding/decoding için
+1 -2
View File
@@ -1,7 +1,6 @@
# kik_mcp_module/models_v2.py
from pydantic import BaseModel, Field, ConfigDict
from typing import List, Optional
from datetime import datetime
from typing import List
from enum import Enum
# New KIK v2 API Models
+2 -2
View File
@@ -2,13 +2,13 @@
import httpx
from bs4 import BeautifulSoup
from typing import List, Optional, Dict, Any
from typing import Optional, Dict, Any
import logging
import os
import re
import io
import math
from urllib.parse import urljoin, urlparse, parse_qs
from urllib.parse import urlparse
from markitdown import MarkItDown
from pydantic import HttpUrl
+1 -1
View File
@@ -1,7 +1,7 @@
# kvkk_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl
from typing import List, Optional, Any
from typing import List, Optional
class KvkkSearchRequest(BaseModel):
"""Model for KVKK (Personal Data Protection Authority) search request via Brave API."""
+1 -1
View File
@@ -36,7 +36,7 @@ def create_clerk_oauth_config() -> OAuthConfig:
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
)
logger.info(f"Created Clerk OAuth config with adapter endpoints")
logger.info("Created Clerk OAuth config with adapter endpoints")
logger.info(f"Clerk domain: {clerk_domain}")
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
logger.debug(f"Token endpoint: {config.token_endpoint}")
+1 -1
View File
@@ -9,7 +9,7 @@ import time
import logging
from dataclasses import dataclass
from datetime import datetime, timedelta
from typing import Any, Optional
from typing import Any
from urllib.parse import urlencode
import httpx
+1 -1
View File
@@ -18,7 +18,7 @@ from fastapi.responses import RedirectResponse, JSONResponse
try:
from clerk_backend_api import Clerk
CLERK_AVAILABLE = True
except ImportError as e:
except ImportError:
CLERK_AVAILABLE = False
Clerk = None
+3 -5
View File
@@ -6,7 +6,7 @@ Uses Redis for authorization code storage to support multi-machine deployment
import os
import logging
from typing import Optional
from urllib.parse import urlencode, quote
from urllib.parse import urlencode
from fastapi import APIRouter, Request, Query, HTTPException
from fastapi.responses import RedirectResponse, JSONResponse
@@ -39,7 +39,6 @@ def get_redis_session_store():
if redis_store is None:
try:
import concurrent.futures
import functools
# Use thread pool with timeout to prevent hanging
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
@@ -206,7 +205,6 @@ async def oauth_callback(
auth_code = f"clerk_auth_{os.urandom(16).hex()}"
# Prepare code data
import time
code_data = {
"user_id": user_id,
"session_id": session_id,
@@ -225,7 +223,7 @@ async def oauth_callback(
if success:
logger.info(f"Stored authorization code {auth_code[:10]}... in Redis with real JWT token")
else:
logger.error(f"Failed to store authorization code in Redis, falling back to in-memory")
logger.error("Failed to store authorization code in Redis, falling back to in-memory")
# Fall back to in-memory storage
if not hasattr(oauth_callback, '_code_storage'):
oauth_callback._code_storage = {}
@@ -236,7 +234,7 @@ async def oauth_callback(
if not hasattr(oauth_callback, '_code_storage'):
oauth_callback._code_storage = {}
oauth_callback._code_storage[auth_code] = code_data
logger.info(f"Stored authorization code in memory (fallback)")
logger.info("Stored authorization code in memory (fallback)")
# Redirect back to client with authorization code
redirect_params = {
+261 -57
View File
@@ -8,8 +8,7 @@ import json
import time
from collections import defaultdict
from pydantic import HttpUrl, Field
from typing import Optional, Dict, List, Literal, Any, Union
import urllib.parse
from typing import Dict, List, Literal, Any
import tiktoken
from fastmcp.server.middleware import Middleware, MiddlewareContext
from fastmcp.server.dependencies import get_access_token, AccessToken
@@ -159,7 +158,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("tool_call_error", input_tokens, 0,
tool_name, duration_ms)
@@ -189,7 +188,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("resource_read_error", 0, 0,
resource_uri, duration_ms)
@@ -219,7 +218,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("prompt_get_error", 0, 0,
prompt_name, duration_ms)
@@ -255,21 +254,18 @@ def create_app(auth=None):
# --- Module Imports ---
from yargitay_mcp_module.client import YargitayOfficialApiClient
from yargitay_mcp_module.models import (
YargitayDetailedSearchRequest, YargitayDocumentMarkdown, CompactYargitaySearchResult,
YargitayBirimEnum, CleanYargitayDecisionEntry
)
from bedesten_mcp_module.client import BedestenApiClient
from bedesten_mcp_module.models import (
BedestenSearchRequest, BedestenSearchData,
BedestenDocumentMarkdown, BedestenCourtTypeEnum
)
from bedesten_mcp_module.enums import BirimAdiEnum
# Semantic Search Module Imports
from semantic_search.embedder import EmbeddingGemma
from semantic_search.vector_store import VectorStore
from semantic_search.processor import DocumentProcessor
from danistay_mcp_module.client import DanistayApiClient
from danistay_mcp_module.models import (
DanistayKeywordSearchRequest, DanistayDetailedSearchRequest,
DanistayDocumentMarkdown, CompactDanistaySearchResult
)
from emsal_mcp_module.client import EmsalApiClient
from emsal_mcp_module.models import (
EmsalSearchRequest, EmsalDocumentMarkdown, CompactEmsalSearchResult
@@ -283,24 +279,12 @@ from anayasa_mcp_module.client import AnayasaMahkemesiApiClient
from anayasa_mcp_module.bireysel_client import AnayasaBireyselBasvuruApiClient
from anayasa_mcp_module.unified_client import AnayasaUnifiedClient
from anayasa_mcp_module.models import (
AnayasaNormDenetimiSearchRequest,
AnayasaSearchResult,
AnayasaDocumentMarkdown,
AnayasaBireyselReportSearchRequest,
AnayasaBireyselReportSearchResult,
AnayasaBireyselBasvuruDocumentMarkdown,
AnayasaUnifiedSearchRequest,
AnayasaUnifiedSearchResult,
AnayasaUnifiedDocumentMarkdown,
# Removed enum imports - now using Literal strings in models
)
# KIK v2 Module Imports (New API)
from kik_mcp_module.client_v2 import KikV2ApiClient
from kik_mcp_module.models_v2 import KikV2DecisionType
from kik_mcp_module.models_v2 import (
KikV2SearchResult,
KikV2DocumentMarkdown
)
from rekabet_mcp_module.client import RekabetKurumuApiClient
from rekabet_mcp_module.models import (
@@ -312,14 +296,9 @@ from rekabet_mcp_module.models import (
from sayistay_mcp_module.client import SayistayApiClient
from sayistay_mcp_module.models import (
GenelKurulSearchRequest, GenelKurulSearchResponse,
TemyizKuruluSearchRequest, TemyizKuruluSearchResponse,
DaireSearchRequest, DaireSearchResponse,
SayistayDocumentMarkdown,
SayistayUnifiedSearchRequest, SayistayUnifiedSearchResult,
SayistayUnifiedDocumentMarkdown
)
from sayistay_mcp_module.enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
from sayistay_mcp_module.unified_client import SayistayUnifiedClient
# KVKK Module Imports
@@ -333,14 +312,11 @@ from kvkk_mcp_module.models import (
# BDDK Module Imports
from bddk_mcp_module.client import BddkApiClient
from bddk_mcp_module.models import (
BddkSearchRequest,
BddkSearchResult,
BddkDocumentMarkdown
BddkSearchRequest
)
# Create a placeholder app that will be properly initialized after tools are defined
from fastmcp import FastMCP
# MCP app for Turkish legal databases with explicit capabilities
app = FastMCP(
@@ -647,7 +623,7 @@ async def search_emsal_detailed_decisions(
page_size=page_size
)
logger.info(f"Tool 'search_emsal_detailed_decisions' called.")
logger.info("Tool 'search_emsal_detailed_decisions' called.")
try:
api_response = await emsal_client_instance.search_detailed_decisions(search_query)
if api_response.data:
@@ -659,8 +635,8 @@ async def search_emsal_detailed_decisions(
)
logger.warning("API response for Emsal search did not contain expected data structure.")
return CompactEmsalSearchResult(decisions=[], total_records=0, requested_page=search_query.page_number, page_size=search_query.page_size)
except Exception as e:
logger.exception(f"Error in tool 'search_emsal_detailed_decisions'.")
except Exception:
logger.exception("Error in tool 'search_emsal_detailed_decisions'.")
raise
@app.tool(
@@ -676,8 +652,8 @@ async def get_emsal_document_markdown(id: str) -> EmsalDocumentMarkdown:
if not id or not id.strip(): raise ValueError("Document ID required for Emsal.")
try:
return await emsal_client_instance.get_decision_document_as_markdown(id)
except Exception as e:
logger.exception(f"Error in tool 'get_emsal_document_markdown'.")
except Exception:
logger.exception("Error in tool 'get_emsal_document_markdown'.")
raise
# --- MCP Tools for Uyusmazlik ---
@@ -745,11 +721,11 @@ async def search_uyusmazlik_decisions(
not_hepsi=not_hepsi
)
logger.info(f"Tool 'search_uyusmazlik_decisions' called.")
logger.info("Tool 'search_uyusmazlik_decisions' called.")
try:
return await uyusmazlik_client_instance.search_decisions(search_params)
except Exception as e:
logger.exception(f"Error in tool 'search_uyusmazlik_decisions'.")
except Exception:
logger.exception("Error in tool 'search_uyusmazlik_decisions'.")
raise
@app.tool(
@@ -768,8 +744,8 @@ async def get_uyusmazlik_document_markdown_from_url(
raise ValueError("Document URL (document_url) is required for Uyuşmazlık document retrieval.")
try:
return await uyusmazlik_client_instance.get_decision_document_as_markdown(str(document_url))
except Exception as e:
logger.exception(f"Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
except Exception:
logger.exception("Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
raise
# --- DEACTIVATED: MCP Tools for Anayasa Mahkemesi (Individual Tools) ---
@@ -860,8 +836,8 @@ async def search_anayasa_unified(
result = await anayasa_unified_client_instance.search_unified(request)
return json.dumps(result.model_dump(), ensure_ascii=False, indent=2)
except Exception as e:
logger.exception(f"Error in tool 'search_anayasa_unified'.")
except Exception:
logger.exception("Error in tool 'search_anayasa_unified'.")
raise
@app.tool(
@@ -882,8 +858,8 @@ async def get_anayasa_document_unified(
result = await anayasa_unified_client_instance.get_document_unified(document_url, page_number)
return json.dumps(result.model_dump(mode='json'), ensure_ascii=False, indent=2)
except Exception as e:
logger.exception(f"Error in tool 'get_anayasa_document_unified'.")
except Exception:
logger.exception("Error in tool 'get_anayasa_document_unified'.")
raise
# --- MCP Tools for KIK v2 (Kamu İhale Kurulu - New API) ---
@@ -1057,7 +1033,7 @@ async def search_rekabet_kurumu_decisions(
try:
return await rekabet_client_instance.search_decisions(search_query)
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_rekabet_kurumu_decisions'.")
return RekabetSearchResult(decisions=[], retrieved_page_number=page, total_records_found=0, total_pages=0)
@@ -1080,7 +1056,7 @@ async def get_rekabet_kurumu_document(
try:
return await rekabet_client_instance.get_decision_document(karar_id, page_number=current_page_to_fetch)
except Exception as e:
except Exception:
logger.exception(f"Error in tool 'get_rekabet_kurumu_document'. Karar ID: {karar_id}")
raise
@@ -1191,7 +1167,7 @@ For best results, use exact phrases with quotes for legal terms."""),
"page_size": pageSize,
"searched_courts": court_types
}
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_bedesten_unified'")
raise
@@ -1213,7 +1189,7 @@ async def get_bedesten_document_markdown(
try:
return await bedesten_client_instance.get_document_as_markdown(documentId)
except Exception as e:
except Exception:
logger.exception("Error in tool 'get_kyb_bedesten_document_markdown'")
raise
@@ -1402,7 +1378,7 @@ async def search_sayistay_unified(
web_karar_metni=web_karar_metni
)
return await sayistay_unified_client_instance.search_unified(search_request)
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_sayistay_unified'")
raise
@@ -1426,7 +1402,7 @@ async def get_sayistay_document_unified(
try:
return await sayistay_unified_client_instance.get_document_unified(decision_id, decision_type)
except Exception as e:
except Exception:
logger.exception("Error in tool 'get_sayistay_document_unified'")
raise
@@ -1839,6 +1815,234 @@ async def get_bddk_document_markdown(
"error": str(e)
}
# --- Semantic Search Tool ---
@app.tool(
description="Semantic search for Turkish legal decisions using EmbeddingGemma for intelligent ranking",
annotations={
"readOnlyHint": True,
"openWorldHint": True,
"idempotentHint": True
}
)
async def search_bedesten_semantic(
query: str = Field(..., description="Search query in Turkish for semantic matching"),
initial_keyword: str = Field(..., description="Initial keyword for Bedesten API search (broad term)"),
court_types: List[BedestenCourtTypeEnum] = Field(
default=["YARGITAYKARARI", "DANISTAYKARAR", "YERELHUKUK", "ISTINAFHUKUK", "KYB"],
description="Court types to search: YARGITAYKARARI, DANISTAYKARAR, YERELHUKUK, ISTINAFHUKUK, KYB (default: all)"
),
top_k: int = Field(10, ge=1, le=50, description="Number of top results to return (1-50)")
) -> Dict[str, Any]:
"""
Perform semantic search on Turkish legal decisions using EmbeddingGemma.
This tool:
1. Searches Bedesten API with initial keyword (retrieves 100 results)
2. Fetches full document content for each result
3. Generates embeddings using Google's EmbeddingGemma model
4. Performs semantic similarity search with the query
5. Returns re-ranked results based on semantic relevance
Benefits over keyword search:
- Better understanding of context and meaning
- Finds semantically similar documents even with different wording
- More accurate ranking based on relevance
- Supports multilingual queries (100+ languages)
"""
logger.info(f"Semantic search tool called with query: {query}, keyword: {initial_keyword}")
try:
# Initialize components
embedder = EmbeddingGemma()
vector_store = VectorStore(dimension=256) # Always use 256 for optimal speed/quality balance
processor = DocumentProcessor(chunk_size=1500, chunk_overlap=300)
# Step 1: Initial keyword search to get document IDs
logger.info(f"Step 1: Searching Bedesten API with keyword: {initial_keyword}")
all_decisions = []
# Search each court type
for court_type in court_types:
try:
# Calculate page size per court type to get 100 total
per_court_limit = max(20, 100 // len(court_types))
search_results = await bedesten_client_instance.search_documents(
BedestenSearchRequest(
data=BedestenSearchData(
phrase=initial_keyword,
itemTypeList=[court_type],
pageSize=per_court_limit, # Distribute 100 across court types
pageNumber=1
)
)
)
if search_results.data and search_results.data.emsalKararList:
all_decisions.extend(search_results.data.emsalKararList)
logger.info(f"Found {len(search_results.data.emsalKararList)} results from {court_type}")
except Exception as e:
logger.warning(f"Error searching {court_type}: {e}")
if not all_decisions:
logger.warning("No documents found from initial search")
return {
"status": "no_results",
"message": "No documents found matching the initial keyword",
"results": []
}
logger.info(f"Total documents found: {len(all_decisions)}")
# Step 2: Fetch document content and process
logger.info("Step 2: Fetching and processing document content...")
documents_data = []
failed_fetches = 0
# Process up to 100 documents total
decisions_to_process = all_decisions[:100]
for i, decision in enumerate(decisions_to_process):
try:
# Fetch document content
doc = await bedesten_client_instance.get_document_as_markdown(decision.documentId)
if doc.markdown_content:
# Process document into chunks
metadata = {
"document_id": decision.documentId,
"birim_adi": decision.birimAdi,
"esas_no": decision.esasNo,
"karar_no": decision.kararNo,
"karar_tarihi": decision.kararTarihiStr,
"court_type": decision.itemType.name if decision.itemType else None
}
chunks = processor.process_document(
document_id=decision.documentId,
text=doc.markdown_content,
metadata=metadata
)
# For now, use the full document as one chunk (can be optimized later)
if chunks:
full_text = " ".join([chunk.text for chunk in chunks])
documents_data.append({
"id": decision.documentId,
"text": full_text[:3000], # Limit text for embedding
"metadata": metadata
})
# Log progress every 10 documents
if (i + 1) % 10 == 0:
logger.info(f"Processed {i + 1}/{len(decisions_to_process)} documents")
except Exception as e:
logger.warning(f"Failed to fetch document {decision.documentId}: {e}")
failed_fetches += 1
if not documents_data:
logger.warning("No documents could be processed")
return {
"status": "processing_error",
"message": "Could not process any documents",
"results": []
}
logger.info(f"Successfully processed {len(documents_data)} documents, {failed_fetches} failed")
# Step 3: Generate embeddings
logger.info("Step 3: Generating embeddings...")
# Generate query embedding
query_embedding = embedder.encode_query(query, task="search result")
# Generate document embeddings
doc_texts = [doc["text"] for doc in documents_data]
doc_titles = [doc["metadata"].get("birim_adi", "none") for doc in documents_data]
doc_embeddings = embedder.encode_documents(doc_texts, titles=doc_titles)
# Always reduce to 256 dimensions for optimal speed/quality balance
query_embedding = embedder.reduce_dimensions(query_embedding, 256)
doc_embeddings = embedder.reduce_dimensions(doc_embeddings, 256)
# Step 4: Add to vector store and search
logger.info("Step 4: Performing semantic search...")
# Add documents to vector store
doc_ids = [doc["id"] for doc in documents_data]
doc_metadatas = [doc["metadata"] for doc in documents_data]
vector_store.add_documents(
ids=doc_ids,
texts=doc_texts,
embeddings=doc_embeddings,
metadata=doc_metadatas
)
# Perform semantic search
search_results = vector_store.search(
query_embedding=query_embedding,
top_k=top_k,
threshold=0.3 # Minimum similarity threshold
)
# Step 5: Format results
logger.info(f"Step 5: Formatting {len(search_results)} results")
formatted_results = []
for doc, score in search_results:
# Build title from metadata
title_parts = []
if doc.metadata.get("birim_adi"):
title_parts.append(doc.metadata["birim_adi"])
if doc.metadata.get("esas_no"):
title_parts.append(f"Esas: {doc.metadata['esas_no']}")
if doc.metadata.get("karar_no"):
title_parts.append(f"Karar: {doc.metadata['karar_no']}")
if doc.metadata.get("karar_tarihi"):
title_parts.append(f"Tarih: {doc.metadata['karar_tarihi']}")
title = " - ".join(title_parts) if title_parts else f"Document {doc.id}"
formatted_results.append({
"document_id": doc.id,
"title": title,
"similarity_score": float(score),
"preview": doc.text[:500] + "..." if len(doc.text) > 500 else doc.text,
"metadata": doc.metadata,
"source_url": f"https://mevzuat.adalet.gov.tr/ictihat/{doc.id}"
})
# Get vector store stats
stats = vector_store.get_stats()
return {
"status": "success",
"query": query,
"initial_keyword": initial_keyword,
"total_documents_processed": len(documents_data),
"embedding_dimension": 256, # Fixed at 256 for optimal performance
"results": formatted_results,
"stats": {
"documents_in_store": stats["num_documents"],
"memory_usage_mb": round(stats["memory_usage_mb"], 2),
"failed_fetches": failed_fetches
}
}
except Exception as e:
logger.exception(f"Error in semantic search: {e}")
return {
"status": "error",
"message": str(e),
"results": []
}
# --- ChatGPT Deep Research Compatible Tools ---
def get_preview_text(markdown_content: str, skip_chars: int = 100, preview_chars: int = 200) -> str:
@@ -2012,7 +2216,7 @@ async def search(
logger.info(f"ChatGPT Deep Research search completed. Found {len(results)} results via Bedesten API.")
return {"results": results}
except Exception as e:
except Exception:
logger.exception("Error in ChatGPT Deep Research search tool")
# Return partial results if any were found
if results:
@@ -2141,7 +2345,7 @@ async def fetch(
doc = await bedesten_client_instance.get_document_as_markdown(doc_id)
"""
except Exception as e:
except Exception:
logger.exception(f"Error fetching ChatGPT Deep Research document {id}")
raise
@@ -2192,7 +2396,7 @@ def main():
app.run()
except KeyboardInterrupt:
logger.info("Server shut down by user (KeyboardInterrupt).")
except Exception as e:
except Exception:
logger.exception("Server failed to start or crashed.")
finally:
logger.info(f"{app.name} server has shut down.")
+5
View File
@@ -31,6 +31,11 @@ dependencies = [
"fastapi>=0.115.14",
"PyJWT>=2.8.0",
"tiktoken>=0.5.0",
"sentence-transformers>=3.0.0",
"torch>=2.0.0",
"numpy>=1.24.0",
"scikit-learn>=1.3.0",
"transformers>=4.57.0.dev0",
]
[project.optional-dependencies]
+2 -2
View File
@@ -11,8 +11,8 @@ import os
import json
import time
import logging
from typing import Optional, Dict, Any, Union
from datetime import datetime, timedelta
from typing import Optional, Dict, Any
from datetime import datetime
logger = logging.getLogger(__name__)
+2 -4
View File
@@ -4,10 +4,9 @@ import httpx
from bs4 import BeautifulSoup
from typing import List, Optional, Tuple, Dict, Any
import logging
import html
import re
import io # For io.BytesIO
from urllib.parse import urlencode, urljoin, quote, parse_qs, urlparse
from urllib.parse import urljoin, parse_qs, urlparse
from markitdown import MarkItDown
import math
@@ -18,8 +17,7 @@ from .models import (
RekabetKurumuSearchRequest,
RekabetDecisionSummary,
RekabetSearchResult,
RekabetDocument,
RekabetKararTuruGuidEnum
RekabetDocument
)
from pydantic import HttpUrl # Ensure HttpUrl is imported from pydantic
+1 -1
View File
@@ -1,7 +1,7 @@
# rekabet_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl
from typing import List, Optional, Any
from typing import List, Optional
from enum import Enum
# Enum for decision type GUIDs (used by the client and expected by the website)
+1 -1
View File
@@ -93,7 +93,7 @@ def main():
config["workers"] = args.workers
# Print startup information
print(f"Starting Yargı MCP server...")
print("Starting Yargı MCP server...")
print(f"Host: {args.host}")
print(f"Port: {args.port}")
print(f"Transport: {args.transport}")
@@ -2,13 +2,13 @@
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
import httpx
from bs4 import BeautifulSoup, Tag
from typing import Dict, Any, List, Optional, Tuple
from bs4 import BeautifulSoup
from typing import List, Optional, Tuple
import logging
import html
import re
import io
from urllib.parse import urlencode, urljoin, quote
from urllib.parse import urljoin
from markitdown import MarkItDown
import math # For math.ceil for pagination
@@ -3,12 +3,12 @@
import httpx
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Tuple
from typing import List, Optional, Tuple
import logging
import html
import re
import io
from urllib.parse import urlencode, urljoin, quote
from urllib.parse import urljoin
from markitdown import MarkItDown
import math # For math.ceil for pagination
@@ -2,7 +2,6 @@
# Unified client for both Norm Denetimi and Bireysel Başvuru
import logging
from typing import Optional
from urllib.parse import urlparse
from .models import (
+8 -13
View File
@@ -10,16 +10,12 @@ Usage:
"""
import os
import time
import logging
from datetime import datetime, timedelta
from fastapi import FastAPI, Request, HTTPException, Query
from fastapi.responses import JSONResponse, HTMLResponse
from fastapi.exception_handlers import http_exception_handler
from starlette.middleware import Middleware
from starlette.middleware.cors import CORSMiddleware
from starlette.responses import Response
from starlette.requests import Request as StarletteRequest
# Import the MCP app creator function
from mcp_server_main import create_app
@@ -103,7 +99,6 @@ mcp_app = mcp_server.http_app(path="/")
# Configure JSON encoder for proper Turkish character support
import json
from fastapi.responses import JSONResponse
class UTF8JSONResponse(JSONResponse):
def __init__(self, content=None, status_code=200, headers=None, **kwargs):
@@ -461,14 +456,14 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
logger.info("User authenticated successfully via Clerk")
# Return success response
return HTMLResponse(f"""
return HTMLResponse("""
<html>
<head>
<title>MCP Connection Successful</title>
<style>
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
.success {{ color: #28a745; }}
.token {{ background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }}
body { font-family: Arial, sans-serif; text-align: center; padding: 50px; }
.success { color: #28a745; }
.token { background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }
</style>
</head>
<body>
@@ -481,13 +476,13 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
<p>You can now close this window and return to your MCP client.</p>
<script>
// Try to close the popup if opened as such
if (window.opener) {{
window.opener.postMessage({{
if (window.opener) {
window.opener.postMessage({
type: 'MCP_AUTH_SUCCESS',
token: 'use_clerk_jwt_token'
}}, '*');
}, '*');
setTimeout(() => window.close(), 3000);
}}
}
</script>
</body>
</html>
@@ -1,13 +1,12 @@
# bddk_mcp_module/client.py
import httpx
from typing import List, Optional, Dict, Any
from typing import Optional
import logging
import os
import re
import io
import math
from urllib.parse import urlparse
from markitdown import MarkItDown
from .models import (
@@ -1,7 +1,7 @@
# bddk_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import List, Optional
from typing import List
class BddkSearchRequest(BaseModel):
"""
@@ -1,8 +1,7 @@
# bedesten_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import List, Optional, Dict, Any, Literal, Union
from datetime import datetime
from typing import List, Optional, Dict, Any, Literal
# Import compressed BirimAdiEnum for chamber filtering
from .enums import BirimAdiEnum
@@ -1,11 +1,9 @@
# danistay_mcp_module/client.py
import httpx
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional
from typing import Dict, List, Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
@@ -2,10 +2,9 @@
import httpx
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
from typing import Dict, Any, List, Optional
from typing import Dict, Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
@@ -7,11 +7,9 @@ import os
from typing import List, Dict, Any, Optional
from datetime import datetime
from fastapi import FastAPI, HTTPException, Query, Depends, Body
from fastapi import FastAPI, HTTPException, Query
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import JSONResponse
from pydantic import BaseModel, Field
import json
# Import the main MCP app
from mcp_server_main import app as mcp_server
@@ -10,13 +10,12 @@ from playwright.async_api import (
)
from bs4 import BeautifulSoup
import logging
from typing import Dict, Any, List, Optional
from typing import List, Optional
import urllib.parse
import base64 # Base64 için
import re
import html as html_parser
from markitdown import MarkItDown
import os
import math
import io
import random
@@ -907,7 +906,7 @@ class KikApiClient:
event_target_for_submit = self.FIELD_LOCATORS['search_button_id']
# Use human-like clicking for search button
search_button_selector = f"a[id='{event_target_for_submit}']"
logger.info(f"Performing human-like search button click...")
logger.info("Performing human-like search button click...")
try:
# Hide datepicker first to prevent interference
@@ -1181,7 +1180,7 @@ class KikApiClient:
try:
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=2000):
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
await current_main_page.wait_for_selector(f"div#detayPopUp:not(.in)", timeout=5000)
await current_main_page.wait_for_selector("div#detayPopUp:not(.in)", timeout=5000)
except: pass
return KikDocumentMarkdown(
@@ -1,5 +1,5 @@
# kik_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
from pydantic import BaseModel, Field, computed_field, ConfigDict
from typing import List, Optional
from enum import Enum
import base64 # Base64 encoding/decoding için
@@ -2,13 +2,13 @@
import httpx
from bs4 import BeautifulSoup
from typing import List, Optional, Dict, Any
from typing import Optional, Dict, Any
import logging
import os
import re
import io
import math
from urllib.parse import urljoin, urlparse, parse_qs
from urllib.parse import urlparse
from markitdown import MarkItDown
from pydantic import HttpUrl
@@ -1,7 +1,7 @@
# kvkk_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl
from typing import List, Optional, Any
from typing import List, Optional
class KvkkSearchRequest(BaseModel):
"""Model for KVKK (Personal Data Protection Authority) search request via Brave API."""
@@ -36,7 +36,7 @@ def create_clerk_oauth_config() -> OAuthConfig:
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
)
logger.info(f"Created Clerk OAuth config with adapter endpoints")
logger.info("Created Clerk OAuth config with adapter endpoints")
logger.info(f"Clerk domain: {clerk_domain}")
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
logger.debug(f"Token endpoint: {config.token_endpoint}")
@@ -9,7 +9,7 @@ import time
import logging
from dataclasses import dataclass
from datetime import datetime, timedelta
from typing import Any, Optional
from typing import Any
from urllib.parse import urlencode
import httpx
@@ -18,7 +18,7 @@ from fastapi.responses import RedirectResponse, JSONResponse
try:
from clerk_backend_api import Clerk
CLERK_AVAILABLE = True
except ImportError as e:
except ImportError:
CLERK_AVAILABLE = False
Clerk = None
@@ -6,7 +6,7 @@ Uses Redis for authorization code storage to support multi-machine deployment
import os
import logging
from typing import Optional
from urllib.parse import urlencode, quote
from urllib.parse import urlencode
from fastapi import APIRouter, Request, Query, HTTPException
from fastapi.responses import RedirectResponse, JSONResponse
@@ -39,7 +39,6 @@ def get_redis_session_store():
if redis_store is None:
try:
import concurrent.futures
import functools
# Use thread pool with timeout to prevent hanging
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
@@ -206,7 +205,6 @@ async def oauth_callback(
auth_code = f"clerk_auth_{os.urandom(16).hex()}"
# Prepare code data
import time
code_data = {
"user_id": user_id,
"session_id": session_id,
@@ -225,7 +223,7 @@ async def oauth_callback(
if success:
logger.info(f"Stored authorization code {auth_code[:10]}... in Redis with real JWT token")
else:
logger.error(f"Failed to store authorization code in Redis, falling back to in-memory")
logger.error("Failed to store authorization code in Redis, falling back to in-memory")
# Fall back to in-memory storage
if not hasattr(oauth_callback, '_code_storage'):
oauth_callback._code_storage = {}
@@ -236,7 +234,7 @@ async def oauth_callback(
if not hasattr(oauth_callback, '_code_storage'):
oauth_callback._code_storage = {}
oauth_callback._code_storage[auth_code] = code_data
logger.info(f"Stored authorization code in memory (fallback)")
logger.info("Stored authorization code in memory (fallback)")
# Redirect back to client with authorization code
redirect_params = {
+32 -57
View File
@@ -8,8 +8,7 @@ import json
import time
from collections import defaultdict
from pydantic import HttpUrl, Field
from typing import Optional, Dict, List, Literal, Any, Union
import urllib.parse
from typing import Dict, List, Literal, Any
import tiktoken
from fastmcp.server.middleware import Middleware, MiddlewareContext
from fastmcp.server.dependencies import get_access_token, AccessToken
@@ -159,7 +158,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("tool_call_error", input_tokens, 0,
tool_name, duration_ms)
@@ -189,7 +188,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("resource_read_error", 0, 0,
resource_uri, duration_ms)
@@ -219,7 +218,7 @@ class TokenCountingMiddleware(Middleware):
return result
except Exception as e:
except Exception:
duration_ms = (time.perf_counter() - start_time) * 1000
self.log_token_usage("prompt_get_error", 0, 0,
prompt_name, duration_ms)
@@ -260,10 +259,6 @@ def create_app(auth=None):
# --- Module Imports ---
from yargitay_mcp_module.client import YargitayOfficialApiClient
from yargitay_mcp_module.models import (
YargitayDetailedSearchRequest, YargitayDocumentMarkdown, CompactYargitaySearchResult,
YargitayBirimEnum, CleanYargitayDecisionEntry
)
from bedesten_mcp_module.client import BedestenApiClient
from bedesten_mcp_module.models import (
BedestenSearchRequest, BedestenSearchData,
@@ -271,10 +266,6 @@ from bedesten_mcp_module.models import (
)
from bedesten_mcp_module.enums import BirimAdiEnum
from danistay_mcp_module.client import DanistayApiClient
from danistay_mcp_module.models import (
DanistayKeywordSearchRequest, DanistayDetailedSearchRequest,
DanistayDocumentMarkdown, CompactDanistaySearchResult
)
from emsal_mcp_module.client import EmsalApiClient
from emsal_mcp_module.models import (
EmsalSearchRequest, EmsalDocumentMarkdown, CompactEmsalSearchResult
@@ -288,15 +279,7 @@ from anayasa_mcp_module.client import AnayasaMahkemesiApiClient
from anayasa_mcp_module.bireysel_client import AnayasaBireyselBasvuruApiClient
from anayasa_mcp_module.unified_client import AnayasaUnifiedClient
from anayasa_mcp_module.models import (
AnayasaNormDenetimiSearchRequest,
AnayasaSearchResult,
AnayasaDocumentMarkdown,
AnayasaBireyselReportSearchRequest,
AnayasaBireyselReportSearchResult,
AnayasaBireyselBasvuruDocumentMarkdown,
AnayasaUnifiedSearchRequest,
AnayasaUnifiedSearchResult,
AnayasaUnifiedDocumentMarkdown,
# Removed enum imports - now using Literal strings in models
)
# KIK Module Imports
@@ -318,14 +301,9 @@ from rekabet_mcp_module.models import (
from sayistay_mcp_module.client import SayistayApiClient
from sayistay_mcp_module.models import (
GenelKurulSearchRequest, GenelKurulSearchResponse,
TemyizKuruluSearchRequest, TemyizKuruluSearchResponse,
DaireSearchRequest, DaireSearchResponse,
SayistayDocumentMarkdown,
SayistayUnifiedSearchRequest, SayistayUnifiedSearchResult,
SayistayUnifiedDocumentMarkdown
)
from sayistay_mcp_module.enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
from sayistay_mcp_module.unified_client import SayistayUnifiedClient
# KVKK Module Imports
@@ -339,14 +317,11 @@ from kvkk_mcp_module.models import (
# BDDK Module Imports
from bddk_mcp_module.client import BddkApiClient
from bddk_mcp_module.models import (
BddkSearchRequest,
BddkSearchResult,
BddkDocumentMarkdown
BddkSearchRequest
)
# Create a placeholder app that will be properly initialized after tools are defined
from fastmcp import FastMCP
# Placeholder app for decorators - will be replaced in create_app() after all tools are defined
app = FastMCP("Yargı MCP Server Placeholder")
@@ -1372,7 +1347,7 @@ async def search_emsal_detailed_decisions(
page_size=page_size
)
logger.info(f"Tool 'search_emsal_detailed_decisions' called.")
logger.info("Tool 'search_emsal_detailed_decisions' called.")
try:
api_response = await emsal_client_instance.search_detailed_decisions(search_query)
if api_response.data:
@@ -1384,8 +1359,8 @@ async def search_emsal_detailed_decisions(
)
logger.warning("API response for Emsal search did not contain expected data structure.")
return CompactEmsalSearchResult(decisions=[], total_records=0, requested_page=search_query.page_number, page_size=search_query.page_size)
except Exception as e:
logger.exception(f"Error in tool 'search_emsal_detailed_decisions'.")
except Exception:
logger.exception("Error in tool 'search_emsal_detailed_decisions'.")
raise
@app.tool(
@@ -1401,8 +1376,8 @@ async def get_emsal_document_markdown(id: str) -> EmsalDocumentMarkdown:
if not id or not id.strip(): raise ValueError("Document ID required for Emsal.")
try:
return await emsal_client_instance.get_decision_document_as_markdown(id)
except Exception as e:
logger.exception(f"Error in tool 'get_emsal_document_markdown'.")
except Exception:
logger.exception("Error in tool 'get_emsal_document_markdown'.")
raise
# --- MCP Tools for Uyusmazlik ---
@@ -1470,11 +1445,11 @@ async def search_uyusmazlik_decisions(
not_hepsi=not_hepsi
)
logger.info(f"Tool 'search_uyusmazlik_decisions' called.")
logger.info("Tool 'search_uyusmazlik_decisions' called.")
try:
return await uyusmazlik_client_instance.search_decisions(search_params)
except Exception as e:
logger.exception(f"Error in tool 'search_uyusmazlik_decisions'.")
except Exception:
logger.exception("Error in tool 'search_uyusmazlik_decisions'.")
raise
@app.tool(
@@ -1493,8 +1468,8 @@ async def get_uyusmazlik_document_markdown_from_url(
raise ValueError("Document URL (document_url) is required for Uyuşmazlık document retrieval.")
try:
return await uyusmazlik_client_instance.get_decision_document_as_markdown(str(document_url))
except Exception as e:
logger.exception(f"Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
except Exception:
logger.exception("Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
raise
# --- DEACTIVATED: MCP Tools for Anayasa Mahkemesi (Individual Tools) ---
@@ -1585,8 +1560,8 @@ async def search_anayasa_unified(
result = await anayasa_unified_client_instance.search_unified(request)
return json.dumps(result.model_dump(), ensure_ascii=False, indent=2)
except Exception as e:
logger.exception(f"Error in tool 'search_anayasa_unified'.")
except Exception:
logger.exception("Error in tool 'search_anayasa_unified'.")
raise
@app.tool(
@@ -1607,8 +1582,8 @@ async def get_anayasa_document_unified(
result = await anayasa_unified_client_instance.get_document_unified(document_url, page_number)
return json.dumps(result.model_dump(mode='json'), ensure_ascii=False, indent=2)
except Exception as e:
logger.exception(f"Error in tool 'get_anayasa_document_unified'.")
except Exception:
logger.exception("Error in tool 'get_anayasa_document_unified'.")
raise
# --- MCP Tools for KIK (Kamu İhale Kurulu) ---
@@ -1654,15 +1629,15 @@ async def search_kik_decisions(
page=page
)
logger.info(f"Tool 'search_kik_decisions' called.")
logger.info("Tool 'search_kik_decisions' called.")
try:
api_response = await kik_client_instance.search_decisions(search_query)
page_param_for_log = search_query.page if hasattr(search_query, 'page') else 1
if not api_response.decisions and api_response.total_records == 0 and page_param_for_log == 1:
logger.warning(f"KIK search returned no decisions for query.")
logger.warning("KIK search returned no decisions for query.")
return api_response
except Exception as e:
logger.exception(f"Error in KIK search tool 'search_kik_decisions'.")
except Exception:
logger.exception("Error in KIK search tool 'search_kik_decisions'.")
current_page_val = search_query.page if hasattr(search_query, 'page') else 1
return KikSearchResult(decisions=[], total_records=0, current_page=current_page_val)
@@ -1758,7 +1733,7 @@ async def search_rekabet_kurumu_decisions(
try:
return await rekabet_client_instance.search_decisions(search_query)
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_rekabet_kurumu_decisions'.")
return RekabetSearchResult(decisions=[], retrieved_page_number=page, total_records_found=0, total_pages=0)
@@ -1781,7 +1756,7 @@ async def get_rekabet_kurumu_document(
try:
return await rekabet_client_instance.get_decision_document(karar_id, page_number=current_page_to_fetch)
except Exception as e:
except Exception:
logger.exception(f"Error in tool 'get_rekabet_kurumu_document'. Karar ID: {karar_id}")
raise
@@ -1876,7 +1851,7 @@ For best results, use exact phrases with quotes for legal terms."""),
"page_size": pageSize,
"searched_courts": court_types
}
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_bedesten_unified'")
raise
@@ -1898,7 +1873,7 @@ async def get_bedesten_document_markdown(
try:
return await bedesten_client_instance.get_document_as_markdown(documentId)
except Exception as e:
except Exception:
logger.exception("Error in tool 'get_kyb_bedesten_document_markdown'")
raise
@@ -2087,7 +2062,7 @@ async def search_sayistay_unified(
web_karar_metni=web_karar_metni
)
return await sayistay_unified_client_instance.search_unified(search_request)
except Exception as e:
except Exception:
logger.exception("Error in tool 'search_sayistay_unified'")
raise
@@ -2111,7 +2086,7 @@ async def get_sayistay_document_unified(
try:
return await sayistay_unified_client_instance.get_document_unified(decision_id, decision_type)
except Exception as e:
except Exception:
logger.exception("Error in tool 'get_sayistay_document_unified'")
raise
@@ -2698,7 +2673,7 @@ async def search(
logger.info(f"ChatGPT Deep Research search completed. Found {len(results)} results via Bedesten API.")
return {"results": results}
except Exception as e:
except Exception:
logger.exception("Error in ChatGPT Deep Research search tool")
# Return partial results if any were found
if results:
@@ -2827,7 +2802,7 @@ async def fetch(
doc = await bedesten_client_instance.get_document_as_markdown(doc_id)
"""
except Exception as e:
except Exception:
logger.exception(f"Error fetching ChatGPT Deep Research document {id}")
raise
@@ -2874,7 +2849,7 @@ def main():
app.run()
except KeyboardInterrupt:
logger.info("Server shut down by user (KeyboardInterrupt).")
except Exception as e:
except Exception:
logger.exception("Server failed to start or crashed.")
finally:
logger.info(f"{app.name} server has shut down.")
@@ -11,8 +11,8 @@ import os
import json
import time
import logging
from typing import Optional, Dict, Any, Union
from datetime import datetime, timedelta
from typing import Optional, Dict, Any
from datetime import datetime
logger = logging.getLogger(__name__)
@@ -4,10 +4,9 @@ import httpx
from bs4 import BeautifulSoup
from typing import List, Optional, Tuple, Dict, Any
import logging
import html
import re
import io # For io.BytesIO
from urllib.parse import urlencode, urljoin, quote, parse_qs, urlparse
from urllib.parse import urljoin, parse_qs, urlparse
from markitdown import MarkItDown
import math
@@ -18,8 +17,7 @@ from .models import (
RekabetKurumuSearchRequest,
RekabetDecisionSummary,
RekabetSearchResult,
RekabetDocument,
RekabetKararTuruGuidEnum
RekabetDocument
)
from pydantic import HttpUrl # Ensure HttpUrl is imported from pydantic
@@ -1,7 +1,7 @@
# rekabet_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl
from typing import List, Optional, Any
from typing import List, Optional
from enum import Enum
# Enum for decision type GUIDs (used by the client and expected by the website)
+1 -1
View File
@@ -93,7 +93,7 @@ def main():
config["workers"] = args.workers
# Print startup information
print(f"Starting Yargı MCP server...")
print("Starting Yargı MCP server...")
print(f"Host: {args.host}")
print(f"Port: {args.port}")
print(f"Transport: {args.transport}")
@@ -1,13 +1,11 @@
# sayistay_mcp_module/client.py
import httpx
import re
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Tuple
from typing import Dict, List, Optional, Tuple
import logging
import html
import io
from urllib.parse import urlencode, urljoin
from urllib.parse import urlencode
from markitdown import MarkItDown
from .models import (
@@ -16,7 +14,7 @@ from .models import (
DaireSearchRequest, DaireSearchResponse, DaireDecision,
SayistayDocumentMarkdown
)
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum, WEB_KARAR_KONUSU_MAPPING
from .enums import WEB_KARAR_KONUSU_MAPPING
logger = logging.getLogger(__name__)
if not logger.hasHandlers():
@@ -1,7 +1,7 @@
# sayistay_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import Optional, List, Union, Dict, Any, Literal
from typing import Optional, List, Dict, Any, Literal
from enum import Enum
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
@@ -2,8 +2,6 @@
# Unified client for all three Sayıştay decision types
import logging
from typing import Optional, Dict, Any
from urllib.parse import urlparse
from .models import (
SayistayUnifiedSearchRequest,
@@ -13,7 +13,7 @@ import os
from starlette.applications import Starlette
from starlette.routing import Mount, Route
from starlette.requests import Request
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
from starlette.responses import JSONResponse
from starlette.middleware import Middleware
from starlette.middleware.cors import CORSMiddleware
from starlette.middleware.authentication import AuthenticationMiddleware
@@ -1,4 +1,5 @@
import os, stripe
import os
import stripe
from clerk_backend_api import Clerk # Clerk backend SDK
from fastapi import APIRouter, Request, HTTPException
@@ -3,7 +3,7 @@
import httpx
import aiohttp
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Union, Tuple
from typing import List, Optional, Tuple
import logging
import html
import re
@@ -1,20 +1,16 @@
# yargitay_mcp_module/client.py
import httpx
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
from typing import Dict, Any, List, Optional
from typing import Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
from .models import (
YargitayDetailedSearchRequest,
YargitayApiSearchResponse,
YargitayApiDecisionEntry,
YargitayDocumentMarkdown,
CompactYargitaySearchResult
YargitayDocumentMarkdown
)
logger = logging.getLogger(__name__)
@@ -1,7 +1,7 @@
# yargitay_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl, ConfigDict
from typing import List, Optional, Dict, Any, Literal
from typing import List, Optional, Literal
# Yargıtay Chamber/Board Options
YargitayBirimEnum = Literal[
+3 -5
View File
@@ -1,13 +1,11 @@
# sayistay_mcp_module/client.py
import httpx
import re
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Tuple
from typing import Dict, List, Optional, Tuple
import logging
import html
import io
from urllib.parse import urlencode, urljoin
from urllib.parse import urlencode
from markitdown import MarkItDown
from .models import (
@@ -16,7 +14,7 @@ from .models import (
DaireSearchRequest, DaireSearchResponse, DaireDecision,
SayistayDocumentMarkdown
)
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum, WEB_KARAR_KONUSU_MAPPING
from .enums import WEB_KARAR_KONUSU_MAPPING
logger = logging.getLogger(__name__)
if not logger.hasHandlers():
+1 -1
View File
@@ -1,7 +1,7 @@
# sayistay_mcp_module/models.py
from pydantic import BaseModel, Field
from typing import Optional, List, Union, Dict, Any, Literal
from typing import Optional, List, Dict, Any, Literal
from enum import Enum
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
-2
View File
@@ -2,8 +2,6 @@
# Unified client for all three Sayıştay decision types
import logging
from typing import Optional, Dict, Any
from urllib.parse import urlparse
from .models import (
SayistayUnifiedSearchRequest,
+7
View File
@@ -0,0 +1,7 @@
# semantic_search/__init__.py
from .embedder import EmbeddingGemma
from .vector_store import VectorStore
from .processor import DocumentProcessor
__all__ = ['EmbeddingGemma', 'VectorStore', 'DocumentProcessor']
+173
View File
@@ -0,0 +1,173 @@
# semantic_search/embedder.py
import logging
from typing import List, Optional
import numpy as np
from sentence_transformers import SentenceTransformer
import torch
logger = logging.getLogger(__name__)
class EmbeddingGemma:
"""
Wrapper for Google's EmbeddingGemma model.
Handles query and document encoding with proper prompt templates.
"""
def __init__(self, model_name: str = "google/embeddinggemma-300m", device: Optional[str] = None):
"""
Initialize EmbeddingGemma model.
Args:
model_name: HuggingFace model name
device: Device to run model on ('cuda', 'cpu', or None for auto)
"""
self.model_name = model_name
# Auto-detect device if not specified
if device is None:
self.device = 'cuda' if torch.cuda.is_available() else 'cpu'
else:
self.device = device
logger.info(f"Initializing EmbeddingGemma on device: {self.device}")
try:
# Load model with float32 precision (EmbeddingGemma doesn't support float16)
self.model = SentenceTransformer(model_name, device=self.device)
self.model.eval() # Set to evaluation mode
# Set precision to float32 or bfloat16
if self.device == 'cuda' and torch.cuda.is_bf16_supported():
logger.info("Using bfloat16 precision for CUDA")
self.dtype = torch.bfloat16
else:
logger.info("Using float32 precision")
self.dtype = torch.float32
logger.info(f"Successfully loaded model: {model_name}")
except Exception as e:
logger.error(f"Failed to load EmbeddingGemma model: {e}")
raise
def encode_query(self, query: str, task: str = "search result") -> np.ndarray:
"""
Encode a search query with appropriate prompt template.
Args:
query: The search query text
task: Task type for prompt template (search result, question answering, etc.)
Returns:
Numpy array of embeddings (768 dimensions)
"""
# Apply query prompt template
prompted_query = f"task: {task} | query: {query}"
try:
with torch.no_grad():
# Encode with model
embeddings = self.model.encode(
prompted_query,
convert_to_numpy=True,
normalize_embeddings=True, # L2 normalization for cosine similarity
show_progress_bar=False
)
logger.debug(f"Encoded query: {query[:50]}... -> shape: {embeddings.shape}")
return embeddings
except Exception as e:
logger.error(f"Failed to encode query: {e}")
raise
def encode_documents(self, documents: List[str], titles: Optional[List[str]] = None) -> np.ndarray:
"""
Encode multiple documents with appropriate prompt template.
Args:
documents: List of document texts
titles: Optional list of document titles
Returns:
Numpy array of embeddings (N x 768 dimensions)
"""
if not documents:
return np.array([])
# Apply document prompt template
prompted_docs = []
for i, doc in enumerate(documents):
title = titles[i] if titles and i < len(titles) else "none"
prompted_doc = f"title: {title} | text: {doc}"
prompted_docs.append(prompted_doc)
try:
with torch.no_grad():
# Batch encode documents
embeddings = self.model.encode(
prompted_docs,
convert_to_numpy=True,
normalize_embeddings=True,
show_progress_bar=len(documents) > 10,
batch_size=8 # Adjust based on memory
)
logger.info(f"Encoded {len(documents)} documents -> shape: {embeddings.shape}")
return embeddings
except Exception as e:
logger.error(f"Failed to encode documents: {e}")
raise
def reduce_dimensions(self, embeddings: np.ndarray, target_dim: int = 512) -> np.ndarray:
"""
Reduce embedding dimensions using Matryoshka Representation Learning.
Args:
embeddings: Original embeddings (N x 768)
target_dim: Target dimension (512, 256, or 128)
Returns:
Reduced embeddings (N x target_dim)
"""
if target_dim not in [512, 256, 128]:
raise ValueError(f"Target dimension must be 512, 256, or 128, got {target_dim}")
if len(embeddings.shape) == 1:
# Single embedding
reduced = embeddings[:target_dim]
# Re-normalize after truncation
norm = np.linalg.norm(reduced)
if norm > 0:
reduced = reduced / norm
else:
# Multiple embeddings
reduced = embeddings[:, :target_dim]
# Re-normalize each embedding
norms = np.linalg.norm(reduced, axis=1, keepdims=True)
reduced = reduced / (norms + 1e-8) # Avoid division by zero
logger.debug(f"Reduced dimensions: {embeddings.shape} -> {reduced.shape}")
return reduced
def compute_similarity(self, query_embedding: np.ndarray, document_embeddings: np.ndarray) -> np.ndarray:
"""
Compute cosine similarity between query and documents.
Args:
query_embedding: Query embedding (768,)
document_embeddings: Document embeddings (N x 768)
Returns:
Similarity scores (N,)
"""
# Ensure query is 2D for matrix multiplication
if len(query_embedding.shape) == 1:
query_embedding = query_embedding.reshape(1, -1)
# Compute cosine similarity (embeddings are already normalized)
similarities = np.dot(document_embeddings, query_embedding.T).squeeze()
return similarities
+305
View File
@@ -0,0 +1,305 @@
# semantic_search/processor.py
import logging
import re
from typing import List, Dict, Any, Optional
from dataclasses import dataclass
import hashlib
logger = logging.getLogger(__name__)
@dataclass
class DocumentChunk:
"""Represents a chunk of a document."""
chunk_id: str
document_id: str
text: str
metadata: Dict[str, Any]
chunk_index: int
total_chunks: int
class DocumentProcessor:
"""
Processes legal documents for semantic search.
Handles chunking, cleaning, and metadata extraction.
"""
def __init__(self,
chunk_size: int = 1000,
chunk_overlap: int = 200,
min_chunk_size: int = 100):
"""
Initialize document processor.
Args:
chunk_size: Target size for each chunk in characters
chunk_overlap: Number of overlapping characters between chunks
min_chunk_size: Minimum chunk size to keep
"""
self.chunk_size = chunk_size
self.chunk_overlap = chunk_overlap
self.min_chunk_size = min_chunk_size
logger.info(f"Initialized DocumentProcessor (chunk_size={chunk_size}, overlap={chunk_overlap})")
def process_document(self,
document_id: str,
text: str,
metadata: Optional[Dict[str, Any]] = None) -> List[DocumentChunk]:
"""
Process a single document into chunks.
Args:
document_id: Unique document identifier
text: Document text content
metadata: Optional document metadata
Returns:
List of document chunks
"""
if not text or len(text.strip()) < self.min_chunk_size:
logger.warning(f"Document {document_id} too short to process")
return []
# Clean text
cleaned_text = self._clean_text(text)
# Extract metadata from text if not provided
if metadata is None:
metadata = {}
# Add extracted metadata
extracted_metadata = self._extract_metadata(cleaned_text)
metadata.update(extracted_metadata)
# Create chunks
chunks = self._create_chunks(cleaned_text)
# Create DocumentChunk objects
document_chunks = []
for i, chunk_text in enumerate(chunks):
chunk_id = self._generate_chunk_id(document_id, i)
chunk = DocumentChunk(
chunk_id=chunk_id,
document_id=document_id,
text=chunk_text,
metadata={
**metadata,
'chunk_index': i,
'total_chunks': len(chunks)
},
chunk_index=i,
total_chunks=len(chunks)
)
document_chunks.append(chunk)
logger.info(f"Processed document {document_id} into {len(chunks)} chunks")
return document_chunks
def _clean_text(self, text: str) -> str:
"""
Clean and normalize text for processing.
Args:
text: Raw text
Returns:
Cleaned text
"""
# Remove excessive whitespace
text = re.sub(r'\s+', ' ', text)
# Remove special characters but keep Turkish characters
# Keep: letters, numbers, spaces, and common punctuation
text = re.sub(r'[^\w\s\.\,\;\:\!\?\-\(\)\"\'ÇĞIİÖŞÜçğıiöşü]', ' ', text)
# Remove multiple spaces
text = re.sub(r' +', ' ', text)
# Trim
text = text.strip()
return text
def _extract_metadata(self, text: str) -> Dict[str, Any]:
"""
Extract metadata from legal document text.
Args:
text: Document text
Returns:
Extracted metadata
"""
metadata = {}
# Extract case numbers (Esas/Karar)
esas_pattern = r'E(?:sas)?[\s\.\:]*(\d{4})[\/\-](\d+)'
karar_pattern = r'K(?:arar)?[\s\.\:]*(\d{4})[\/\-](\d+)'
esas_match = re.search(esas_pattern, text[:500]) # Look in first 500 chars
if esas_match:
metadata['esas_no'] = f"E.{esas_match.group(1)}/{esas_match.group(2)}"
karar_match = re.search(karar_pattern, text[:500])
if karar_match:
metadata['karar_no'] = f"K.{karar_match.group(1)}/{karar_match.group(2)}"
# Extract dates (DD.MM.YYYY or DD/MM/YYYY format)
date_pattern = r'(\d{1,2})[\.\/](\d{1,2})[\.\/](\d{4})'
dates = re.findall(date_pattern, text[:1000]) # Look in first 1000 chars
if dates:
# Take the first date as decision date
day, month, year = dates[0]
metadata['karar_tarihi'] = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
# Extract court/chamber name
chamber_patterns = [
r'(\d+)\.\s*Hukuk\s+Dairesi',
r'(\d+)\.\s*Ceza\s+Dairesi',
r'Hukuk\s+Genel\s+Kurulu',
r'Ceza\s+Genel\s+Kurulu',
r'(\d+)\.\s*Daire'
]
for pattern in chamber_patterns:
match = re.search(pattern, text[:500], re.IGNORECASE)
if match:
metadata['chamber'] = match.group(0)
break
return metadata
def _create_chunks(self, text: str) -> List[str]:
"""
Create overlapping chunks from text.
Args:
text: Cleaned document text
Returns:
List of text chunks
"""
chunks = []
# Split by sentences for better semantic coherence
sentences = self._split_sentences(text)
current_chunk = []
current_size = 0
for sentence in sentences:
sentence_size = len(sentence)
# If adding this sentence exceeds chunk size
if current_size + sentence_size > self.chunk_size and current_chunk:
# Save current chunk
chunk_text = ' '.join(current_chunk)
chunks.append(chunk_text)
# Create overlap for next chunk
overlap_size = 0
overlap_sentences = []
# Add sentences from the end until we reach overlap size
for sent in reversed(current_chunk):
overlap_size += len(sent)
overlap_sentences.insert(0, sent)
if overlap_size >= self.chunk_overlap:
break
# Start new chunk with overlap
current_chunk = overlap_sentences
current_size = sum(len(s) for s in current_chunk)
# Add sentence to current chunk
current_chunk.append(sentence)
current_size += sentence_size
# Add final chunk if not empty
if current_chunk:
chunk_text = ' '.join(current_chunk)
if len(chunk_text) >= self.min_chunk_size:
chunks.append(chunk_text)
return chunks
def _split_sentences(self, text: str) -> List[str]:
"""
Split text into sentences.
Args:
text: Text to split
Returns:
List of sentences
"""
# Simple sentence splitting for Turkish text
# Split on period, question mark, exclamation, but not on abbreviations
# Common Turkish abbreviations to preserve
abbreviations = ['Dr', 'Prof', 'Av', 'Md', 'Yrd', 'Doç', 'No', 'S', 'vs', 'vb', 'bkz']
# Replace abbreviations temporarily
temp_text = text
replacements = {}
for i, abbr in enumerate(abbreviations):
placeholder = f"__ABBR{i}__"
temp_text = temp_text.replace(f"{abbr}.", placeholder)
replacements[placeholder] = f"{abbr}."
# Split sentences
sentence_endings = re.compile(r'[.!?]+')
sentences = sentence_endings.split(temp_text)
# Restore abbreviations and clean
cleaned_sentences = []
for sentence in sentences:
# Restore abbreviations
for placeholder, original in replacements.items():
sentence = sentence.replace(placeholder, original)
# Clean and add if not empty
sentence = sentence.strip()
if sentence and len(sentence) > 10: # Minimum sentence length
cleaned_sentences.append(sentence)
return cleaned_sentences
def _generate_chunk_id(self, document_id: str, chunk_index: int) -> str:
"""
Generate unique chunk ID.
Args:
document_id: Parent document ID
chunk_index: Index of chunk in document
Returns:
Unique chunk ID
"""
chunk_string = f"{document_id}_chunk_{chunk_index}"
chunk_hash = hashlib.md5(chunk_string.encode()).hexdigest()[:8]
return f"{document_id}_c{chunk_index}_{chunk_hash}"
def combine_chunks(self, chunks: List[DocumentChunk]) -> str:
"""
Combine chunks back into full document text.
Args:
chunks: List of document chunks
Returns:
Combined text
"""
if not chunks:
return ""
# Sort by chunk index
sorted_chunks = sorted(chunks, key=lambda x: x.chunk_index)
# For overlapping chunks, we need to be careful about duplication
# Simple approach: just concatenate with space
combined = " ".join([chunk.text for chunk in sorted_chunks])
return combined
+235
View File
@@ -0,0 +1,235 @@
# semantic_search/vector_store.py
import logging
import numpy as np
from typing import List, Dict, Any, Tuple, Optional
from dataclasses import dataclass
import json
logger = logging.getLogger(__name__)
@dataclass
class Document:
"""Represents a document with its embedding and metadata."""
id: str
text: str
embedding: np.ndarray
metadata: Dict[str, Any]
def to_dict(self) -> Dict[str, Any]:
"""Convert to dictionary (excluding embedding for serialization)."""
return {
'id': self.id,
'text': self.text,
'metadata': self.metadata
}
class VectorStore:
"""
In-memory vector storage with similarity search capabilities.
Future versions can use Faiss, ChromaDB, or other vector databases.
"""
def __init__(self, dimension: int = 768):
"""
Initialize vector store.
Args:
dimension: Embedding dimension size
"""
self.dimension = dimension
self.documents: List[Document] = []
self.embeddings: Optional[np.ndarray] = None
self.index_built = False
logger.info(f"Initialized VectorStore with dimension: {dimension}")
def add_documents(self,
ids: List[str],
texts: List[str],
embeddings: np.ndarray,
metadata: Optional[List[Dict[str, Any]]] = None) -> int:
"""
Add documents to the vector store.
Args:
ids: Document IDs
texts: Document texts
embeddings: Document embeddings (N x dimension)
metadata: Optional metadata for each document
Returns:
Number of documents added
"""
if len(ids) != len(texts) or len(ids) != embeddings.shape[0]:
raise ValueError("Mismatched lengths for ids, texts, and embeddings")
if metadata and len(metadata) != len(ids):
raise ValueError("Metadata length doesn't match document count")
# Add documents
for i in range(len(ids)):
doc = Document(
id=ids[i],
text=texts[i],
embedding=embeddings[i],
metadata=metadata[i] if metadata else {}
)
self.documents.append(doc)
# Rebuild index
self._build_index()
logger.info(f"Added {len(ids)} documents to vector store. Total: {len(self.documents)}")
return len(ids)
def _build_index(self):
"""Build or rebuild the embedding index."""
if not self.documents:
self.embeddings = None
self.index_built = False
return
# Stack all embeddings into a single array
self.embeddings = np.vstack([doc.embedding for doc in self.documents])
self.index_built = True
logger.debug(f"Built index with shape: {self.embeddings.shape}")
def search(self,
query_embedding: np.ndarray,
top_k: int = 10,
threshold: Optional[float] = None) -> List[Tuple[Document, float]]:
"""
Search for similar documents using cosine similarity.
Args:
query_embedding: Query embedding vector
top_k: Number of results to return
threshold: Optional similarity threshold (0-1)
Returns:
List of (Document, similarity_score) tuples
"""
if not self.index_built or self.embeddings is None:
logger.warning("No documents in vector store")
return []
# Ensure query is 2D
if len(query_embedding.shape) == 1:
query_embedding = query_embedding.reshape(1, -1)
# Compute cosine similarities (assuming normalized embeddings)
similarities = np.dot(self.embeddings, query_embedding.T).squeeze()
# Apply threshold if specified
if threshold is not None:
valid_indices = np.where(similarities >= threshold)[0]
if len(valid_indices) == 0:
logger.info(f"No documents above threshold {threshold}")
return []
similarities = similarities[valid_indices]
valid_docs = [self.documents[i] for i in valid_indices]
else:
valid_docs = self.documents
# Get top-k indices
top_k = min(top_k, len(valid_docs))
if top_k == 0:
return []
# Use argpartition for efficiency with large arrays
if len(similarities) > top_k:
top_indices = np.argpartition(similarities, -top_k)[-top_k:]
top_indices = top_indices[np.argsort(similarities[top_indices])[::-1]]
else:
top_indices = np.argsort(similarities)[::-1]
# Create results
results = []
for idx in top_indices:
doc = valid_docs[idx] if threshold else self.documents[idx]
score = float(similarities[idx])
results.append((doc, score))
logger.info(f"Search returned {len(results)} results (top_k={top_k})")
return results
def hybrid_search(self,
query_embedding: np.ndarray,
keyword_scores: Dict[str, float],
top_k: int = 10,
alpha: float = 0.5) -> List[Tuple[Document, float]]:
"""
Hybrid search combining vector similarity and keyword scores.
Args:
query_embedding: Query embedding vector
keyword_scores: Document ID to keyword relevance score mapping
top_k: Number of results to return
alpha: Weight for vector similarity (1-alpha for keyword score)
Returns:
List of (Document, combined_score) tuples
"""
if not self.index_built:
logger.warning("No documents in vector store")
return []
# Get vector similarities
vector_results = self.search(query_embedding, top_k=len(self.documents))
# Combine scores
combined_scores = []
for doc, vector_score in vector_results:
keyword_score = keyword_scores.get(doc.id, 0.0)
# Normalize keyword score to 0-1 range if needed
if keyword_score > 1.0:
keyword_score = keyword_score / max(keyword_scores.values())
combined_score = alpha * vector_score + (1 - alpha) * keyword_score
combined_scores.append((doc, combined_score))
# Sort by combined score and return top-k
combined_scores.sort(key=lambda x: x[1], reverse=True)
results = combined_scores[:top_k]
logger.info(f"Hybrid search returned {len(results)} results")
return results
def clear(self):
"""Clear all documents from the store."""
self.documents = []
self.embeddings = None
self.index_built = False
logger.info("Cleared vector store")
def size(self) -> int:
"""Get number of documents in store."""
return len(self.documents)
def get_by_id(self, doc_id: str) -> Optional[Document]:
"""Get document by ID."""
for doc in self.documents:
if doc.id == doc_id:
return doc
return None
def get_stats(self) -> Dict[str, Any]:
"""Get statistics about the vector store."""
stats = {
'num_documents': len(self.documents),
'dimension': self.dimension,
'index_built': self.index_built,
'memory_usage_mb': 0
}
if self.embeddings is not None:
# Estimate memory usage
memory_bytes = self.embeddings.nbytes
for doc in self.documents:
memory_bytes += len(doc.text.encode('utf-8'))
memory_bytes += len(json.dumps(doc.metadata).encode('utf-8'))
stats['memory_usage_mb'] = memory_bytes / (1024 * 1024)
return stats
+1 -1
View File
@@ -13,7 +13,7 @@ import os
from starlette.applications import Starlette
from starlette.routing import Mount, Route
from starlette.requests import Request
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
from starlette.responses import JSONResponse
from starlette.middleware import Middleware
from starlette.middleware.cors import CORSMiddleware
from starlette.middleware.authentication import AuthenticationMiddleware
+2 -1
View File
@@ -1,4 +1,5 @@
import os, stripe
import os
import stripe
from clerk_backend_api import Clerk # Clerk backend SDK
from fastapi import APIRouter, Request, HTTPException
+1 -1
View File
@@ -2,7 +2,7 @@
import httpx
from bs4 import BeautifulSoup
from typing import Dict, Any, List, Optional, Union, Tuple
from typing import List, Optional, Tuple
import logging
import html
import re
+2 -6
View File
@@ -1,20 +1,16 @@
# yargitay_mcp_module/client.py
import httpx
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
from typing import Dict, Any, List, Optional
from typing import Optional
import logging
import html
import re
import io
from markitdown import MarkItDown
from .models import (
YargitayDetailedSearchRequest,
YargitayApiSearchResponse,
YargitayApiDecisionEntry,
YargitayDocumentMarkdown,
CompactYargitaySearchResult
YargitayDocumentMarkdown
)
logger = logging.getLogger(__name__)
+1 -1
View File
@@ -1,7 +1,7 @@
# yargitay_mcp_module/models.py
from pydantic import BaseModel, Field, HttpUrl, ConfigDict
from typing import List, Optional, Dict, Any, Literal
from typing import List, Optional, Literal
# Yargıtay Chamber/Board Options
YargitayBirimEnum = Literal[