Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9b43070754 |
@@ -8,7 +8,6 @@ and trying to reverse engineer the hash generation logic.
|
||||
import asyncio
|
||||
import json
|
||||
import hashlib
|
||||
import hmac
|
||||
import base64
|
||||
from fastmcp import Client
|
||||
from mcp_server_main import app
|
||||
@@ -152,7 +151,7 @@ async def test_hash_generation_comprehensive():
|
||||
|
||||
# Test with first decision
|
||||
sample_decision = decisions[0]
|
||||
print(f"\n📋 Sample decision for hash analysis:")
|
||||
print("\n📋 Sample decision for hash analysis:")
|
||||
for key, value in sample_decision.items():
|
||||
print(f" {key}: {value}")
|
||||
|
||||
@@ -162,20 +161,20 @@ async def test_hash_generation_comprehensive():
|
||||
all_hashes = {}
|
||||
|
||||
# Test different hash generation methods
|
||||
print(f"\n🔨 Testing webpack-style hashing...")
|
||||
print("\n🔨 Testing webpack-style hashing...")
|
||||
webpack_hashes = test_webpack_style_hashing(sample_decision)
|
||||
all_hashes.update(webpack_hashes)
|
||||
|
||||
print(f"🔨 Testing Angular routing hashes...")
|
||||
print("🔨 Testing Angular routing hashes...")
|
||||
angular_hashes = test_angular_routing_hashes(sample_decision)
|
||||
all_hashes.update(angular_hashes)
|
||||
|
||||
print(f"🔨 Testing base64 encoding variants...")
|
||||
print("🔨 Testing base64 encoding variants...")
|
||||
b64_hashes = test_base64_encoding_variants(sample_decision)
|
||||
all_hashes.update(b64_hashes)
|
||||
|
||||
# Check for matches
|
||||
print(f"\n🎯 Checking for hash matches...")
|
||||
print("\n🎯 Checking for hash matches...")
|
||||
matches_found = []
|
||||
partial_matches = []
|
||||
|
||||
@@ -191,13 +190,13 @@ async def test_hash_generation_comprehensive():
|
||||
print(f" 🔍 Partial match (last 8): {hash_name} -> ...{hash_value[-16:]}")
|
||||
|
||||
if not matches_found and not partial_matches:
|
||||
print(f" ❌ No matches found")
|
||||
print(f"\n📝 Sample generated hashes (first 10):")
|
||||
print(" ❌ No matches found")
|
||||
print("\n📝 Sample generated hashes (first 10):")
|
||||
for i, (hash_name, hash_value) in enumerate(list(all_hashes.items())[:10]):
|
||||
print(f" {hash_name}: {hash_value}")
|
||||
|
||||
# Try combinations with other decisions
|
||||
print(f"\n🔄 Testing hash combinations with multiple decisions...")
|
||||
print("\n🔄 Testing hash combinations with multiple decisions...")
|
||||
if len(decisions) > 1:
|
||||
for i, decision in enumerate(decisions[1:3]): # Test 2 more
|
||||
print(f"\n Testing decision {i+2}: {decision.get('kararNo')}")
|
||||
@@ -209,7 +208,7 @@ async def test_hash_generation_comprehensive():
|
||||
matches_found.append((f"decision_{i+2}_{hash_name}", hash_value))
|
||||
|
||||
# Try composite hashes (combining multiple fields)
|
||||
print(f"\n🔗 Testing composite hash generation...")
|
||||
print("\n🔗 Testing composite hash generation...")
|
||||
composite_tests = [
|
||||
f"{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararNo')}",
|
||||
f"{sample_decision.get('kararNo')}_{sample_decision.get('kararTarihi')}",
|
||||
@@ -224,7 +223,7 @@ async def test_hash_generation_comprehensive():
|
||||
print(f" 🎉 COMPOSITE MATCH FOUND: test_{i} -> {composite_str[:50]}...")
|
||||
matches_found.append((f"composite_{i}", composite_hash))
|
||||
|
||||
print(f"\n🎯 Hash analysis completed!")
|
||||
print("\n🎯 Hash analysis completed!")
|
||||
print(f" Total matches found: {len(matches_found)}")
|
||||
print(f" Partial matches: {len(partial_matches)}")
|
||||
|
||||
|
||||
@@ -2,13 +2,13 @@
|
||||
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup, Tag
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from urllib.parse import urljoin
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from urllib.parse import urljoin
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
# Unified client for both Norm Denetimi and Bireysel Başvuru
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .models import (
|
||||
|
||||
+8
-11
@@ -10,16 +10,13 @@ Usage:
|
||||
"""
|
||||
|
||||
import os
|
||||
import time
|
||||
import logging
|
||||
import json
|
||||
from datetime import datetime, timedelta
|
||||
from fastapi import FastAPI, Request, HTTPException, Query
|
||||
from fastapi.responses import JSONResponse, HTMLResponse, Response
|
||||
from fastapi.exception_handlers import http_exception_handler
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.middleware.base import BaseHTTPMiddleware
|
||||
|
||||
# Import the proper create_app function that includes all middleware
|
||||
from mcp_server_main import create_app
|
||||
@@ -505,14 +502,14 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
logger.info("User authenticated successfully via Clerk")
|
||||
|
||||
# Return success response
|
||||
return HTMLResponse(f"""
|
||||
return HTMLResponse("""
|
||||
<html>
|
||||
<head>
|
||||
<title>MCP Connection Successful</title>
|
||||
<style>
|
||||
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
|
||||
.success {{ color: #28a745; }}
|
||||
.token {{ background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }}
|
||||
body { font-family: Arial, sans-serif; text-align: center; padding: 50px; }
|
||||
.success { color: #28a745; }
|
||||
.token { background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
@@ -525,13 +522,13 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
<p>You can now close this window and return to your MCP client.</p>
|
||||
<script>
|
||||
// Try to close the popup if opened as such
|
||||
if (window.opener) {{
|
||||
window.opener.postMessage({{
|
||||
if (window.opener) {
|
||||
window.opener.postMessage({
|
||||
type: 'MCP_AUTH_SUCCESS',
|
||||
token: 'use_clerk_jwt_token'
|
||||
}}, '*');
|
||||
}, '*');
|
||||
setTimeout(() => window.close(), 3000);
|
||||
}}
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
# bddk_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from typing import List, Optional, Dict, Any
|
||||
from typing import Optional
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urlparse
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# bddk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional
|
||||
from typing import List
|
||||
|
||||
class BddkSearchRequest(BaseModel):
|
||||
"""
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
# bedesten_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional, Dict, Any, Literal, Union
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Dict, Any, Literal
|
||||
|
||||
# Import compressed BirimAdiEnum for chamber filtering
|
||||
from .enums import BirimAdiEnum
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
# danistay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Dict, List, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
|
||||
@@ -2,10 +2,9 @@
|
||||
|
||||
import httpx
|
||||
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Dict, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
|
||||
@@ -7,11 +7,9 @@ import os
|
||||
from typing import List, Dict, Any, Optional
|
||||
from datetime import datetime
|
||||
|
||||
from fastapi import FastAPI, HTTPException, Query, Depends, Body
|
||||
from fastapi import FastAPI, HTTPException, Query
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import JSONResponse
|
||||
from pydantic import BaseModel, Field
|
||||
import json
|
||||
|
||||
# Import the main MCP app
|
||||
from mcp_server_main import app as mcp_server
|
||||
|
||||
@@ -10,13 +10,12 @@ from playwright.async_api import (
|
||||
)
|
||||
from bs4 import BeautifulSoup
|
||||
import logging
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import List, Optional
|
||||
import urllib.parse
|
||||
import base64 # Base64 için
|
||||
import re
|
||||
import html as html_parser
|
||||
from markitdown import MarkItDown
|
||||
import os
|
||||
import math
|
||||
import io
|
||||
import random
|
||||
@@ -907,7 +906,7 @@ class KikApiClient:
|
||||
event_target_for_submit = self.FIELD_LOCATORS['search_button_id']
|
||||
# Use human-like clicking for search button
|
||||
search_button_selector = f"a[id='{event_target_for_submit}']"
|
||||
logger.info(f"Performing human-like search button click...")
|
||||
logger.info("Performing human-like search button click...")
|
||||
|
||||
try:
|
||||
# Hide datepicker first to prevent interference
|
||||
@@ -1181,7 +1180,7 @@ class KikApiClient:
|
||||
try:
|
||||
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=2000):
|
||||
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
|
||||
await current_main_page.wait_for_selector(f"div#detayPopUp:not(.in)", timeout=5000)
|
||||
await current_main_page.wait_for_selector("div#detayPopUp:not(.in)", timeout=5000)
|
||||
except: pass
|
||||
|
||||
return KikDocumentMarkdown(
|
||||
|
||||
@@ -1,13 +1,9 @@
|
||||
# kik_mcp_module/client_v2.py
|
||||
|
||||
import httpx
|
||||
import requests
|
||||
import logging
|
||||
import uuid
|
||||
import base64
|
||||
import ssl
|
||||
from typing import Optional
|
||||
from datetime import datetime
|
||||
|
||||
from .models_v2 import (
|
||||
KikV2DecisionType, KikV2SearchPayload, KikV2SearchPayloadDk, KikV2SearchPayloadMk,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
# kik_mcp_module/models.py
|
||||
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
|
||||
from pydantic import BaseModel, Field, computed_field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
import base64 # Base64 encoding/decoding için
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
# kik_mcp_module/models_v2.py
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from datetime import datetime
|
||||
from typing import List
|
||||
from enum import Enum
|
||||
|
||||
# New KIK v2 API Models
|
||||
|
||||
@@ -2,13 +2,13 @@
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Dict, Any
|
||||
from typing import Optional, Dict, Any
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urljoin, urlparse, parse_qs
|
||||
from urllib.parse import urlparse
|
||||
from markitdown import MarkItDown
|
||||
from pydantic import HttpUrl
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# kvkk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Any
|
||||
from typing import List, Optional
|
||||
|
||||
class KvkkSearchRequest(BaseModel):
|
||||
"""Model for KVKK (Personal Data Protection Authority) search request via Brave API."""
|
||||
|
||||
@@ -36,7 +36,7 @@ def create_clerk_oauth_config() -> OAuthConfig:
|
||||
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
|
||||
)
|
||||
|
||||
logger.info(f"Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info("Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info(f"Clerk domain: {clerk_domain}")
|
||||
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
|
||||
logger.debug(f"Token endpoint: {config.token_endpoint}")
|
||||
|
||||
+1
-1
@@ -9,7 +9,7 @@ import time
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Any, Optional
|
||||
from typing import Any
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import httpx
|
||||
|
||||
@@ -18,7 +18,7 @@ from fastapi.responses import RedirectResponse, JSONResponse
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError as e:
|
||||
except ImportError:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ Uses Redis for authorization code storage to support multi-machine deployment
|
||||
import os
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlencode, quote
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from fastapi import APIRouter, Request, Query, HTTPException
|
||||
from fastapi.responses import RedirectResponse, JSONResponse
|
||||
@@ -39,7 +39,6 @@ def get_redis_session_store():
|
||||
if redis_store is None:
|
||||
try:
|
||||
import concurrent.futures
|
||||
import functools
|
||||
|
||||
# Use thread pool with timeout to prevent hanging
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
|
||||
@@ -206,7 +205,6 @@ async def oauth_callback(
|
||||
auth_code = f"clerk_auth_{os.urandom(16).hex()}"
|
||||
|
||||
# Prepare code data
|
||||
import time
|
||||
code_data = {
|
||||
"user_id": user_id,
|
||||
"session_id": session_id,
|
||||
@@ -225,7 +223,7 @@ async def oauth_callback(
|
||||
if success:
|
||||
logger.info(f"Stored authorization code {auth_code[:10]}... in Redis with real JWT token")
|
||||
else:
|
||||
logger.error(f"Failed to store authorization code in Redis, falling back to in-memory")
|
||||
logger.error("Failed to store authorization code in Redis, falling back to in-memory")
|
||||
# Fall back to in-memory storage
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
@@ -236,7 +234,7 @@ async def oauth_callback(
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
oauth_callback._code_storage[auth_code] = code_data
|
||||
logger.info(f"Stored authorization code in memory (fallback)")
|
||||
logger.info("Stored authorization code in memory (fallback)")
|
||||
|
||||
# Redirect back to client with authorization code
|
||||
redirect_params = {
|
||||
|
||||
+261
-57
@@ -8,8 +8,7 @@ import json
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from pydantic import HttpUrl, Field
|
||||
from typing import Optional, Dict, List, Literal, Any, Union
|
||||
import urllib.parse
|
||||
from typing import Dict, List, Literal, Any
|
||||
import tiktoken
|
||||
from fastmcp.server.middleware import Middleware, MiddlewareContext
|
||||
from fastmcp.server.dependencies import get_access_token, AccessToken
|
||||
@@ -159,7 +158,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("tool_call_error", input_tokens, 0,
|
||||
tool_name, duration_ms)
|
||||
@@ -189,7 +188,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("resource_read_error", 0, 0,
|
||||
resource_uri, duration_ms)
|
||||
@@ -219,7 +218,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("prompt_get_error", 0, 0,
|
||||
prompt_name, duration_ms)
|
||||
@@ -255,21 +254,18 @@ def create_app(auth=None):
|
||||
|
||||
# --- Module Imports ---
|
||||
from yargitay_mcp_module.client import YargitayOfficialApiClient
|
||||
from yargitay_mcp_module.models import (
|
||||
YargitayDetailedSearchRequest, YargitayDocumentMarkdown, CompactYargitaySearchResult,
|
||||
YargitayBirimEnum, CleanYargitayDecisionEntry
|
||||
)
|
||||
from bedesten_mcp_module.client import BedestenApiClient
|
||||
from bedesten_mcp_module.models import (
|
||||
BedestenSearchRequest, BedestenSearchData,
|
||||
BedestenDocumentMarkdown, BedestenCourtTypeEnum
|
||||
)
|
||||
from bedesten_mcp_module.enums import BirimAdiEnum
|
||||
|
||||
# Semantic Search Module Imports
|
||||
from semantic_search.embedder import EmbeddingGemma
|
||||
from semantic_search.vector_store import VectorStore
|
||||
from semantic_search.processor import DocumentProcessor
|
||||
from danistay_mcp_module.client import DanistayApiClient
|
||||
from danistay_mcp_module.models import (
|
||||
DanistayKeywordSearchRequest, DanistayDetailedSearchRequest,
|
||||
DanistayDocumentMarkdown, CompactDanistaySearchResult
|
||||
)
|
||||
from emsal_mcp_module.client import EmsalApiClient
|
||||
from emsal_mcp_module.models import (
|
||||
EmsalSearchRequest, EmsalDocumentMarkdown, CompactEmsalSearchResult
|
||||
@@ -283,24 +279,12 @@ from anayasa_mcp_module.client import AnayasaMahkemesiApiClient
|
||||
from anayasa_mcp_module.bireysel_client import AnayasaBireyselBasvuruApiClient
|
||||
from anayasa_mcp_module.unified_client import AnayasaUnifiedClient
|
||||
from anayasa_mcp_module.models import (
|
||||
AnayasaNormDenetimiSearchRequest,
|
||||
AnayasaSearchResult,
|
||||
AnayasaDocumentMarkdown,
|
||||
AnayasaBireyselReportSearchRequest,
|
||||
AnayasaBireyselReportSearchResult,
|
||||
AnayasaBireyselBasvuruDocumentMarkdown,
|
||||
AnayasaUnifiedSearchRequest,
|
||||
AnayasaUnifiedSearchResult,
|
||||
AnayasaUnifiedDocumentMarkdown,
|
||||
# Removed enum imports - now using Literal strings in models
|
||||
)
|
||||
# KIK v2 Module Imports (New API)
|
||||
from kik_mcp_module.client_v2 import KikV2ApiClient
|
||||
from kik_mcp_module.models_v2 import KikV2DecisionType
|
||||
from kik_mcp_module.models_v2 import (
|
||||
KikV2SearchResult,
|
||||
KikV2DocumentMarkdown
|
||||
)
|
||||
|
||||
from rekabet_mcp_module.client import RekabetKurumuApiClient
|
||||
from rekabet_mcp_module.models import (
|
||||
@@ -312,14 +296,9 @@ from rekabet_mcp_module.models import (
|
||||
|
||||
from sayistay_mcp_module.client import SayistayApiClient
|
||||
from sayistay_mcp_module.models import (
|
||||
GenelKurulSearchRequest, GenelKurulSearchResponse,
|
||||
TemyizKuruluSearchRequest, TemyizKuruluSearchResponse,
|
||||
DaireSearchRequest, DaireSearchResponse,
|
||||
SayistayDocumentMarkdown,
|
||||
SayistayUnifiedSearchRequest, SayistayUnifiedSearchResult,
|
||||
SayistayUnifiedDocumentMarkdown
|
||||
)
|
||||
from sayistay_mcp_module.enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
from sayistay_mcp_module.unified_client import SayistayUnifiedClient
|
||||
|
||||
# KVKK Module Imports
|
||||
@@ -333,14 +312,11 @@ from kvkk_mcp_module.models import (
|
||||
# BDDK Module Imports
|
||||
from bddk_mcp_module.client import BddkApiClient
|
||||
from bddk_mcp_module.models import (
|
||||
BddkSearchRequest,
|
||||
BddkSearchResult,
|
||||
BddkDocumentMarkdown
|
||||
BddkSearchRequest
|
||||
)
|
||||
|
||||
|
||||
# Create a placeholder app that will be properly initialized after tools are defined
|
||||
from fastmcp import FastMCP
|
||||
|
||||
# MCP app for Turkish legal databases with explicit capabilities
|
||||
app = FastMCP(
|
||||
@@ -647,7 +623,7 @@ async def search_emsal_detailed_decisions(
|
||||
page_size=page_size
|
||||
)
|
||||
|
||||
logger.info(f"Tool 'search_emsal_detailed_decisions' called.")
|
||||
logger.info("Tool 'search_emsal_detailed_decisions' called.")
|
||||
try:
|
||||
api_response = await emsal_client_instance.search_detailed_decisions(search_query)
|
||||
if api_response.data:
|
||||
@@ -659,8 +635,8 @@ async def search_emsal_detailed_decisions(
|
||||
)
|
||||
logger.warning("API response for Emsal search did not contain expected data structure.")
|
||||
return CompactEmsalSearchResult(decisions=[], total_records=0, requested_page=search_query.page_number, page_size=search_query.page_size)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_emsal_detailed_decisions'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_emsal_detailed_decisions'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -676,8 +652,8 @@ async def get_emsal_document_markdown(id: str) -> EmsalDocumentMarkdown:
|
||||
if not id or not id.strip(): raise ValueError("Document ID required for Emsal.")
|
||||
try:
|
||||
return await emsal_client_instance.get_decision_document_as_markdown(id)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_emsal_document_markdown'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_emsal_document_markdown'.")
|
||||
raise
|
||||
|
||||
# --- MCP Tools for Uyusmazlik ---
|
||||
@@ -745,11 +721,11 @@ async def search_uyusmazlik_decisions(
|
||||
not_hepsi=not_hepsi
|
||||
)
|
||||
|
||||
logger.info(f"Tool 'search_uyusmazlik_decisions' called.")
|
||||
logger.info("Tool 'search_uyusmazlik_decisions' called.")
|
||||
try:
|
||||
return await uyusmazlik_client_instance.search_decisions(search_params)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_uyusmazlik_decisions'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_uyusmazlik_decisions'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -768,8 +744,8 @@ async def get_uyusmazlik_document_markdown_from_url(
|
||||
raise ValueError("Document URL (document_url) is required for Uyuşmazlık document retrieval.")
|
||||
try:
|
||||
return await uyusmazlik_client_instance.get_decision_document_as_markdown(str(document_url))
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
|
||||
raise
|
||||
|
||||
# --- DEACTIVATED: MCP Tools for Anayasa Mahkemesi (Individual Tools) ---
|
||||
@@ -860,8 +836,8 @@ async def search_anayasa_unified(
|
||||
result = await anayasa_unified_client_instance.search_unified(request)
|
||||
return json.dumps(result.model_dump(), ensure_ascii=False, indent=2)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_anayasa_unified'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_anayasa_unified'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -882,8 +858,8 @@ async def get_anayasa_document_unified(
|
||||
result = await anayasa_unified_client_instance.get_document_unified(document_url, page_number)
|
||||
return json.dumps(result.model_dump(mode='json'), ensure_ascii=False, indent=2)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_anayasa_document_unified'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_anayasa_document_unified'.")
|
||||
raise
|
||||
|
||||
# --- MCP Tools for KIK v2 (Kamu İhale Kurulu - New API) ---
|
||||
@@ -1057,7 +1033,7 @@ async def search_rekabet_kurumu_decisions(
|
||||
try:
|
||||
|
||||
return await rekabet_client_instance.search_decisions(search_query)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_rekabet_kurumu_decisions'.")
|
||||
return RekabetSearchResult(decisions=[], retrieved_page_number=page, total_records_found=0, total_pages=0)
|
||||
|
||||
@@ -1080,7 +1056,7 @@ async def get_rekabet_kurumu_document(
|
||||
try:
|
||||
|
||||
return await rekabet_client_instance.get_decision_document(karar_id, page_number=current_page_to_fetch)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception(f"Error in tool 'get_rekabet_kurumu_document'. Karar ID: {karar_id}")
|
||||
raise
|
||||
|
||||
@@ -1191,7 +1167,7 @@ For best results, use exact phrases with quotes for legal terms."""),
|
||||
"page_size": pageSize,
|
||||
"searched_courts": court_types
|
||||
}
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_bedesten_unified'")
|
||||
raise
|
||||
|
||||
@@ -1213,7 +1189,7 @@ async def get_bedesten_document_markdown(
|
||||
|
||||
try:
|
||||
return await bedesten_client_instance.get_document_as_markdown(documentId)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_kyb_bedesten_document_markdown'")
|
||||
raise
|
||||
|
||||
@@ -1402,7 +1378,7 @@ async def search_sayistay_unified(
|
||||
web_karar_metni=web_karar_metni
|
||||
)
|
||||
return await sayistay_unified_client_instance.search_unified(search_request)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_sayistay_unified'")
|
||||
raise
|
||||
|
||||
@@ -1426,7 +1402,7 @@ async def get_sayistay_document_unified(
|
||||
|
||||
try:
|
||||
return await sayistay_unified_client_instance.get_document_unified(decision_id, decision_type)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_sayistay_document_unified'")
|
||||
raise
|
||||
|
||||
@@ -1839,6 +1815,234 @@ async def get_bddk_document_markdown(
|
||||
"error": str(e)
|
||||
}
|
||||
|
||||
# --- Semantic Search Tool ---
|
||||
|
||||
@app.tool(
|
||||
description="Semantic search for Turkish legal decisions using EmbeddingGemma for intelligent ranking",
|
||||
annotations={
|
||||
"readOnlyHint": True,
|
||||
"openWorldHint": True,
|
||||
"idempotentHint": True
|
||||
}
|
||||
)
|
||||
async def search_bedesten_semantic(
|
||||
query: str = Field(..., description="Search query in Turkish for semantic matching"),
|
||||
initial_keyword: str = Field(..., description="Initial keyword for Bedesten API search (broad term)"),
|
||||
court_types: List[BedestenCourtTypeEnum] = Field(
|
||||
default=["YARGITAYKARARI", "DANISTAYKARAR", "YERELHUKUK", "ISTINAFHUKUK", "KYB"],
|
||||
description="Court types to search: YARGITAYKARARI, DANISTAYKARAR, YERELHUKUK, ISTINAFHUKUK, KYB (default: all)"
|
||||
),
|
||||
top_k: int = Field(10, ge=1, le=50, description="Number of top results to return (1-50)")
|
||||
) -> Dict[str, Any]:
|
||||
"""
|
||||
Perform semantic search on Turkish legal decisions using EmbeddingGemma.
|
||||
|
||||
This tool:
|
||||
1. Searches Bedesten API with initial keyword (retrieves 100 results)
|
||||
2. Fetches full document content for each result
|
||||
3. Generates embeddings using Google's EmbeddingGemma model
|
||||
4. Performs semantic similarity search with the query
|
||||
5. Returns re-ranked results based on semantic relevance
|
||||
|
||||
Benefits over keyword search:
|
||||
- Better understanding of context and meaning
|
||||
- Finds semantically similar documents even with different wording
|
||||
- More accurate ranking based on relevance
|
||||
- Supports multilingual queries (100+ languages)
|
||||
"""
|
||||
logger.info(f"Semantic search tool called with query: {query}, keyword: {initial_keyword}")
|
||||
|
||||
try:
|
||||
# Initialize components
|
||||
embedder = EmbeddingGemma()
|
||||
vector_store = VectorStore(dimension=256) # Always use 256 for optimal speed/quality balance
|
||||
processor = DocumentProcessor(chunk_size=1500, chunk_overlap=300)
|
||||
|
||||
# Step 1: Initial keyword search to get document IDs
|
||||
logger.info(f"Step 1: Searching Bedesten API with keyword: {initial_keyword}")
|
||||
|
||||
all_decisions = []
|
||||
|
||||
# Search each court type
|
||||
for court_type in court_types:
|
||||
try:
|
||||
# Calculate page size per court type to get 100 total
|
||||
per_court_limit = max(20, 100 // len(court_types))
|
||||
|
||||
search_results = await bedesten_client_instance.search_documents(
|
||||
BedestenSearchRequest(
|
||||
data=BedestenSearchData(
|
||||
phrase=initial_keyword,
|
||||
itemTypeList=[court_type],
|
||||
pageSize=per_court_limit, # Distribute 100 across court types
|
||||
pageNumber=1
|
||||
)
|
||||
)
|
||||
)
|
||||
|
||||
if search_results.data and search_results.data.emsalKararList:
|
||||
all_decisions.extend(search_results.data.emsalKararList)
|
||||
logger.info(f"Found {len(search_results.data.emsalKararList)} results from {court_type}")
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"Error searching {court_type}: {e}")
|
||||
|
||||
if not all_decisions:
|
||||
logger.warning("No documents found from initial search")
|
||||
return {
|
||||
"status": "no_results",
|
||||
"message": "No documents found matching the initial keyword",
|
||||
"results": []
|
||||
}
|
||||
|
||||
logger.info(f"Total documents found: {len(all_decisions)}")
|
||||
|
||||
# Step 2: Fetch document content and process
|
||||
logger.info("Step 2: Fetching and processing document content...")
|
||||
|
||||
documents_data = []
|
||||
failed_fetches = 0
|
||||
|
||||
# Process up to 100 documents total
|
||||
decisions_to_process = all_decisions[:100]
|
||||
|
||||
for i, decision in enumerate(decisions_to_process):
|
||||
try:
|
||||
# Fetch document content
|
||||
doc = await bedesten_client_instance.get_document_as_markdown(decision.documentId)
|
||||
|
||||
if doc.markdown_content:
|
||||
# Process document into chunks
|
||||
metadata = {
|
||||
"document_id": decision.documentId,
|
||||
"birim_adi": decision.birimAdi,
|
||||
"esas_no": decision.esasNo,
|
||||
"karar_no": decision.kararNo,
|
||||
"karar_tarihi": decision.kararTarihiStr,
|
||||
"court_type": decision.itemType.name if decision.itemType else None
|
||||
}
|
||||
|
||||
chunks = processor.process_document(
|
||||
document_id=decision.documentId,
|
||||
text=doc.markdown_content,
|
||||
metadata=metadata
|
||||
)
|
||||
|
||||
# For now, use the full document as one chunk (can be optimized later)
|
||||
if chunks:
|
||||
full_text = " ".join([chunk.text for chunk in chunks])
|
||||
documents_data.append({
|
||||
"id": decision.documentId,
|
||||
"text": full_text[:3000], # Limit text for embedding
|
||||
"metadata": metadata
|
||||
})
|
||||
|
||||
# Log progress every 10 documents
|
||||
if (i + 1) % 10 == 0:
|
||||
logger.info(f"Processed {i + 1}/{len(decisions_to_process)} documents")
|
||||
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to fetch document {decision.documentId}: {e}")
|
||||
failed_fetches += 1
|
||||
|
||||
if not documents_data:
|
||||
logger.warning("No documents could be processed")
|
||||
return {
|
||||
"status": "processing_error",
|
||||
"message": "Could not process any documents",
|
||||
"results": []
|
||||
}
|
||||
|
||||
logger.info(f"Successfully processed {len(documents_data)} documents, {failed_fetches} failed")
|
||||
|
||||
# Step 3: Generate embeddings
|
||||
logger.info("Step 3: Generating embeddings...")
|
||||
|
||||
# Generate query embedding
|
||||
query_embedding = embedder.encode_query(query, task="search result")
|
||||
|
||||
# Generate document embeddings
|
||||
doc_texts = [doc["text"] for doc in documents_data]
|
||||
doc_titles = [doc["metadata"].get("birim_adi", "none") for doc in documents_data]
|
||||
doc_embeddings = embedder.encode_documents(doc_texts, titles=doc_titles)
|
||||
|
||||
# Always reduce to 256 dimensions for optimal speed/quality balance
|
||||
query_embedding = embedder.reduce_dimensions(query_embedding, 256)
|
||||
doc_embeddings = embedder.reduce_dimensions(doc_embeddings, 256)
|
||||
|
||||
# Step 4: Add to vector store and search
|
||||
logger.info("Step 4: Performing semantic search...")
|
||||
|
||||
# Add documents to vector store
|
||||
doc_ids = [doc["id"] for doc in documents_data]
|
||||
doc_metadatas = [doc["metadata"] for doc in documents_data]
|
||||
|
||||
vector_store.add_documents(
|
||||
ids=doc_ids,
|
||||
texts=doc_texts,
|
||||
embeddings=doc_embeddings,
|
||||
metadata=doc_metadatas
|
||||
)
|
||||
|
||||
# Perform semantic search
|
||||
search_results = vector_store.search(
|
||||
query_embedding=query_embedding,
|
||||
top_k=top_k,
|
||||
threshold=0.3 # Minimum similarity threshold
|
||||
)
|
||||
|
||||
# Step 5: Format results
|
||||
logger.info(f"Step 5: Formatting {len(search_results)} results")
|
||||
|
||||
formatted_results = []
|
||||
for doc, score in search_results:
|
||||
# Build title from metadata
|
||||
title_parts = []
|
||||
if doc.metadata.get("birim_adi"):
|
||||
title_parts.append(doc.metadata["birim_adi"])
|
||||
if doc.metadata.get("esas_no"):
|
||||
title_parts.append(f"Esas: {doc.metadata['esas_no']}")
|
||||
if doc.metadata.get("karar_no"):
|
||||
title_parts.append(f"Karar: {doc.metadata['karar_no']}")
|
||||
if doc.metadata.get("karar_tarihi"):
|
||||
title_parts.append(f"Tarih: {doc.metadata['karar_tarihi']}")
|
||||
|
||||
title = " - ".join(title_parts) if title_parts else f"Document {doc.id}"
|
||||
|
||||
formatted_results.append({
|
||||
"document_id": doc.id,
|
||||
"title": title,
|
||||
"similarity_score": float(score),
|
||||
"preview": doc.text[:500] + "..." if len(doc.text) > 500 else doc.text,
|
||||
"metadata": doc.metadata,
|
||||
"source_url": f"https://mevzuat.adalet.gov.tr/ictihat/{doc.id}"
|
||||
})
|
||||
|
||||
# Get vector store stats
|
||||
stats = vector_store.get_stats()
|
||||
|
||||
return {
|
||||
"status": "success",
|
||||
"query": query,
|
||||
"initial_keyword": initial_keyword,
|
||||
"total_documents_processed": len(documents_data),
|
||||
"embedding_dimension": 256, # Fixed at 256 for optimal performance
|
||||
"results": formatted_results,
|
||||
"stats": {
|
||||
"documents_in_store": stats["num_documents"],
|
||||
"memory_usage_mb": round(stats["memory_usage_mb"], 2),
|
||||
"failed_fetches": failed_fetches
|
||||
}
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in semantic search: {e}")
|
||||
return {
|
||||
"status": "error",
|
||||
"message": str(e),
|
||||
"results": []
|
||||
}
|
||||
|
||||
# --- ChatGPT Deep Research Compatible Tools ---
|
||||
|
||||
def get_preview_text(markdown_content: str, skip_chars: int = 100, preview_chars: int = 200) -> str:
|
||||
@@ -2012,7 +2216,7 @@ async def search(
|
||||
logger.info(f"ChatGPT Deep Research search completed. Found {len(results)} results via Bedesten API.")
|
||||
return {"results": results}
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in ChatGPT Deep Research search tool")
|
||||
# Return partial results if any were found
|
||||
if results:
|
||||
@@ -2141,7 +2345,7 @@ async def fetch(
|
||||
doc = await bedesten_client_instance.get_document_as_markdown(doc_id)
|
||||
"""
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception(f"Error fetching ChatGPT Deep Research document {id}")
|
||||
raise
|
||||
|
||||
@@ -2192,7 +2396,7 @@ def main():
|
||||
app.run()
|
||||
except KeyboardInterrupt:
|
||||
logger.info("Server shut down by user (KeyboardInterrupt).")
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Server failed to start or crashed.")
|
||||
finally:
|
||||
logger.info(f"{app.name} server has shut down.")
|
||||
|
||||
@@ -31,6 +31,11 @@ dependencies = [
|
||||
"fastapi>=0.115.14",
|
||||
"PyJWT>=2.8.0",
|
||||
"tiktoken>=0.5.0",
|
||||
"sentence-transformers>=3.0.0",
|
||||
"torch>=2.0.0",
|
||||
"numpy>=1.24.0",
|
||||
"scikit-learn>=1.3.0",
|
||||
"transformers>=4.57.0.dev0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
|
||||
@@ -11,8 +11,8 @@ import os
|
||||
import json
|
||||
import time
|
||||
import logging
|
||||
from typing import Optional, Dict, Any, Union
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Optional, Dict, Any
|
||||
from datetime import datetime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -4,10 +4,9 @@ import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple, Dict, Any
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io # For io.BytesIO
|
||||
from urllib.parse import urlencode, urljoin, quote, parse_qs, urlparse
|
||||
from urllib.parse import urljoin, parse_qs, urlparse
|
||||
from markitdown import MarkItDown
|
||||
import math
|
||||
|
||||
@@ -18,8 +17,7 @@ from .models import (
|
||||
RekabetKurumuSearchRequest,
|
||||
RekabetDecisionSummary,
|
||||
RekabetSearchResult,
|
||||
RekabetDocument,
|
||||
RekabetKararTuruGuidEnum
|
||||
RekabetDocument
|
||||
)
|
||||
from pydantic import HttpUrl # Ensure HttpUrl is imported from pydantic
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# rekabet_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Any
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
|
||||
# Enum for decision type GUIDs (used by the client and expected by the website)
|
||||
|
||||
+1
-1
@@ -93,7 +93,7 @@ def main():
|
||||
config["workers"] = args.workers
|
||||
|
||||
# Print startup information
|
||||
print(f"Starting Yargı MCP server...")
|
||||
print("Starting Yargı MCP server...")
|
||||
print(f"Host: {args.host}")
|
||||
print(f"Port: {args.port}")
|
||||
print(f"Transport: {args.transport}")
|
||||
|
||||
@@ -2,13 +2,13 @@
|
||||
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup, Tag
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from urllib.parse import urljoin
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
|
||||
|
||||
@@ -3,12 +3,12 @@
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from urllib.parse import urljoin
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
|
||||
|
||||
@@ -2,7 +2,6 @@
|
||||
# Unified client for both Norm Denetimi and Bireysel Başvuru
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .models import (
|
||||
|
||||
@@ -10,16 +10,12 @@ Usage:
|
||||
"""
|
||||
|
||||
import os
|
||||
import time
|
||||
import logging
|
||||
from datetime import datetime, timedelta
|
||||
from fastapi import FastAPI, Request, HTTPException, Query
|
||||
from fastapi.responses import JSONResponse, HTMLResponse
|
||||
from fastapi.exception_handlers import http_exception_handler
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.responses import Response
|
||||
from starlette.requests import Request as StarletteRequest
|
||||
|
||||
# Import the MCP app creator function
|
||||
from mcp_server_main import create_app
|
||||
@@ -103,7 +99,6 @@ mcp_app = mcp_server.http_app(path="/")
|
||||
|
||||
# Configure JSON encoder for proper Turkish character support
|
||||
import json
|
||||
from fastapi.responses import JSONResponse
|
||||
|
||||
class UTF8JSONResponse(JSONResponse):
|
||||
def __init__(self, content=None, status_code=200, headers=None, **kwargs):
|
||||
@@ -461,14 +456,14 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
logger.info("User authenticated successfully via Clerk")
|
||||
|
||||
# Return success response
|
||||
return HTMLResponse(f"""
|
||||
return HTMLResponse("""
|
||||
<html>
|
||||
<head>
|
||||
<title>MCP Connection Successful</title>
|
||||
<style>
|
||||
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
|
||||
.success {{ color: #28a745; }}
|
||||
.token {{ background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }}
|
||||
body { font-family: Arial, sans-serif; text-align: center; padding: 50px; }
|
||||
.success { color: #28a745; }
|
||||
.token { background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
@@ -481,13 +476,13 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
<p>You can now close this window and return to your MCP client.</p>
|
||||
<script>
|
||||
// Try to close the popup if opened as such
|
||||
if (window.opener) {{
|
||||
window.opener.postMessage({{
|
||||
if (window.opener) {
|
||||
window.opener.postMessage({
|
||||
type: 'MCP_AUTH_SUCCESS',
|
||||
token: 'use_clerk_jwt_token'
|
||||
}}, '*');
|
||||
}, '*');
|
||||
setTimeout(() => window.close(), 3000);
|
||||
}}
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
|
||||
@@ -1,13 +1,12 @@
|
||||
# bddk_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from typing import List, Optional, Dict, Any
|
||||
from typing import Optional
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urlparse
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# bddk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional
|
||||
from typing import List
|
||||
|
||||
class BddkSearchRequest(BaseModel):
|
||||
"""
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
# bedesten_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional, Dict, Any, Literal, Union
|
||||
from datetime import datetime
|
||||
from typing import List, Optional, Dict, Any, Literal
|
||||
|
||||
# Import compressed BirimAdiEnum for chamber filtering
|
||||
from .enums import BirimAdiEnum
|
||||
|
||||
@@ -1,11 +1,9 @@
|
||||
# danistay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Dict, List, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
|
||||
@@ -2,10 +2,9 @@
|
||||
|
||||
import httpx
|
||||
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Dict, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
|
||||
@@ -7,11 +7,9 @@ import os
|
||||
from typing import List, Dict, Any, Optional
|
||||
from datetime import datetime
|
||||
|
||||
from fastapi import FastAPI, HTTPException, Query, Depends, Body
|
||||
from fastapi import FastAPI, HTTPException, Query
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import JSONResponse
|
||||
from pydantic import BaseModel, Field
|
||||
import json
|
||||
|
||||
# Import the main MCP app
|
||||
from mcp_server_main import app as mcp_server
|
||||
|
||||
@@ -10,13 +10,12 @@ from playwright.async_api import (
|
||||
)
|
||||
from bs4 import BeautifulSoup
|
||||
import logging
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import List, Optional
|
||||
import urllib.parse
|
||||
import base64 # Base64 için
|
||||
import re
|
||||
import html as html_parser
|
||||
from markitdown import MarkItDown
|
||||
import os
|
||||
import math
|
||||
import io
|
||||
import random
|
||||
@@ -907,7 +906,7 @@ class KikApiClient:
|
||||
event_target_for_submit = self.FIELD_LOCATORS['search_button_id']
|
||||
# Use human-like clicking for search button
|
||||
search_button_selector = f"a[id='{event_target_for_submit}']"
|
||||
logger.info(f"Performing human-like search button click...")
|
||||
logger.info("Performing human-like search button click...")
|
||||
|
||||
try:
|
||||
# Hide datepicker first to prevent interference
|
||||
@@ -1181,7 +1180,7 @@ class KikApiClient:
|
||||
try:
|
||||
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=2000):
|
||||
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
|
||||
await current_main_page.wait_for_selector(f"div#detayPopUp:not(.in)", timeout=5000)
|
||||
await current_main_page.wait_for_selector("div#detayPopUp:not(.in)", timeout=5000)
|
||||
except: pass
|
||||
|
||||
return KikDocumentMarkdown(
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
# kik_mcp_module/models.py
|
||||
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
|
||||
from pydantic import BaseModel, Field, computed_field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
import base64 # Base64 encoding/decoding için
|
||||
|
||||
@@ -2,13 +2,13 @@
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Dict, Any
|
||||
from typing import Optional, Dict, Any
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urljoin, urlparse, parse_qs
|
||||
from urllib.parse import urlparse
|
||||
from markitdown import MarkItDown
|
||||
from pydantic import HttpUrl
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# kvkk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Any
|
||||
from typing import List, Optional
|
||||
|
||||
class KvkkSearchRequest(BaseModel):
|
||||
"""Model for KVKK (Personal Data Protection Authority) search request via Brave API."""
|
||||
|
||||
@@ -36,7 +36,7 @@ def create_clerk_oauth_config() -> OAuthConfig:
|
||||
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
|
||||
)
|
||||
|
||||
logger.info(f"Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info("Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info(f"Clerk domain: {clerk_domain}")
|
||||
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
|
||||
logger.debug(f"Token endpoint: {config.token_endpoint}")
|
||||
|
||||
@@ -9,7 +9,7 @@ import time
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Any, Optional
|
||||
from typing import Any
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import httpx
|
||||
|
||||
@@ -18,7 +18,7 @@ from fastapi.responses import RedirectResponse, JSONResponse
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError as e:
|
||||
except ImportError:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@ Uses Redis for authorization code storage to support multi-machine deployment
|
||||
import os
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlencode, quote
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from fastapi import APIRouter, Request, Query, HTTPException
|
||||
from fastapi.responses import RedirectResponse, JSONResponse
|
||||
@@ -39,7 +39,6 @@ def get_redis_session_store():
|
||||
if redis_store is None:
|
||||
try:
|
||||
import concurrent.futures
|
||||
import functools
|
||||
|
||||
# Use thread pool with timeout to prevent hanging
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
|
||||
@@ -206,7 +205,6 @@ async def oauth_callback(
|
||||
auth_code = f"clerk_auth_{os.urandom(16).hex()}"
|
||||
|
||||
# Prepare code data
|
||||
import time
|
||||
code_data = {
|
||||
"user_id": user_id,
|
||||
"session_id": session_id,
|
||||
@@ -225,7 +223,7 @@ async def oauth_callback(
|
||||
if success:
|
||||
logger.info(f"Stored authorization code {auth_code[:10]}... in Redis with real JWT token")
|
||||
else:
|
||||
logger.error(f"Failed to store authorization code in Redis, falling back to in-memory")
|
||||
logger.error("Failed to store authorization code in Redis, falling back to in-memory")
|
||||
# Fall back to in-memory storage
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
@@ -236,7 +234,7 @@ async def oauth_callback(
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
oauth_callback._code_storage[auth_code] = code_data
|
||||
logger.info(f"Stored authorization code in memory (fallback)")
|
||||
logger.info("Stored authorization code in memory (fallback)")
|
||||
|
||||
# Redirect back to client with authorization code
|
||||
redirect_params = {
|
||||
|
||||
@@ -8,8 +8,7 @@ import json
|
||||
import time
|
||||
from collections import defaultdict
|
||||
from pydantic import HttpUrl, Field
|
||||
from typing import Optional, Dict, List, Literal, Any, Union
|
||||
import urllib.parse
|
||||
from typing import Dict, List, Literal, Any
|
||||
import tiktoken
|
||||
from fastmcp.server.middleware import Middleware, MiddlewareContext
|
||||
from fastmcp.server.dependencies import get_access_token, AccessToken
|
||||
@@ -159,7 +158,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("tool_call_error", input_tokens, 0,
|
||||
tool_name, duration_ms)
|
||||
@@ -189,7 +188,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("resource_read_error", 0, 0,
|
||||
resource_uri, duration_ms)
|
||||
@@ -219,7 +218,7 @@ class TokenCountingMiddleware(Middleware):
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
duration_ms = (time.perf_counter() - start_time) * 1000
|
||||
self.log_token_usage("prompt_get_error", 0, 0,
|
||||
prompt_name, duration_ms)
|
||||
@@ -260,10 +259,6 @@ def create_app(auth=None):
|
||||
|
||||
# --- Module Imports ---
|
||||
from yargitay_mcp_module.client import YargitayOfficialApiClient
|
||||
from yargitay_mcp_module.models import (
|
||||
YargitayDetailedSearchRequest, YargitayDocumentMarkdown, CompactYargitaySearchResult,
|
||||
YargitayBirimEnum, CleanYargitayDecisionEntry
|
||||
)
|
||||
from bedesten_mcp_module.client import BedestenApiClient
|
||||
from bedesten_mcp_module.models import (
|
||||
BedestenSearchRequest, BedestenSearchData,
|
||||
@@ -271,10 +266,6 @@ from bedesten_mcp_module.models import (
|
||||
)
|
||||
from bedesten_mcp_module.enums import BirimAdiEnum
|
||||
from danistay_mcp_module.client import DanistayApiClient
|
||||
from danistay_mcp_module.models import (
|
||||
DanistayKeywordSearchRequest, DanistayDetailedSearchRequest,
|
||||
DanistayDocumentMarkdown, CompactDanistaySearchResult
|
||||
)
|
||||
from emsal_mcp_module.client import EmsalApiClient
|
||||
from emsal_mcp_module.models import (
|
||||
EmsalSearchRequest, EmsalDocumentMarkdown, CompactEmsalSearchResult
|
||||
@@ -288,15 +279,7 @@ from anayasa_mcp_module.client import AnayasaMahkemesiApiClient
|
||||
from anayasa_mcp_module.bireysel_client import AnayasaBireyselBasvuruApiClient
|
||||
from anayasa_mcp_module.unified_client import AnayasaUnifiedClient
|
||||
from anayasa_mcp_module.models import (
|
||||
AnayasaNormDenetimiSearchRequest,
|
||||
AnayasaSearchResult,
|
||||
AnayasaDocumentMarkdown,
|
||||
AnayasaBireyselReportSearchRequest,
|
||||
AnayasaBireyselReportSearchResult,
|
||||
AnayasaBireyselBasvuruDocumentMarkdown,
|
||||
AnayasaUnifiedSearchRequest,
|
||||
AnayasaUnifiedSearchResult,
|
||||
AnayasaUnifiedDocumentMarkdown,
|
||||
# Removed enum imports - now using Literal strings in models
|
||||
)
|
||||
# KIK Module Imports
|
||||
@@ -318,14 +301,9 @@ from rekabet_mcp_module.models import (
|
||||
|
||||
from sayistay_mcp_module.client import SayistayApiClient
|
||||
from sayistay_mcp_module.models import (
|
||||
GenelKurulSearchRequest, GenelKurulSearchResponse,
|
||||
TemyizKuruluSearchRequest, TemyizKuruluSearchResponse,
|
||||
DaireSearchRequest, DaireSearchResponse,
|
||||
SayistayDocumentMarkdown,
|
||||
SayistayUnifiedSearchRequest, SayistayUnifiedSearchResult,
|
||||
SayistayUnifiedDocumentMarkdown
|
||||
)
|
||||
from sayistay_mcp_module.enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
from sayistay_mcp_module.unified_client import SayistayUnifiedClient
|
||||
|
||||
# KVKK Module Imports
|
||||
@@ -339,14 +317,11 @@ from kvkk_mcp_module.models import (
|
||||
# BDDK Module Imports
|
||||
from bddk_mcp_module.client import BddkApiClient
|
||||
from bddk_mcp_module.models import (
|
||||
BddkSearchRequest,
|
||||
BddkSearchResult,
|
||||
BddkDocumentMarkdown
|
||||
BddkSearchRequest
|
||||
)
|
||||
|
||||
|
||||
# Create a placeholder app that will be properly initialized after tools are defined
|
||||
from fastmcp import FastMCP
|
||||
|
||||
# Placeholder app for decorators - will be replaced in create_app() after all tools are defined
|
||||
app = FastMCP("Yargı MCP Server Placeholder")
|
||||
@@ -1372,7 +1347,7 @@ async def search_emsal_detailed_decisions(
|
||||
page_size=page_size
|
||||
)
|
||||
|
||||
logger.info(f"Tool 'search_emsal_detailed_decisions' called.")
|
||||
logger.info("Tool 'search_emsal_detailed_decisions' called.")
|
||||
try:
|
||||
api_response = await emsal_client_instance.search_detailed_decisions(search_query)
|
||||
if api_response.data:
|
||||
@@ -1384,8 +1359,8 @@ async def search_emsal_detailed_decisions(
|
||||
)
|
||||
logger.warning("API response for Emsal search did not contain expected data structure.")
|
||||
return CompactEmsalSearchResult(decisions=[], total_records=0, requested_page=search_query.page_number, page_size=search_query.page_size)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_emsal_detailed_decisions'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_emsal_detailed_decisions'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -1401,8 +1376,8 @@ async def get_emsal_document_markdown(id: str) -> EmsalDocumentMarkdown:
|
||||
if not id or not id.strip(): raise ValueError("Document ID required for Emsal.")
|
||||
try:
|
||||
return await emsal_client_instance.get_decision_document_as_markdown(id)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_emsal_document_markdown'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_emsal_document_markdown'.")
|
||||
raise
|
||||
|
||||
# --- MCP Tools for Uyusmazlik ---
|
||||
@@ -1470,11 +1445,11 @@ async def search_uyusmazlik_decisions(
|
||||
not_hepsi=not_hepsi
|
||||
)
|
||||
|
||||
logger.info(f"Tool 'search_uyusmazlik_decisions' called.")
|
||||
logger.info("Tool 'search_uyusmazlik_decisions' called.")
|
||||
try:
|
||||
return await uyusmazlik_client_instance.search_decisions(search_params)
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_uyusmazlik_decisions'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_uyusmazlik_decisions'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -1493,8 +1468,8 @@ async def get_uyusmazlik_document_markdown_from_url(
|
||||
raise ValueError("Document URL (document_url) is required for Uyuşmazlık document retrieval.")
|
||||
try:
|
||||
return await uyusmazlik_client_instance.get_decision_document_as_markdown(str(document_url))
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_uyusmazlik_document_markdown_from_url'.")
|
||||
raise
|
||||
|
||||
# --- DEACTIVATED: MCP Tools for Anayasa Mahkemesi (Individual Tools) ---
|
||||
@@ -1585,8 +1560,8 @@ async def search_anayasa_unified(
|
||||
result = await anayasa_unified_client_instance.search_unified(request)
|
||||
return json.dumps(result.model_dump(), ensure_ascii=False, indent=2)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'search_anayasa_unified'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_anayasa_unified'.")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
@@ -1607,8 +1582,8 @@ async def get_anayasa_document_unified(
|
||||
result = await anayasa_unified_client_instance.get_document_unified(document_url, page_number)
|
||||
return json.dumps(result.model_dump(mode='json'), ensure_ascii=False, indent=2)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in tool 'get_anayasa_document_unified'.")
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_anayasa_document_unified'.")
|
||||
raise
|
||||
|
||||
# --- MCP Tools for KIK (Kamu İhale Kurulu) ---
|
||||
@@ -1654,15 +1629,15 @@ async def search_kik_decisions(
|
||||
page=page
|
||||
)
|
||||
|
||||
logger.info(f"Tool 'search_kik_decisions' called.")
|
||||
logger.info("Tool 'search_kik_decisions' called.")
|
||||
try:
|
||||
api_response = await kik_client_instance.search_decisions(search_query)
|
||||
page_param_for_log = search_query.page if hasattr(search_query, 'page') else 1
|
||||
if not api_response.decisions and api_response.total_records == 0 and page_param_for_log == 1:
|
||||
logger.warning(f"KIK search returned no decisions for query.")
|
||||
logger.warning("KIK search returned no decisions for query.")
|
||||
return api_response
|
||||
except Exception as e:
|
||||
logger.exception(f"Error in KIK search tool 'search_kik_decisions'.")
|
||||
except Exception:
|
||||
logger.exception("Error in KIK search tool 'search_kik_decisions'.")
|
||||
current_page_val = search_query.page if hasattr(search_query, 'page') else 1
|
||||
return KikSearchResult(decisions=[], total_records=0, current_page=current_page_val)
|
||||
|
||||
@@ -1758,7 +1733,7 @@ async def search_rekabet_kurumu_decisions(
|
||||
try:
|
||||
|
||||
return await rekabet_client_instance.search_decisions(search_query)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_rekabet_kurumu_decisions'.")
|
||||
return RekabetSearchResult(decisions=[], retrieved_page_number=page, total_records_found=0, total_pages=0)
|
||||
|
||||
@@ -1781,7 +1756,7 @@ async def get_rekabet_kurumu_document(
|
||||
try:
|
||||
|
||||
return await rekabet_client_instance.get_decision_document(karar_id, page_number=current_page_to_fetch)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception(f"Error in tool 'get_rekabet_kurumu_document'. Karar ID: {karar_id}")
|
||||
raise
|
||||
|
||||
@@ -1876,7 +1851,7 @@ For best results, use exact phrases with quotes for legal terms."""),
|
||||
"page_size": pageSize,
|
||||
"searched_courts": court_types
|
||||
}
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_bedesten_unified'")
|
||||
raise
|
||||
|
||||
@@ -1898,7 +1873,7 @@ async def get_bedesten_document_markdown(
|
||||
|
||||
try:
|
||||
return await bedesten_client_instance.get_document_as_markdown(documentId)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_kyb_bedesten_document_markdown'")
|
||||
raise
|
||||
|
||||
@@ -2087,7 +2062,7 @@ async def search_sayistay_unified(
|
||||
web_karar_metni=web_karar_metni
|
||||
)
|
||||
return await sayistay_unified_client_instance.search_unified(search_request)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'search_sayistay_unified'")
|
||||
raise
|
||||
|
||||
@@ -2111,7 +2086,7 @@ async def get_sayistay_document_unified(
|
||||
|
||||
try:
|
||||
return await sayistay_unified_client_instance.get_document_unified(decision_id, decision_type)
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in tool 'get_sayistay_document_unified'")
|
||||
raise
|
||||
|
||||
@@ -2698,7 +2673,7 @@ async def search(
|
||||
logger.info(f"ChatGPT Deep Research search completed. Found {len(results)} results via Bedesten API.")
|
||||
return {"results": results}
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Error in ChatGPT Deep Research search tool")
|
||||
# Return partial results if any were found
|
||||
if results:
|
||||
@@ -2827,7 +2802,7 @@ async def fetch(
|
||||
doc = await bedesten_client_instance.get_document_as_markdown(doc_id)
|
||||
"""
|
||||
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception(f"Error fetching ChatGPT Deep Research document {id}")
|
||||
raise
|
||||
|
||||
@@ -2874,7 +2849,7 @@ def main():
|
||||
app.run()
|
||||
except KeyboardInterrupt:
|
||||
logger.info("Server shut down by user (KeyboardInterrupt).")
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
logger.exception("Server failed to start or crashed.")
|
||||
finally:
|
||||
logger.info(f"{app.name} server has shut down.")
|
||||
|
||||
@@ -11,8 +11,8 @@ import os
|
||||
import json
|
||||
import time
|
||||
import logging
|
||||
from typing import Optional, Dict, Any, Union
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Optional, Dict, Any
|
||||
from datetime import datetime
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -4,10 +4,9 @@ import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple, Dict, Any
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io # For io.BytesIO
|
||||
from urllib.parse import urlencode, urljoin, quote, parse_qs, urlparse
|
||||
from urllib.parse import urljoin, parse_qs, urlparse
|
||||
from markitdown import MarkItDown
|
||||
import math
|
||||
|
||||
@@ -18,8 +17,7 @@ from .models import (
|
||||
RekabetKurumuSearchRequest,
|
||||
RekabetDecisionSummary,
|
||||
RekabetSearchResult,
|
||||
RekabetDocument,
|
||||
RekabetKararTuruGuidEnum
|
||||
RekabetDocument
|
||||
)
|
||||
from pydantic import HttpUrl # Ensure HttpUrl is imported from pydantic
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# rekabet_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Any
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
|
||||
# Enum for decision type GUIDs (used by the client and expected by the website)
|
||||
|
||||
@@ -93,7 +93,7 @@ def main():
|
||||
config["workers"] = args.workers
|
||||
|
||||
# Print startup information
|
||||
print(f"Starting Yargı MCP server...")
|
||||
print("Starting Yargı MCP server...")
|
||||
print(f"Host: {args.host}")
|
||||
print(f"Port: {args.port}")
|
||||
print(f"Transport: {args.transport}")
|
||||
|
||||
@@ -1,13 +1,11 @@
|
||||
# sayistay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
import re
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin
|
||||
from urllib.parse import urlencode
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
@@ -16,7 +14,7 @@ from .models import (
|
||||
DaireSearchRequest, DaireSearchResponse, DaireDecision,
|
||||
SayistayDocumentMarkdown
|
||||
)
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum, WEB_KARAR_KONUSU_MAPPING
|
||||
from .enums import WEB_KARAR_KONUSU_MAPPING
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# sayistay_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import Optional, List, Union, Dict, Any, Literal
|
||||
from typing import Optional, List, Dict, Any, Literal
|
||||
from enum import Enum
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
|
||||
|
||||
@@ -2,8 +2,6 @@
|
||||
# Unified client for all three Sayıştay decision types
|
||||
|
||||
import logging
|
||||
from typing import Optional, Dict, Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .models import (
|
||||
SayistayUnifiedSearchRequest,
|
||||
|
||||
@@ -13,7 +13,7 @@ import os
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
|
||||
from starlette.responses import JSONResponse
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.middleware.authentication import AuthenticationMiddleware
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import os, stripe
|
||||
import os
|
||||
import stripe
|
||||
from clerk_backend_api import Clerk # Clerk backend SDK
|
||||
from fastapi import APIRouter, Request, HTTPException
|
||||
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
import httpx
|
||||
import aiohttp
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Union, Tuple
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
|
||||
@@ -1,20 +1,16 @@
|
||||
# yargitay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
YargitayDetailedSearchRequest,
|
||||
YargitayApiSearchResponse,
|
||||
YargitayApiDecisionEntry,
|
||||
YargitayDocumentMarkdown,
|
||||
CompactYargitaySearchResult
|
||||
YargitayDocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# yargitay_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl, ConfigDict
|
||||
from typing import List, Optional, Dict, Any, Literal
|
||||
from typing import List, Optional, Literal
|
||||
|
||||
# Yargıtay Chamber/Board Options
|
||||
YargitayBirimEnum = Literal[
|
||||
|
||||
@@ -1,13 +1,11 @@
|
||||
# sayistay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
import re
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
from typing import Dict, List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin
|
||||
from urllib.parse import urlencode
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
@@ -16,7 +14,7 @@ from .models import (
|
||||
DaireSearchRequest, DaireSearchResponse, DaireDecision,
|
||||
SayistayDocumentMarkdown
|
||||
)
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum, WEB_KARAR_KONUSU_MAPPING
|
||||
from .enums import WEB_KARAR_KONUSU_MAPPING
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# sayistay_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import Optional, List, Union, Dict, Any, Literal
|
||||
from typing import Optional, List, Dict, Any, Literal
|
||||
from enum import Enum
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
|
||||
|
||||
@@ -2,8 +2,6 @@
|
||||
# Unified client for all three Sayıştay decision types
|
||||
|
||||
import logging
|
||||
from typing import Optional, Dict, Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .models import (
|
||||
SayistayUnifiedSearchRequest,
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
# semantic_search/__init__.py
|
||||
|
||||
from .embedder import EmbeddingGemma
|
||||
from .vector_store import VectorStore
|
||||
from .processor import DocumentProcessor
|
||||
|
||||
__all__ = ['EmbeddingGemma', 'VectorStore', 'DocumentProcessor']
|
||||
@@ -0,0 +1,173 @@
|
||||
# semantic_search/embedder.py
|
||||
|
||||
import logging
|
||||
from typing import List, Optional
|
||||
import numpy as np
|
||||
from sentence_transformers import SentenceTransformer
|
||||
import torch
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class EmbeddingGemma:
|
||||
"""
|
||||
Wrapper for Google's EmbeddingGemma model.
|
||||
Handles query and document encoding with proper prompt templates.
|
||||
"""
|
||||
|
||||
def __init__(self, model_name: str = "google/embeddinggemma-300m", device: Optional[str] = None):
|
||||
"""
|
||||
Initialize EmbeddingGemma model.
|
||||
|
||||
Args:
|
||||
model_name: HuggingFace model name
|
||||
device: Device to run model on ('cuda', 'cpu', or None for auto)
|
||||
"""
|
||||
self.model_name = model_name
|
||||
|
||||
# Auto-detect device if not specified
|
||||
if device is None:
|
||||
self.device = 'cuda' if torch.cuda.is_available() else 'cpu'
|
||||
else:
|
||||
self.device = device
|
||||
|
||||
logger.info(f"Initializing EmbeddingGemma on device: {self.device}")
|
||||
|
||||
try:
|
||||
# Load model with float32 precision (EmbeddingGemma doesn't support float16)
|
||||
self.model = SentenceTransformer(model_name, device=self.device)
|
||||
self.model.eval() # Set to evaluation mode
|
||||
|
||||
# Set precision to float32 or bfloat16
|
||||
if self.device == 'cuda' and torch.cuda.is_bf16_supported():
|
||||
logger.info("Using bfloat16 precision for CUDA")
|
||||
self.dtype = torch.bfloat16
|
||||
else:
|
||||
logger.info("Using float32 precision")
|
||||
self.dtype = torch.float32
|
||||
|
||||
logger.info(f"Successfully loaded model: {model_name}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to load EmbeddingGemma model: {e}")
|
||||
raise
|
||||
|
||||
def encode_query(self, query: str, task: str = "search result") -> np.ndarray:
|
||||
"""
|
||||
Encode a search query with appropriate prompt template.
|
||||
|
||||
Args:
|
||||
query: The search query text
|
||||
task: Task type for prompt template (search result, question answering, etc.)
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (768 dimensions)
|
||||
"""
|
||||
# Apply query prompt template
|
||||
prompted_query = f"task: {task} | query: {query}"
|
||||
|
||||
try:
|
||||
with torch.no_grad():
|
||||
# Encode with model
|
||||
embeddings = self.model.encode(
|
||||
prompted_query,
|
||||
convert_to_numpy=True,
|
||||
normalize_embeddings=True, # L2 normalization for cosine similarity
|
||||
show_progress_bar=False
|
||||
)
|
||||
|
||||
logger.debug(f"Encoded query: {query[:50]}... -> shape: {embeddings.shape}")
|
||||
return embeddings
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode query: {e}")
|
||||
raise
|
||||
|
||||
def encode_documents(self, documents: List[str], titles: Optional[List[str]] = None) -> np.ndarray:
|
||||
"""
|
||||
Encode multiple documents with appropriate prompt template.
|
||||
|
||||
Args:
|
||||
documents: List of document texts
|
||||
titles: Optional list of document titles
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (N x 768 dimensions)
|
||||
"""
|
||||
if not documents:
|
||||
return np.array([])
|
||||
|
||||
# Apply document prompt template
|
||||
prompted_docs = []
|
||||
for i, doc in enumerate(documents):
|
||||
title = titles[i] if titles and i < len(titles) else "none"
|
||||
prompted_doc = f"title: {title} | text: {doc}"
|
||||
prompted_docs.append(prompted_doc)
|
||||
|
||||
try:
|
||||
with torch.no_grad():
|
||||
# Batch encode documents
|
||||
embeddings = self.model.encode(
|
||||
prompted_docs,
|
||||
convert_to_numpy=True,
|
||||
normalize_embeddings=True,
|
||||
show_progress_bar=len(documents) > 10,
|
||||
batch_size=8 # Adjust based on memory
|
||||
)
|
||||
|
||||
logger.info(f"Encoded {len(documents)} documents -> shape: {embeddings.shape}")
|
||||
return embeddings
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode documents: {e}")
|
||||
raise
|
||||
|
||||
def reduce_dimensions(self, embeddings: np.ndarray, target_dim: int = 512) -> np.ndarray:
|
||||
"""
|
||||
Reduce embedding dimensions using Matryoshka Representation Learning.
|
||||
|
||||
Args:
|
||||
embeddings: Original embeddings (N x 768)
|
||||
target_dim: Target dimension (512, 256, or 128)
|
||||
|
||||
Returns:
|
||||
Reduced embeddings (N x target_dim)
|
||||
"""
|
||||
if target_dim not in [512, 256, 128]:
|
||||
raise ValueError(f"Target dimension must be 512, 256, or 128, got {target_dim}")
|
||||
|
||||
if len(embeddings.shape) == 1:
|
||||
# Single embedding
|
||||
reduced = embeddings[:target_dim]
|
||||
# Re-normalize after truncation
|
||||
norm = np.linalg.norm(reduced)
|
||||
if norm > 0:
|
||||
reduced = reduced / norm
|
||||
else:
|
||||
# Multiple embeddings
|
||||
reduced = embeddings[:, :target_dim]
|
||||
# Re-normalize each embedding
|
||||
norms = np.linalg.norm(reduced, axis=1, keepdims=True)
|
||||
reduced = reduced / (norms + 1e-8) # Avoid division by zero
|
||||
|
||||
logger.debug(f"Reduced dimensions: {embeddings.shape} -> {reduced.shape}")
|
||||
return reduced
|
||||
|
||||
def compute_similarity(self, query_embedding: np.ndarray, document_embeddings: np.ndarray) -> np.ndarray:
|
||||
"""
|
||||
Compute cosine similarity between query and documents.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding (768,)
|
||||
document_embeddings: Document embeddings (N x 768)
|
||||
|
||||
Returns:
|
||||
Similarity scores (N,)
|
||||
"""
|
||||
# Ensure query is 2D for matrix multiplication
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarity (embeddings are already normalized)
|
||||
similarities = np.dot(document_embeddings, query_embedding.T).squeeze()
|
||||
|
||||
return similarities
|
||||
@@ -0,0 +1,305 @@
|
||||
# semantic_search/processor.py
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List, Dict, Any, Optional
|
||||
from dataclasses import dataclass
|
||||
import hashlib
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class DocumentChunk:
|
||||
"""Represents a chunk of a document."""
|
||||
chunk_id: str
|
||||
document_id: str
|
||||
text: str
|
||||
metadata: Dict[str, Any]
|
||||
chunk_index: int
|
||||
total_chunks: int
|
||||
|
||||
class DocumentProcessor:
|
||||
"""
|
||||
Processes legal documents for semantic search.
|
||||
Handles chunking, cleaning, and metadata extraction.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
chunk_size: int = 1000,
|
||||
chunk_overlap: int = 200,
|
||||
min_chunk_size: int = 100):
|
||||
"""
|
||||
Initialize document processor.
|
||||
|
||||
Args:
|
||||
chunk_size: Target size for each chunk in characters
|
||||
chunk_overlap: Number of overlapping characters between chunks
|
||||
min_chunk_size: Minimum chunk size to keep
|
||||
"""
|
||||
self.chunk_size = chunk_size
|
||||
self.chunk_overlap = chunk_overlap
|
||||
self.min_chunk_size = min_chunk_size
|
||||
|
||||
logger.info(f"Initialized DocumentProcessor (chunk_size={chunk_size}, overlap={chunk_overlap})")
|
||||
|
||||
def process_document(self,
|
||||
document_id: str,
|
||||
text: str,
|
||||
metadata: Optional[Dict[str, Any]] = None) -> List[DocumentChunk]:
|
||||
"""
|
||||
Process a single document into chunks.
|
||||
|
||||
Args:
|
||||
document_id: Unique document identifier
|
||||
text: Document text content
|
||||
metadata: Optional document metadata
|
||||
|
||||
Returns:
|
||||
List of document chunks
|
||||
"""
|
||||
if not text or len(text.strip()) < self.min_chunk_size:
|
||||
logger.warning(f"Document {document_id} too short to process")
|
||||
return []
|
||||
|
||||
# Clean text
|
||||
cleaned_text = self._clean_text(text)
|
||||
|
||||
# Extract metadata from text if not provided
|
||||
if metadata is None:
|
||||
metadata = {}
|
||||
|
||||
# Add extracted metadata
|
||||
extracted_metadata = self._extract_metadata(cleaned_text)
|
||||
metadata.update(extracted_metadata)
|
||||
|
||||
# Create chunks
|
||||
chunks = self._create_chunks(cleaned_text)
|
||||
|
||||
# Create DocumentChunk objects
|
||||
document_chunks = []
|
||||
for i, chunk_text in enumerate(chunks):
|
||||
chunk_id = self._generate_chunk_id(document_id, i)
|
||||
|
||||
chunk = DocumentChunk(
|
||||
chunk_id=chunk_id,
|
||||
document_id=document_id,
|
||||
text=chunk_text,
|
||||
metadata={
|
||||
**metadata,
|
||||
'chunk_index': i,
|
||||
'total_chunks': len(chunks)
|
||||
},
|
||||
chunk_index=i,
|
||||
total_chunks=len(chunks)
|
||||
)
|
||||
document_chunks.append(chunk)
|
||||
|
||||
logger.info(f"Processed document {document_id} into {len(chunks)} chunks")
|
||||
return document_chunks
|
||||
|
||||
def _clean_text(self, text: str) -> str:
|
||||
"""
|
||||
Clean and normalize text for processing.
|
||||
|
||||
Args:
|
||||
text: Raw text
|
||||
|
||||
Returns:
|
||||
Cleaned text
|
||||
"""
|
||||
# Remove excessive whitespace
|
||||
text = re.sub(r'\s+', ' ', text)
|
||||
|
||||
# Remove special characters but keep Turkish characters
|
||||
# Keep: letters, numbers, spaces, and common punctuation
|
||||
text = re.sub(r'[^\w\s\.\,\;\:\!\?\-\(\)\"\'ÇĞIİÖŞÜçğıiöşü]', ' ', text)
|
||||
|
||||
# Remove multiple spaces
|
||||
text = re.sub(r' +', ' ', text)
|
||||
|
||||
# Trim
|
||||
text = text.strip()
|
||||
|
||||
return text
|
||||
|
||||
def _extract_metadata(self, text: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Extract metadata from legal document text.
|
||||
|
||||
Args:
|
||||
text: Document text
|
||||
|
||||
Returns:
|
||||
Extracted metadata
|
||||
"""
|
||||
metadata = {}
|
||||
|
||||
# Extract case numbers (Esas/Karar)
|
||||
esas_pattern = r'E(?:sas)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
karar_pattern = r'K(?:arar)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
|
||||
esas_match = re.search(esas_pattern, text[:500]) # Look in first 500 chars
|
||||
if esas_match:
|
||||
metadata['esas_no'] = f"E.{esas_match.group(1)}/{esas_match.group(2)}"
|
||||
|
||||
karar_match = re.search(karar_pattern, text[:500])
|
||||
if karar_match:
|
||||
metadata['karar_no'] = f"K.{karar_match.group(1)}/{karar_match.group(2)}"
|
||||
|
||||
# Extract dates (DD.MM.YYYY or DD/MM/YYYY format)
|
||||
date_pattern = r'(\d{1,2})[\.\/](\d{1,2})[\.\/](\d{4})'
|
||||
dates = re.findall(date_pattern, text[:1000]) # Look in first 1000 chars
|
||||
if dates:
|
||||
# Take the first date as decision date
|
||||
day, month, year = dates[0]
|
||||
metadata['karar_tarihi'] = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
|
||||
# Extract court/chamber name
|
||||
chamber_patterns = [
|
||||
r'(\d+)\.\s*Hukuk\s+Dairesi',
|
||||
r'(\d+)\.\s*Ceza\s+Dairesi',
|
||||
r'Hukuk\s+Genel\s+Kurulu',
|
||||
r'Ceza\s+Genel\s+Kurulu',
|
||||
r'(\d+)\.\s*Daire'
|
||||
]
|
||||
|
||||
for pattern in chamber_patterns:
|
||||
match = re.search(pattern, text[:500], re.IGNORECASE)
|
||||
if match:
|
||||
metadata['chamber'] = match.group(0)
|
||||
break
|
||||
|
||||
return metadata
|
||||
|
||||
def _create_chunks(self, text: str) -> List[str]:
|
||||
"""
|
||||
Create overlapping chunks from text.
|
||||
|
||||
Args:
|
||||
text: Cleaned document text
|
||||
|
||||
Returns:
|
||||
List of text chunks
|
||||
"""
|
||||
chunks = []
|
||||
|
||||
# Split by sentences for better semantic coherence
|
||||
sentences = self._split_sentences(text)
|
||||
|
||||
current_chunk = []
|
||||
current_size = 0
|
||||
|
||||
for sentence in sentences:
|
||||
sentence_size = len(sentence)
|
||||
|
||||
# If adding this sentence exceeds chunk size
|
||||
if current_size + sentence_size > self.chunk_size and current_chunk:
|
||||
# Save current chunk
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
chunks.append(chunk_text)
|
||||
|
||||
# Create overlap for next chunk
|
||||
overlap_size = 0
|
||||
overlap_sentences = []
|
||||
|
||||
# Add sentences from the end until we reach overlap size
|
||||
for sent in reversed(current_chunk):
|
||||
overlap_size += len(sent)
|
||||
overlap_sentences.insert(0, sent)
|
||||
if overlap_size >= self.chunk_overlap:
|
||||
break
|
||||
|
||||
# Start new chunk with overlap
|
||||
current_chunk = overlap_sentences
|
||||
current_size = sum(len(s) for s in current_chunk)
|
||||
|
||||
# Add sentence to current chunk
|
||||
current_chunk.append(sentence)
|
||||
current_size += sentence_size
|
||||
|
||||
# Add final chunk if not empty
|
||||
if current_chunk:
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
if len(chunk_text) >= self.min_chunk_size:
|
||||
chunks.append(chunk_text)
|
||||
|
||||
return chunks
|
||||
|
||||
def _split_sentences(self, text: str) -> List[str]:
|
||||
"""
|
||||
Split text into sentences.
|
||||
|
||||
Args:
|
||||
text: Text to split
|
||||
|
||||
Returns:
|
||||
List of sentences
|
||||
"""
|
||||
# Simple sentence splitting for Turkish text
|
||||
# Split on period, question mark, exclamation, but not on abbreviations
|
||||
|
||||
# Common Turkish abbreviations to preserve
|
||||
abbreviations = ['Dr', 'Prof', 'Av', 'Md', 'Yrd', 'Doç', 'No', 'S', 'vs', 'vb', 'bkz']
|
||||
|
||||
# Replace abbreviations temporarily
|
||||
temp_text = text
|
||||
replacements = {}
|
||||
for i, abbr in enumerate(abbreviations):
|
||||
placeholder = f"__ABBR{i}__"
|
||||
temp_text = temp_text.replace(f"{abbr}.", placeholder)
|
||||
replacements[placeholder] = f"{abbr}."
|
||||
|
||||
# Split sentences
|
||||
sentence_endings = re.compile(r'[.!?]+')
|
||||
sentences = sentence_endings.split(temp_text)
|
||||
|
||||
# Restore abbreviations and clean
|
||||
cleaned_sentences = []
|
||||
for sentence in sentences:
|
||||
# Restore abbreviations
|
||||
for placeholder, original in replacements.items():
|
||||
sentence = sentence.replace(placeholder, original)
|
||||
|
||||
# Clean and add if not empty
|
||||
sentence = sentence.strip()
|
||||
if sentence and len(sentence) > 10: # Minimum sentence length
|
||||
cleaned_sentences.append(sentence)
|
||||
|
||||
return cleaned_sentences
|
||||
|
||||
def _generate_chunk_id(self, document_id: str, chunk_index: int) -> str:
|
||||
"""
|
||||
Generate unique chunk ID.
|
||||
|
||||
Args:
|
||||
document_id: Parent document ID
|
||||
chunk_index: Index of chunk in document
|
||||
|
||||
Returns:
|
||||
Unique chunk ID
|
||||
"""
|
||||
chunk_string = f"{document_id}_chunk_{chunk_index}"
|
||||
chunk_hash = hashlib.md5(chunk_string.encode()).hexdigest()[:8]
|
||||
return f"{document_id}_c{chunk_index}_{chunk_hash}"
|
||||
|
||||
def combine_chunks(self, chunks: List[DocumentChunk]) -> str:
|
||||
"""
|
||||
Combine chunks back into full document text.
|
||||
|
||||
Args:
|
||||
chunks: List of document chunks
|
||||
|
||||
Returns:
|
||||
Combined text
|
||||
"""
|
||||
if not chunks:
|
||||
return ""
|
||||
|
||||
# Sort by chunk index
|
||||
sorted_chunks = sorted(chunks, key=lambda x: x.chunk_index)
|
||||
|
||||
# For overlapping chunks, we need to be careful about duplication
|
||||
# Simple approach: just concatenate with space
|
||||
combined = " ".join([chunk.text for chunk in sorted_chunks])
|
||||
|
||||
return combined
|
||||
@@ -0,0 +1,235 @@
|
||||
# semantic_search/vector_store.py
|
||||
|
||||
import logging
|
||||
import numpy as np
|
||||
from typing import List, Dict, Any, Tuple, Optional
|
||||
from dataclasses import dataclass
|
||||
import json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class Document:
|
||||
"""Represents a document with its embedding and metadata."""
|
||||
id: str
|
||||
text: str
|
||||
embedding: np.ndarray
|
||||
metadata: Dict[str, Any]
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Convert to dictionary (excluding embedding for serialization)."""
|
||||
return {
|
||||
'id': self.id,
|
||||
'text': self.text,
|
||||
'metadata': self.metadata
|
||||
}
|
||||
|
||||
class VectorStore:
|
||||
"""
|
||||
In-memory vector storage with similarity search capabilities.
|
||||
Future versions can use Faiss, ChromaDB, or other vector databases.
|
||||
"""
|
||||
|
||||
def __init__(self, dimension: int = 768):
|
||||
"""
|
||||
Initialize vector store.
|
||||
|
||||
Args:
|
||||
dimension: Embedding dimension size
|
||||
"""
|
||||
self.dimension = dimension
|
||||
self.documents: List[Document] = []
|
||||
self.embeddings: Optional[np.ndarray] = None
|
||||
self.index_built = False
|
||||
|
||||
logger.info(f"Initialized VectorStore with dimension: {dimension}")
|
||||
|
||||
def add_documents(self,
|
||||
ids: List[str],
|
||||
texts: List[str],
|
||||
embeddings: np.ndarray,
|
||||
metadata: Optional[List[Dict[str, Any]]] = None) -> int:
|
||||
"""
|
||||
Add documents to the vector store.
|
||||
|
||||
Args:
|
||||
ids: Document IDs
|
||||
texts: Document texts
|
||||
embeddings: Document embeddings (N x dimension)
|
||||
metadata: Optional metadata for each document
|
||||
|
||||
Returns:
|
||||
Number of documents added
|
||||
"""
|
||||
if len(ids) != len(texts) or len(ids) != embeddings.shape[0]:
|
||||
raise ValueError("Mismatched lengths for ids, texts, and embeddings")
|
||||
|
||||
if metadata and len(metadata) != len(ids):
|
||||
raise ValueError("Metadata length doesn't match document count")
|
||||
|
||||
# Add documents
|
||||
for i in range(len(ids)):
|
||||
doc = Document(
|
||||
id=ids[i],
|
||||
text=texts[i],
|
||||
embedding=embeddings[i],
|
||||
metadata=metadata[i] if metadata else {}
|
||||
)
|
||||
self.documents.append(doc)
|
||||
|
||||
# Rebuild index
|
||||
self._build_index()
|
||||
|
||||
logger.info(f"Added {len(ids)} documents to vector store. Total: {len(self.documents)}")
|
||||
return len(ids)
|
||||
|
||||
def _build_index(self):
|
||||
"""Build or rebuild the embedding index."""
|
||||
if not self.documents:
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
return
|
||||
|
||||
# Stack all embeddings into a single array
|
||||
self.embeddings = np.vstack([doc.embedding for doc in self.documents])
|
||||
self.index_built = True
|
||||
|
||||
logger.debug(f"Built index with shape: {self.embeddings.shape}")
|
||||
|
||||
def search(self,
|
||||
query_embedding: np.ndarray,
|
||||
top_k: int = 10,
|
||||
threshold: Optional[float] = None) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Search for similar documents using cosine similarity.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
top_k: Number of results to return
|
||||
threshold: Optional similarity threshold (0-1)
|
||||
|
||||
Returns:
|
||||
List of (Document, similarity_score) tuples
|
||||
"""
|
||||
if not self.index_built or self.embeddings is None:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Ensure query is 2D
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarities (assuming normalized embeddings)
|
||||
similarities = np.dot(self.embeddings, query_embedding.T).squeeze()
|
||||
|
||||
# Apply threshold if specified
|
||||
if threshold is not None:
|
||||
valid_indices = np.where(similarities >= threshold)[0]
|
||||
if len(valid_indices) == 0:
|
||||
logger.info(f"No documents above threshold {threshold}")
|
||||
return []
|
||||
similarities = similarities[valid_indices]
|
||||
valid_docs = [self.documents[i] for i in valid_indices]
|
||||
else:
|
||||
valid_docs = self.documents
|
||||
|
||||
# Get top-k indices
|
||||
top_k = min(top_k, len(valid_docs))
|
||||
if top_k == 0:
|
||||
return []
|
||||
|
||||
# Use argpartition for efficiency with large arrays
|
||||
if len(similarities) > top_k:
|
||||
top_indices = np.argpartition(similarities, -top_k)[-top_k:]
|
||||
top_indices = top_indices[np.argsort(similarities[top_indices])[::-1]]
|
||||
else:
|
||||
top_indices = np.argsort(similarities)[::-1]
|
||||
|
||||
# Create results
|
||||
results = []
|
||||
for idx in top_indices:
|
||||
doc = valid_docs[idx] if threshold else self.documents[idx]
|
||||
score = float(similarities[idx])
|
||||
results.append((doc, score))
|
||||
|
||||
logger.info(f"Search returned {len(results)} results (top_k={top_k})")
|
||||
return results
|
||||
|
||||
def hybrid_search(self,
|
||||
query_embedding: np.ndarray,
|
||||
keyword_scores: Dict[str, float],
|
||||
top_k: int = 10,
|
||||
alpha: float = 0.5) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Hybrid search combining vector similarity and keyword scores.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
keyword_scores: Document ID to keyword relevance score mapping
|
||||
top_k: Number of results to return
|
||||
alpha: Weight for vector similarity (1-alpha for keyword score)
|
||||
|
||||
Returns:
|
||||
List of (Document, combined_score) tuples
|
||||
"""
|
||||
if not self.index_built:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Get vector similarities
|
||||
vector_results = self.search(query_embedding, top_k=len(self.documents))
|
||||
|
||||
# Combine scores
|
||||
combined_scores = []
|
||||
for doc, vector_score in vector_results:
|
||||
keyword_score = keyword_scores.get(doc.id, 0.0)
|
||||
# Normalize keyword score to 0-1 range if needed
|
||||
if keyword_score > 1.0:
|
||||
keyword_score = keyword_score / max(keyword_scores.values())
|
||||
|
||||
combined_score = alpha * vector_score + (1 - alpha) * keyword_score
|
||||
combined_scores.append((doc, combined_score))
|
||||
|
||||
# Sort by combined score and return top-k
|
||||
combined_scores.sort(key=lambda x: x[1], reverse=True)
|
||||
results = combined_scores[:top_k]
|
||||
|
||||
logger.info(f"Hybrid search returned {len(results)} results")
|
||||
return results
|
||||
|
||||
def clear(self):
|
||||
"""Clear all documents from the store."""
|
||||
self.documents = []
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
logger.info("Cleared vector store")
|
||||
|
||||
def size(self) -> int:
|
||||
"""Get number of documents in store."""
|
||||
return len(self.documents)
|
||||
|
||||
def get_by_id(self, doc_id: str) -> Optional[Document]:
|
||||
"""Get document by ID."""
|
||||
for doc in self.documents:
|
||||
if doc.id == doc_id:
|
||||
return doc
|
||||
return None
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""Get statistics about the vector store."""
|
||||
stats = {
|
||||
'num_documents': len(self.documents),
|
||||
'dimension': self.dimension,
|
||||
'index_built': self.index_built,
|
||||
'memory_usage_mb': 0
|
||||
}
|
||||
|
||||
if self.embeddings is not None:
|
||||
# Estimate memory usage
|
||||
memory_bytes = self.embeddings.nbytes
|
||||
for doc in self.documents:
|
||||
memory_bytes += len(doc.text.encode('utf-8'))
|
||||
memory_bytes += len(json.dumps(doc.metadata).encode('utf-8'))
|
||||
stats['memory_usage_mb'] = memory_bytes / (1024 * 1024)
|
||||
|
||||
return stats
|
||||
+1
-1
@@ -13,7 +13,7 @@ import os
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
|
||||
from starlette.responses import JSONResponse
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.middleware.authentication import AuthenticationMiddleware
|
||||
|
||||
+2
-1
@@ -1,4 +1,5 @@
|
||||
import os, stripe
|
||||
import os
|
||||
import stripe
|
||||
from clerk_backend_api import Clerk # Clerk backend SDK
|
||||
from fastapi import APIRouter, Request, HTTPException
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Union, Tuple
|
||||
from typing import List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
|
||||
@@ -1,20 +1,16 @@
|
||||
# yargitay_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
|
||||
from typing import Dict, Any, List, Optional
|
||||
from typing import Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
YargitayDetailedSearchRequest,
|
||||
YargitayApiSearchResponse,
|
||||
YargitayApiDecisionEntry,
|
||||
YargitayDocumentMarkdown,
|
||||
CompactYargitaySearchResult
|
||||
YargitayDocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# yargitay_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl, ConfigDict
|
||||
from typing import List, Optional, Dict, Any, Literal
|
||||
from typing import List, Optional, Literal
|
||||
|
||||
# Yargıtay Chamber/Board Options
|
||||
YargitayBirimEnum = Literal[
|
||||
|
||||
Reference in New Issue
Block a user