Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f1d3b60efb | ||
|
|
93e64bc1fc | ||
|
|
77e2748ade | ||
|
|
da146cf3ec | ||
|
|
e771c5b3c5 | ||
|
|
1223b37adb | ||
|
|
e26f09aced | ||
|
|
ae5bae2f4a | ||
|
|
a2b50951e9 | ||
|
|
91ad04cf09 | ||
|
|
5cec0df785 | ||
|
|
d51f11c7ba | ||
|
|
b207b16ef7 | ||
|
|
f47147ba44 | ||
|
|
4d57a3939f | ||
|
|
50c6963eee | ||
|
|
def7e7d65e | ||
|
|
82a0d13d25 | ||
|
|
18b552ca2f | ||
|
|
1e96b1888e | ||
|
|
815786a09d | ||
|
|
260adb3ac9 | ||
|
|
7164205425 | ||
|
|
3961a23d3a | ||
|
|
25723f070f | ||
|
|
6376037ccf | ||
|
|
d1728ce114 | ||
|
|
69b5da5cef | ||
|
|
91564bf0a1 | ||
|
|
6f94eca33c | ||
|
|
4122790821 | ||
|
|
0f5bae8bb1 | ||
|
|
4d7da0d3ba | ||
|
|
4f48681b09 | ||
|
|
b401bad890 | ||
|
|
0a80bc535b | ||
|
|
b1da034ea9 | ||
|
|
e900bc03dd | ||
|
|
54f81e18f0 | ||
|
|
4e18e792c5 | ||
|
|
4a3edef287 | ||
|
|
e2ca844ab9 | ||
|
|
a49d0859ea | ||
|
|
a7877f34f4 | ||
|
|
364f3761d7 | ||
|
|
673f996f5f | ||
|
|
90a7a23064 | ||
|
|
443657f9e2 | ||
|
|
f5fa0076f8 | ||
|
|
7a346ef3f6 | ||
|
|
217103f0b6 | ||
|
|
1fbcb65031 | ||
|
|
9e40671798 | ||
|
|
6c8a614872 | ||
|
|
861d9e86ef | ||
|
|
c4b5d3608a | ||
|
|
38e0cc032b | ||
|
|
2c1b8c6f9d | ||
|
|
92f04fbab6 | ||
|
|
515347e29c | ||
|
|
ebefe22a4c | ||
|
|
c93244ee10 | ||
|
|
d84f8a2c88 |
@@ -70,6 +70,15 @@ JWT_SECRET_KEY=your_jwt_secret_key_here
|
||||
# MAX_REQUESTS_PER_MINUTE=60
|
||||
# BURST_CAPACITY=20
|
||||
|
||||
# =============================================================================
|
||||
# SEMANTIC SEARCH SETTINGS (Optional)
|
||||
# =============================================================================
|
||||
|
||||
# OpenRouter API Key for semantic search functionality
|
||||
# Get your API key from: https://openrouter.ai/keys
|
||||
# If not set, semantic search tool will be disabled
|
||||
OPENROUTER_API_KEY=sk-or-v1-your_openrouter_api_key_here
|
||||
|
||||
# =============================================================================
|
||||
# USAGE INSTRUCTIONS
|
||||
# =============================================================================
|
||||
|
||||
@@ -213,3 +213,4 @@ measure_mcp_directly.py
|
||||
playwright_mcp_overhead.json
|
||||
simple_test.py
|
||||
analyze_anayasa_html.py
|
||||
CLAUDE.md
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
/cache
|
||||
@@ -0,0 +1,84 @@
|
||||
# list of languages for which language servers are started; choose from:
|
||||
# al bash clojure cpp csharp csharp_omnisharp
|
||||
# dart elixir elm erlang fortran go
|
||||
# haskell java julia kotlin lua markdown
|
||||
# nix perl php python python_jedi r
|
||||
# rego ruby ruby_solargraph rust scala swift
|
||||
# terraform typescript typescript_vts yaml zig
|
||||
# Note:
|
||||
# - For C, use cpp
|
||||
# - For JavaScript, use typescript
|
||||
# Special requirements:
|
||||
# - csharp: Requires the presence of a .sln file in the project folder.
|
||||
# When using multiple languages, the first language server that supports a given file will be used for that file.
|
||||
# The first language is the default language and the respective language server will be used as a fallback.
|
||||
# Note that when using the JetBrains backend, language servers are not used and this list is correspondingly ignored.
|
||||
languages:
|
||||
- python
|
||||
|
||||
# the encoding used by text files in the project
|
||||
# For a list of possible encodings, see https://docs.python.org/3.11/library/codecs.html#standard-encodings
|
||||
encoding: "utf-8"
|
||||
|
||||
# whether to use the project's gitignore file to ignore files
|
||||
# Added on 2025-04-07
|
||||
ignore_all_files_in_gitignore: true
|
||||
|
||||
# list of additional paths to ignore
|
||||
# same syntax as gitignore, so you can use * and **
|
||||
# Was previously called `ignored_dirs`, please update your config if you are using that.
|
||||
# Added (renamed) on 2025-04-07
|
||||
ignored_paths: []
|
||||
|
||||
# whether the project is in read-only mode
|
||||
# If set to true, all editing tools will be disabled and attempts to use them will result in an error
|
||||
# Added on 2025-04-18
|
||||
read_only: false
|
||||
|
||||
# list of tool names to exclude. We recommend not excluding any tools, see the readme for more details.
|
||||
# Below is the complete list of tools for convenience.
|
||||
# To make sure you have the latest list of tools, and to view their descriptions,
|
||||
# execute `uv run scripts/print_tool_overview.py`.
|
||||
#
|
||||
# * `activate_project`: Activates a project by name.
|
||||
# * `check_onboarding_performed`: Checks whether project onboarding was already performed.
|
||||
# * `create_text_file`: Creates/overwrites a file in the project directory.
|
||||
# * `delete_lines`: Deletes a range of lines within a file.
|
||||
# * `delete_memory`: Deletes a memory from Serena's project-specific memory store.
|
||||
# * `execute_shell_command`: Executes a shell command.
|
||||
# * `find_referencing_code_snippets`: Finds code snippets in which the symbol at the given location is referenced.
|
||||
# * `find_referencing_symbols`: Finds symbols that reference the symbol at the given location (optionally filtered by type).
|
||||
# * `find_symbol`: Performs a global (or local) search for symbols with/containing a given name/substring (optionally filtered by type).
|
||||
# * `get_current_config`: Prints the current configuration of the agent, including the active and available projects, tools, contexts, and modes.
|
||||
# * `get_symbols_overview`: Gets an overview of the top-level symbols defined in a given file.
|
||||
# * `initial_instructions`: Gets the initial instructions for the current project.
|
||||
# Should only be used in settings where the system prompt cannot be set,
|
||||
# e.g. in clients you have no control over, like Claude Desktop.
|
||||
# * `insert_after_symbol`: Inserts content after the end of the definition of a given symbol.
|
||||
# * `insert_at_line`: Inserts content at a given line in a file.
|
||||
# * `insert_before_symbol`: Inserts content before the beginning of the definition of a given symbol.
|
||||
# * `list_dir`: Lists files and directories in the given directory (optionally with recursion).
|
||||
# * `list_memories`: Lists memories in Serena's project-specific memory store.
|
||||
# * `onboarding`: Performs onboarding (identifying the project structure and essential tasks, e.g. for testing or building).
|
||||
# * `prepare_for_new_conversation`: Provides instructions for preparing for a new conversation (in order to continue with the necessary context).
|
||||
# * `read_file`: Reads a file within the project directory.
|
||||
# * `read_memory`: Reads the memory with the given name from Serena's project-specific memory store.
|
||||
# * `remove_project`: Removes a project from the Serena configuration.
|
||||
# * `replace_lines`: Replaces a range of lines within a file with new content.
|
||||
# * `replace_symbol_body`: Replaces the full definition of a symbol.
|
||||
# * `restart_language_server`: Restarts the language server, may be necessary when edits not through Serena happen.
|
||||
# * `search_for_pattern`: Performs a search for a pattern in the project.
|
||||
# * `summarize_changes`: Provides instructions for summarizing the changes made to the codebase.
|
||||
# * `switch_modes`: Activates modes by providing a list of their names
|
||||
# * `think_about_collected_information`: Thinking tool for pondering the completeness of collected information.
|
||||
# * `think_about_task_adherence`: Thinking tool for determining whether the agent is still on track with the current task.
|
||||
# * `think_about_whether_you_are_done`: Thinking tool for determining whether the task is truly completed.
|
||||
# * `write_memory`: Writes a named memory (for future reference) to Serena's project-specific memory store.
|
||||
excluded_tools: []
|
||||
|
||||
# initial prompt for the project. It will always be given to the LLM upon activating the project
|
||||
# (contrary to the memories, which are loaded on demand).
|
||||
initial_prompt: ""
|
||||
|
||||
project_name: "yargi-mcp"
|
||||
included_optional_tools: []
|
||||
+7
-2
@@ -1,5 +1,5 @@
|
||||
# -------- BASE IMAGE (includes Chromium & deps) ----------------------------
|
||||
FROM mcr.microsoft.com/playwright/python:v1.53.0-noble
|
||||
# -------- BASE IMAGE ---------------------------------------------------------
|
||||
FROM python:3.12-slim
|
||||
|
||||
# -------- Runtime setup ----------------------------------------------------
|
||||
WORKDIR /app
|
||||
@@ -9,8 +9,13 @@ COPY pyproject.toml poetry.lock* requirements*.txt* ./
|
||||
|
||||
# Fast, deterministic install with `uv`
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system --no-cache-dir . && \
|
||||
uv pip install --system --no-cache-dir .[asgi,saas]
|
||||
|
||||
# Cache buster - force rebuild
|
||||
ARG CACHE_BUST=202510061202
|
||||
RUN echo "Cache bust: $CACHE_BUST"
|
||||
|
||||
# Copy application source
|
||||
COPY . .
|
||||
|
||||
|
||||
@@ -4,6 +4,28 @@
|
||||
|
||||
Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kararlar, Uyuşmazlık Mahkemesi, Anayasa Mahkemesi - Norm Denetimi ile Bireysel Başvuru Kararları, Kamu İhale Kurulu Kararları, Rekabet Kurumu Kararları, Sayıştay Kararları, KVKK Kararları ve BDDK Kararları) erişimi kolaylaştıran bir [FastMCP](https://gofastmcp.com/) sunucusu oluşturur. Bu sayede, bu kaynaklardan veri arama ve belge getirme işlemleri, Model Context Protocol (MCP) destekleyen LLM (Büyük Dil Modeli) uygulamaları (örneğin Claude Desktop veya [5ire](https://5ire.app)) ve diğer istemciler tarafından araç (tool) olarak kullanılabilir hale gelir.
|
||||
|
||||
---
|
||||
|
||||
## 🚀 5 Dakikada Başla (Remote MCP)
|
||||
|
||||
### ✅ Kurulum Gerektirmez! Hemen Kullan!
|
||||
|
||||
🔗 **Remote MCP Adresi:** `https://yargimcp.fastmcp.app/mcp`
|
||||
|
||||
### Claude Desktop ile Kullanım
|
||||
|
||||
1. **Claude Desktop'ı açın**
|
||||
2. **Settings → Connectors → Add Custom Connector**
|
||||
3. **Bilgileri girin:**
|
||||
- **Name:** `Yargı MCP`
|
||||
- **URL:** `https://yargimcp.fastmcp.app/mcp`
|
||||
4. **Add** butonuna tıklayın
|
||||
5. **Hemen kullanmaya başlayın!** 🎉
|
||||
|
||||
> 💡 **İpucu:** Remote MCP sayesinde Python, uv veya herhangi bir kurulum yapmadan doğrudan Claude Desktop üzerinden Türk hukuk kaynaklarına erişebilirsiniz!
|
||||
|
||||
---
|
||||
|
||||

|
||||
|
||||
🎯 **Temel Özellikler**
|
||||
@@ -132,10 +154,65 @@ Yargı MCP'yi Gemini CLI ile kullanmak için:
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
<details>
|
||||
<summary>🧠 <strong>Semantik Arama (Opsiyonel - OpenRouter API)</strong></summary>
|
||||
|
||||
Yargı MCP, **semantik arama** özelliği ile kararları anlamsal olarak sıralayabilir. Bu özellik opsiyoneldir ve `OPENROUTER_API_KEY` ayarlandığında otomatik olarak etkinleşir.
|
||||
|
||||
### Semantik Arama Nasıl Çalışır?
|
||||
1. `initial_keyword` ile Bedesten API'den 100 karar çekilir
|
||||
2. `query` ile bu kararlar embedding modeli kullanılarak anlamsal olarak sıralanır
|
||||
3. En alakalı kararlar döndürülür
|
||||
|
||||
### OpenRouter API Anahtarı Alma
|
||||
1. [OpenRouter](https://openrouter.ai/) sitesine gidin
|
||||
2. Hesap oluşturun ve API anahtarı alın (ücretsiz kredi ile başlayabilirsiniz)
|
||||
|
||||
### Claude Desktop için Yapılandırma
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"Yargı MCP": {
|
||||
"command": "uvx",
|
||||
"args": ["yargi-mcp"],
|
||||
"env": {
|
||||
"OPENROUTER_API_KEY": "sk-or-v1-xxx..."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### 5ire için Yapılandırma
|
||||
Tool ayarlarında **Environment Variables** alanına ekleyin:
|
||||
```
|
||||
OPENROUTER_API_KEY=sk-or-v1-xxx...
|
||||
```
|
||||
|
||||
### Gemini CLI için Yapılandırma
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"yargi_mcp": {
|
||||
"command": "uvx",
|
||||
"args": ["yargi-mcp"],
|
||||
"env": {
|
||||
"OPENROUTER_API_KEY": "sk-or-v1-xxx..."
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
> 💡 **Not:** `OPENROUTER_API_KEY` ayarlanmazsa semantik arama aracı görünmez, diğer 19 araç normal şekilde çalışmaya devam eder.
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary>🛠️ <strong>Kullanılabilir Araçlar (MCP Tools)</strong></summary>
|
||||
|
||||
Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliği için optimize edilmiş):
|
||||
Bu FastMCP sunucusu **19 temel MCP aracı** + **1 opsiyonel semantik arama aracı** sunar (token verimliliği için optimize edilmiş):
|
||||
|
||||
### **Yargıtay Araçları (Birleşik Bedesten API - Token Optimized)**
|
||||
*Not: Yargıtay araçları token verimliliği için birleşik Bedesten API'ye entegre edilmiştir*
|
||||
@@ -200,7 +277,7 @@ Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliğ
|
||||
|
||||
**GENEL İSTATİSTİKLER:**
|
||||
- **Toplam Mahkeme/Kurum:** 13 farklı hukuki kurum (KVKK dahil)
|
||||
- **Toplam MCP Tool:** 19 optimize edilmiş arama ve belge getirme aracı
|
||||
- **Toplam MCP Tool:** 19 temel araç + 1 opsiyonel semantik arama aracı
|
||||
- **Daire/Kurul Filtreleme:** 87 farklı seçenek (52 Yargıtay + 27 Danıştay + 8 Sayıştay)
|
||||
- **Tarih Filtreleme:** Birleşik Bedesten API aracında ISO 8601 formatında tam tarih aralığı desteği
|
||||
- **Kesin Cümle Arama:** Birleşik Bedesten API aracında çift tırnak ile tam cümle arama (`"\"mülkiyet kararı\""` formatı)
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
"""
|
||||
Analyze KİK v2 hash generation by examining JavaScript code patterns
|
||||
and trying to reverse engineer the hash generation logic.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import hashlib
|
||||
import hmac
|
||||
import base64
|
||||
from fastmcp import Client
|
||||
from mcp_server_main import app
|
||||
|
||||
def analyze_webpack_hash_patterns():
|
||||
"""
|
||||
Analyze the webpack JavaScript code you provided to find hash generation patterns
|
||||
"""
|
||||
print("🔍 Analyzing webpack hash generation patterns...")
|
||||
|
||||
# From the JavaScript code, I can see several hash/ID generation patterns:
|
||||
hash_patterns = {
|
||||
# Webpack chunk system hashes (from the JS code)
|
||||
"webpack_chunks": {
|
||||
315: "d9a9486a4f5ba326",
|
||||
531: "cd8fb385c88033ae",
|
||||
671: "04c48b287646627a",
|
||||
856: "682c9a7b87351f90",
|
||||
1017: "9de022378fc275f6",
|
||||
# ... many more from the __webpack_require__.u function
|
||||
},
|
||||
|
||||
# Symbol generation from Zone.js
|
||||
"zone_symbols": [
|
||||
"__zone_symbol__",
|
||||
"__Zone_symbol_prefix",
|
||||
"Zone.__symbol__"
|
||||
],
|
||||
|
||||
# Angular module federation patterns
|
||||
"module_federation": [
|
||||
"__webpack_modules__",
|
||||
"__webpack_module_cache__",
|
||||
"__webpack_require__"
|
||||
]
|
||||
}
|
||||
|
||||
# The target hash format
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"🎯 Target hash: {target_hash}")
|
||||
print(f" Length: {len(target_hash)} characters")
|
||||
print(f" Format: {'SHA256' if len(target_hash) == 64 else 'Other'} (64 chars = SHA256)")
|
||||
|
||||
return hash_patterns
|
||||
|
||||
def test_webpack_style_hashing(data_dict):
|
||||
"""Test webpack-style hash generation methods"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try various webpack-style hash methods
|
||||
hashes[f"webpack_md5_{key}"] = hashlib.md5(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha1_{key}"] = hashlib.sha1(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha256_{key}"] = hashlib.sha256(test_string.encode()).hexdigest()
|
||||
|
||||
# Try with various prefixes/suffixes (common in webpack)
|
||||
prefixed = f"__webpack__{test_string}"
|
||||
hashes[f"webpack_prefixed_sha256_{key}"] = hashlib.sha256(prefixed.encode()).hexdigest()
|
||||
|
||||
# Try with module federation style
|
||||
module_style = f"shell:{test_string}"
|
||||
hashes[f"module_fed_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
# Try JSON stringified
|
||||
json_style = json.dumps({"id": value, "type": "decision"}, separators=(',', ':'))
|
||||
hashes[f"json_sha256_{key}"] = hashlib.sha256(json_style.encode()).hexdigest()
|
||||
|
||||
# Try with timestamp or sequence
|
||||
with_seq = f"{test_string}_0"
|
||||
hashes[f"seq_sha256_{key}"] = hashlib.sha256(with_seq.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_angular_routing_hashes(data_dict):
|
||||
"""Test Angular routing/state management hash generation"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
# Angular often uses route parameters for hash generation
|
||||
route_style = f"/kurul-kararlari/{value}"
|
||||
hashes[f"route_sha256_{key}"] = hashlib.sha256(route_style.encode()).hexdigest()
|
||||
|
||||
# Component state style
|
||||
state_style = f"KurulKararGoster_{value}"
|
||||
hashes[f"state_sha256_{key}"] = hashlib.sha256(state_style.encode()).hexdigest()
|
||||
|
||||
# Angular module style
|
||||
module_style = f"kik.kurul.karar.{value}"
|
||||
hashes[f"module_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_base64_encoding_variants(data_dict):
|
||||
"""Test various base64 and encoding variants"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try base64 encoding then hashing
|
||||
b64_encoded = base64.b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64_sha256_{key}"] = hashlib.sha256(b64_encoded.encode()).hexdigest()
|
||||
|
||||
# Try URL-safe base64
|
||||
b64_url = base64.urlsafe_b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64url_sha256_{key}"] = hashlib.sha256(b64_url.encode()).hexdigest()
|
||||
|
||||
# Try hex encoding
|
||||
hex_encoded = test_string.encode().hex()
|
||||
hashes[f"hex_sha256_{key}"] = hashlib.sha256(hex_encoded.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
async def test_hash_generation_comprehensive():
|
||||
print("🔐 Comprehensive KİK document hash generation analysis...")
|
||||
print("=" * 70)
|
||||
|
||||
# First analyze the webpack patterns
|
||||
webpack_patterns = analyze_webpack_hash_patterns()
|
||||
|
||||
client = Client(app)
|
||||
|
||||
async with client:
|
||||
print("✅ MCP client connected")
|
||||
|
||||
# Get sample decisions
|
||||
print("\n📊 Getting sample decisions for hash analysis...")
|
||||
search_result = await client.call_tool("search_kik_v2_decisions", {
|
||||
"decision_type": "uyusmazlik",
|
||||
"karar_metni": "2024"
|
||||
})
|
||||
|
||||
if hasattr(search_result, 'content') and search_result.content:
|
||||
search_data = json.loads(search_result.content[0].text)
|
||||
decisions = search_data.get('decisions', [])
|
||||
|
||||
if decisions:
|
||||
print(f"✅ Found {len(decisions)} decisions")
|
||||
|
||||
# Test with first decision
|
||||
sample_decision = decisions[0]
|
||||
print(f"\n📋 Sample decision for hash analysis:")
|
||||
for key, value in sample_decision.items():
|
||||
print(f" {key}: {value}")
|
||||
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"\n🎯 Target hash to match: {target_hash}")
|
||||
|
||||
all_hashes = {}
|
||||
|
||||
# Test different hash generation methods
|
||||
print(f"\n🔨 Testing webpack-style hashing...")
|
||||
webpack_hashes = test_webpack_style_hashing(sample_decision)
|
||||
all_hashes.update(webpack_hashes)
|
||||
|
||||
print(f"🔨 Testing Angular routing hashes...")
|
||||
angular_hashes = test_angular_routing_hashes(sample_decision)
|
||||
all_hashes.update(angular_hashes)
|
||||
|
||||
print(f"🔨 Testing base64 encoding variants...")
|
||||
b64_hashes = test_base64_encoding_variants(sample_decision)
|
||||
all_hashes.update(b64_hashes)
|
||||
|
||||
# Check for matches
|
||||
print(f"\n🎯 Checking for hash matches...")
|
||||
matches_found = []
|
||||
partial_matches = []
|
||||
|
||||
for hash_name, hash_value in all_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
matches_found.append((hash_name, hash_value))
|
||||
print(f" 🎉 EXACT MATCH FOUND: {hash_name}")
|
||||
elif hash_value[:8] == target_hash[:8]: # First 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (first 8): {hash_name} -> {hash_value[:16]}...")
|
||||
elif hash_value[-8:] == target_hash[-8:]: # Last 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (last 8): {hash_name} -> ...{hash_value[-16:]}")
|
||||
|
||||
if not matches_found and not partial_matches:
|
||||
print(f" ❌ No matches found")
|
||||
print(f"\n📝 Sample generated hashes (first 10):")
|
||||
for i, (hash_name, hash_value) in enumerate(list(all_hashes.items())[:10]):
|
||||
print(f" {hash_name}: {hash_value}")
|
||||
|
||||
# Try combinations with other decisions
|
||||
print(f"\n🔄 Testing hash combinations with multiple decisions...")
|
||||
if len(decisions) > 1:
|
||||
for i, decision in enumerate(decisions[1:3]): # Test 2 more
|
||||
print(f"\n Testing decision {i+2}: {decision.get('kararNo')}")
|
||||
decision_hashes = test_webpack_style_hashing(decision)
|
||||
|
||||
for hash_name, hash_value in decision_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
print(f" 🎉 MATCH FOUND in decision {i+2}: {hash_name}")
|
||||
matches_found.append((f"decision_{i+2}_{hash_name}", hash_value))
|
||||
|
||||
# Try composite hashes (combining multiple fields)
|
||||
print(f"\n🔗 Testing composite hash generation...")
|
||||
composite_tests = [
|
||||
f"{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararNo')}",
|
||||
f"{sample_decision.get('kararNo')}_{sample_decision.get('kararTarihi')}",
|
||||
f"uyusmazlik_{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararTarihi')}",
|
||||
json.dumps(sample_decision, separators=(',', ':'), sort_keys=True),
|
||||
f"{sample_decision.get('basvuran')}_{sample_decision.get('gundemMaddesiId')}",
|
||||
]
|
||||
|
||||
for i, composite_str in enumerate(composite_tests):
|
||||
composite_hash = hashlib.sha256(composite_str.encode()).hexdigest()
|
||||
if composite_hash == target_hash:
|
||||
print(f" 🎉 COMPOSITE MATCH FOUND: test_{i} -> {composite_str[:50]}...")
|
||||
matches_found.append((f"composite_{i}", composite_hash))
|
||||
|
||||
print(f"\n🎯 Hash analysis completed!")
|
||||
print(f" Total matches found: {len(matches_found)}")
|
||||
print(f" Partial matches: {len(partial_matches)}")
|
||||
|
||||
else:
|
||||
print("❌ No decisions found")
|
||||
else:
|
||||
print("❌ Search failed")
|
||||
|
||||
print("=" * 70)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_hash_generation_comprehensive())
|
||||
Regular → Executable
+227
-316
@@ -3,7 +3,7 @@ ASGI application for Yargı MCP Server
|
||||
|
||||
This module provides ASGI/HTTP access to the Yargı MCP server,
|
||||
allowing it to be deployed as a web service with FastAPI wrapper
|
||||
for Stripe webhook integration.
|
||||
for OAuth integration and proper middleware support.
|
||||
|
||||
Usage:
|
||||
uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
@@ -12,52 +12,92 @@ Usage:
|
||||
import os
|
||||
import time
|
||||
import logging
|
||||
import json
|
||||
from datetime import datetime, timedelta
|
||||
from fastapi import FastAPI, Request, HTTPException, Query
|
||||
from fastapi.responses import JSONResponse, HTMLResponse
|
||||
from fastapi.responses import JSONResponse, HTMLResponse, Response
|
||||
from fastapi.exception_handlers import http_exception_handler
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.responses import Response
|
||||
from starlette.middleware.base import BaseHTTPMiddleware
|
||||
|
||||
# Import the fully configured MCP app with all tools
|
||||
from mcp_server_main import app as mcp_server
|
||||
# Import the proper create_app function that includes all middleware
|
||||
from mcp_server_main import create_app
|
||||
|
||||
# Import Stripe webhook router
|
||||
from stripe_webhook import router as stripe_router
|
||||
# Conditional auth-related imports (only if auth enabled)
|
||||
_auth_check = os.getenv("ENABLE_AUTH", "false").lower() == "true"
|
||||
|
||||
# Import simplified MCP Auth HTTP adapter
|
||||
from mcp_auth_http_simple import router as mcp_auth_router
|
||||
if _auth_check:
|
||||
# Import MCP Auth HTTP adapter (OAuth endpoints)
|
||||
try:
|
||||
from mcp_auth_http_simple import router as mcp_auth_router
|
||||
except ImportError:
|
||||
mcp_auth_router = None
|
||||
|
||||
# Import Stripe webhook router
|
||||
try:
|
||||
from stripe_webhook import router as stripe_router
|
||||
except ImportError:
|
||||
stripe_router = None
|
||||
else:
|
||||
mcp_auth_router = None
|
||||
stripe_router = None
|
||||
|
||||
# OAuth configuration from environment variables
|
||||
CLERK_ISSUER = os.getenv("CLERK_ISSUER", "https://accounts.yargimcp.com")
|
||||
BASE_URL = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
CLERK_ISSUER = os.getenv("CLERK_ISSUER", "https://clerk.yargimcp.com")
|
||||
BASE_URL = os.getenv("BASE_URL", "https://api.yargimcp.com")
|
||||
CLERK_SECRET_KEY = os.getenv("CLERK_SECRET_KEY")
|
||||
CLERK_PUBLISHABLE_KEY = os.getenv("CLERK_PUBLISHABLE_KEY")
|
||||
|
||||
# Setup logging
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Configure CORS middleware
|
||||
# Configure CORS and Auth middleware
|
||||
cors_origins = os.getenv("ALLOWED_ORIGINS", "*").split(",")
|
||||
custom_middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=cors_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST", "OPTIONS"],
|
||||
allow_headers=["Content-Type", "Authorization", "X-Request-ID"],
|
||||
),
|
||||
]
|
||||
|
||||
# Create MCP Starlette sub-application (without auth wrapper)
|
||||
mcp_app = mcp_server.http_app(
|
||||
path="/",
|
||||
middleware=custom_middleware
|
||||
)
|
||||
# Import FastMCP Bearer Auth Provider
|
||||
from fastmcp.server.auth import BearerAuthProvider
|
||||
from fastmcp.server.auth.providers.bearer import RSAKeyPair
|
||||
|
||||
# Import Clerk SDK at module level for performance
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_SDK_AVAILABLE = True
|
||||
except ImportError:
|
||||
CLERK_SDK_AVAILABLE = False
|
||||
logger.warning("Clerk SDK not available - falling back to development mode")
|
||||
|
||||
# Configure Bearer token authentication based on ENABLE_AUTH
|
||||
auth_enabled = os.getenv("ENABLE_AUTH", "false").lower() == "true"
|
||||
bearer_auth = None
|
||||
|
||||
if CLERK_SECRET_KEY and CLERK_ISSUER:
|
||||
# Production: Use Clerk JWKS endpoint for token validation
|
||||
bearer_auth = BearerAuthProvider(
|
||||
jwks_uri=f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
issuer=None,
|
||||
algorithm="RS256",
|
||||
audience=None,
|
||||
required_scopes=[]
|
||||
)
|
||||
else:
|
||||
# Development: Generate RSA key pair for testing
|
||||
dev_key_pair = RSAKeyPair.generate()
|
||||
bearer_auth = BearerAuthProvider(
|
||||
public_key=dev_key_pair.public_key,
|
||||
issuer="https://dev.yargimcp.com",
|
||||
audience="dev-mcp-server",
|
||||
required_scopes=["yargi.read"]
|
||||
)
|
||||
|
||||
# Create MCP app with Bearer authentication
|
||||
mcp_server = create_app(auth=bearer_auth if auth_enabled else None)
|
||||
|
||||
# Create MCP Starlette sub-application with root path - mount will add /mcp prefix
|
||||
mcp_app = mcp_server.http_app(path="/")
|
||||
|
||||
|
||||
# Configure JSON encoder for proper Turkish character support
|
||||
import json
|
||||
from fastapi.responses import JSONResponse
|
||||
|
||||
class UTF8JSONResponse(JSONResponse):
|
||||
def __init__(self, content=None, status_code=200, headers=None, **kwargs):
|
||||
if headers is None:
|
||||
@@ -74,21 +114,32 @@ class UTF8JSONResponse(JSONResponse):
|
||||
separators=(",", ":"),
|
||||
).encode("utf-8")
|
||||
|
||||
# Create FastAPI wrapper application with MCP lifespan
|
||||
custom_middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=cors_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST", "OPTIONS", "DELETE"],
|
||||
allow_headers=["Content-Type", "Authorization", "X-Request-ID", "X-Session-ID"],
|
||||
),
|
||||
]
|
||||
|
||||
# Create FastAPI wrapper application
|
||||
app = FastAPI(
|
||||
title="Yargı MCP Server",
|
||||
description="MCP server for Turkish legal databases with OAuth authentication",
|
||||
version="0.1.0",
|
||||
middleware=custom_middleware,
|
||||
lifespan=mcp_app.lifespan, # MCP app lifespan
|
||||
default_response_class=UTF8JSONResponse # Use UTF-8 JSON encoder
|
||||
default_response_class=UTF8JSONResponse, # Use UTF-8 JSON encoder
|
||||
redirect_slashes=False # Disable to prevent 307 redirects on /mcp endpoint
|
||||
)
|
||||
|
||||
# Add Stripe webhook router to FastAPI
|
||||
app.include_router(stripe_router, prefix="/api")
|
||||
# Add auth-related routers to FastAPI (only if available)
|
||||
if stripe_router:
|
||||
app.include_router(stripe_router, prefix="/api/stripe")
|
||||
|
||||
# Add MCP Auth HTTP adapter to FastAPI (handles OAuth endpoints)
|
||||
app.include_router(mcp_auth_router)
|
||||
if mcp_auth_router:
|
||||
app.include_router(mcp_auth_router)
|
||||
|
||||
# Custom 401 exception handler for MCP spec compliance
|
||||
@app.exception_handler(401)
|
||||
@@ -107,157 +158,110 @@ async def custom_401_handler(request: Request, exc: HTTPException):
|
||||
|
||||
return response
|
||||
|
||||
# Mount MCP app as sub-application at /mcp-server to avoid path conflicts
|
||||
app.mount("/mcp-server", mcp_app)
|
||||
|
||||
# Add custom route to handle /mcp requests and forward to mounted app
|
||||
@app.api_route("/mcp", methods=["POST", "DELETE", "OPTIONS"])
|
||||
@app.api_route("/mcp/", methods=["POST", "DELETE", "OPTIONS"])
|
||||
async def mcp_protocol_handler(request: Request):
|
||||
"""Handle MCP protocol requests by forwarding to mounted app"""
|
||||
|
||||
# Handle DELETE requests for session termination
|
||||
if request.method == "DELETE":
|
||||
logger.info("DELETE request received for session termination")
|
||||
# For session termination, we just return 200 OK
|
||||
# The actual session cleanup is handled by the underlying MCP transport
|
||||
from starlette.responses import Response
|
||||
return Response(
|
||||
status_code=200,
|
||||
content="Session terminated successfully"
|
||||
)
|
||||
|
||||
# REQUIRED: Validate Bearer JWT tokens for all MCP requests
|
||||
auth_header = request.headers.get("Authorization")
|
||||
if not auth_header or not auth_header.startswith("Bearer "):
|
||||
logger.error("Missing or invalid Authorization header")
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Missing or invalid Authorization header. Bearer token required."
|
||||
)
|
||||
|
||||
token = auth_header.split(" ")[1]
|
||||
try:
|
||||
# Check if this is a mock token for development/testing
|
||||
if token.startswith("mock_clerk_jwt_"):
|
||||
logger.info(f"Using mock JWT token for development: {token[:30]}...")
|
||||
# For mock tokens, we'll allow access with a mock user
|
||||
request.state.user_id = "mock_user_dev"
|
||||
request.state.session_id = "mock_session_dev"
|
||||
request.state.token_scopes = ["read", "search"]
|
||||
logger.info("Mock JWT token accepted for development")
|
||||
elif token.startswith("eyJ"):
|
||||
# This looks like a real JWT token (starts with eyJ which is base64 encoded '{"')
|
||||
logger.info(f"Processing real JWT token: {token[:30]}...")
|
||||
# Validate real Clerk JWT token
|
||||
from clerk_backend_api import Clerk, models
|
||||
import jwt
|
||||
|
||||
# Decode JWT token and extract user info
|
||||
try:
|
||||
decoded_token = jwt.decode(token, options={"verify_signature": False})
|
||||
user_id = decoded_token.get("user_id") or decoded_token.get("sub")
|
||||
user_email = decoded_token.get("email")
|
||||
token_scopes = decoded_token.get("scopes", ["read", "search"])
|
||||
session_id = decoded_token.get("sid", "jwt_session")
|
||||
|
||||
logger.info(f"JWT token claims - user_id: {user_id}, email: {user_email}, scopes: {token_scopes}")
|
||||
|
||||
if user_id and user_email:
|
||||
# JWT token is signed by Clerk and contains valid user info
|
||||
request.state.user_id = user_id
|
||||
request.state.user_email = user_email
|
||||
request.state.session_id = session_id
|
||||
request.state.token_scopes = token_scopes
|
||||
logger.info(f"Real JWT token accepted for user: {user_id}")
|
||||
else:
|
||||
logger.error(f"Missing required fields in JWT token - user_id: {bool(user_id)}, email: {bool(user_email)}")
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Invalid token - missing user_id or email in claims"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"JWT token decoding failed: {e}")
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Invalid JWT token format"
|
||||
)
|
||||
else:
|
||||
# Invalid token format - doesn't start with expected patterns
|
||||
logger.error(f"Invalid token format: {token[:30]}...")
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail="Invalid token format - must be a valid JWT token"
|
||||
)
|
||||
|
||||
except HTTPException:
|
||||
# Re-raise HTTPException as-is
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"Bearer token validation failed: {str(e)}")
|
||||
raise HTTPException(
|
||||
status_code=401,
|
||||
detail=f"Token validation failed: {str(e)}"
|
||||
)
|
||||
|
||||
# Forward the request to the mounted MCP app
|
||||
async def receive():
|
||||
return await request.receive()
|
||||
|
||||
# Create new scope for the mounted app
|
||||
scope = request.scope.copy()
|
||||
scope["path"] = "/" # Root path for mounted app
|
||||
scope["path_info"] = "/"
|
||||
|
||||
# Capture the response
|
||||
response_parts = {"status": 200, "headers": [], "body": b""}
|
||||
|
||||
async def send(message):
|
||||
if message["type"] == "http.response.start":
|
||||
response_parts["status"] = message["status"]
|
||||
response_parts["headers"] = message["headers"]
|
||||
elif message["type"] == "http.response.body":
|
||||
response_parts["body"] += message.get("body", b"")
|
||||
|
||||
# Call the mounted MCP app
|
||||
await mcp_app(scope, receive, send)
|
||||
|
||||
# Return the response
|
||||
from starlette.responses import Response
|
||||
|
||||
# Convert ASGI headers to dict
|
||||
headers = {}
|
||||
for name, value in response_parts["headers"]:
|
||||
headers[name.decode()] = value.decode()
|
||||
|
||||
return Response(
|
||||
content=response_parts["body"],
|
||||
status_code=response_parts["status"],
|
||||
headers=headers
|
||||
)
|
||||
|
||||
|
||||
# SSE transport deprecated - removed
|
||||
|
||||
|
||||
# FastAPI health check endpoint
|
||||
# FastAPI health check endpoint - BEFORE mounting MCP app
|
||||
@app.get("/health")
|
||||
async def health_check():
|
||||
"""Health check endpoint for monitoring"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"auth_enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true"
|
||||
})
|
||||
}
|
||||
|
||||
# Add explicit redirect for /mcp to /mcp/ with method preservation
|
||||
@app.api_route("/mcp", methods=["GET", "POST", "HEAD", "OPTIONS"])
|
||||
async def redirect_to_slash(request: Request):
|
||||
"""Redirect /mcp to /mcp/ preserving HTTP method with 308"""
|
||||
from fastapi.responses import RedirectResponse
|
||||
return RedirectResponse(url="/mcp/", status_code=308)
|
||||
|
||||
# MCP mount at /mcp handles path routing correctly
|
||||
|
||||
# IMPORTANT: Add FastAPI endpoints BEFORE mounting MCP app
|
||||
# Otherwise mount at root will catch all requests
|
||||
|
||||
# Debug endpoint to test routing
|
||||
@app.get("/debug/test")
|
||||
async def debug_test():
|
||||
"""Debug endpoint to test if FastAPI routes work"""
|
||||
return {"message": "FastAPI routes working", "debug": True}
|
||||
|
||||
# Clerk CORS proxy endpoints
|
||||
@app.api_route("/clerk-proxy/{path:path}", methods=["GET", "POST", "PUT", "DELETE", "OPTIONS"])
|
||||
async def clerk_cors_proxy(request: Request, path: str):
|
||||
"""
|
||||
Proxy requests to Clerk to bypass CORS restrictions.
|
||||
Forwards requests from Claude AI to clerk.yargimcp.com with proper CORS headers.
|
||||
"""
|
||||
import httpx
|
||||
|
||||
# Build target URL
|
||||
clerk_url = f"https://clerk.yargimcp.com/{path}"
|
||||
|
||||
# Forward query parameters
|
||||
if request.url.query:
|
||||
clerk_url += f"?{request.url.query}"
|
||||
|
||||
# Copy headers (exclude host/origin)
|
||||
headers = dict(request.headers)
|
||||
headers.pop('host', None)
|
||||
headers.pop('origin', None)
|
||||
headers['origin'] = 'https://yargimcp.com' # Use our frontend domain
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient() as client:
|
||||
# Forward the request to Clerk
|
||||
if request.method == "OPTIONS":
|
||||
# Handle preflight
|
||||
response = await client.request(
|
||||
method=request.method,
|
||||
url=clerk_url,
|
||||
headers=headers
|
||||
)
|
||||
else:
|
||||
# Forward body for POST/PUT requests
|
||||
body = None
|
||||
if request.method in ["POST", "PUT", "PATCH"]:
|
||||
body = await request.body()
|
||||
|
||||
response = await client.request(
|
||||
method=request.method,
|
||||
url=clerk_url,
|
||||
headers=headers,
|
||||
content=body
|
||||
)
|
||||
|
||||
# Create response with CORS headers
|
||||
response_headers = dict(response.headers)
|
||||
response_headers.update({
|
||||
"Access-Control-Allow-Origin": "*",
|
||||
"Access-Control-Allow-Methods": "GET, POST, PUT, DELETE, OPTIONS",
|
||||
"Access-Control-Allow-Headers": "Content-Type, Authorization, Accept, Origin, X-Requested-With",
|
||||
"Access-Control-Allow-Credentials": "true",
|
||||
"Access-Control-Max-Age": "86400"
|
||||
})
|
||||
|
||||
return Response(
|
||||
content=response.content,
|
||||
status_code=response.status_code,
|
||||
headers=response_headers,
|
||||
media_type=response.headers.get("content-type")
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
return JSONResponse(
|
||||
{"error": "proxy_error", "message": str(e)},
|
||||
status_code=500,
|
||||
headers={"Access-Control-Allow-Origin": "*"}
|
||||
)
|
||||
|
||||
# FastAPI root endpoint
|
||||
@app.get("/")
|
||||
async def root():
|
||||
"""Root endpoint with service information"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"service": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases with OAuth authentication",
|
||||
"endpoints": {
|
||||
@@ -282,25 +286,26 @@ async def root():
|
||||
"Kamu İhale Kurulu (Public Procurement Authority)",
|
||||
"Rekabet Kurumu (Competition Authority)",
|
||||
"Sayıştay (Court of Accounts)",
|
||||
"KVKK (Personal Data Protection Authority)",
|
||||
"BDDK (Banking Regulation and Supervision Agency)",
|
||||
"Bedesten API (Multiple courts)"
|
||||
],
|
||||
"authentication": {
|
||||
"enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true",
|
||||
"type": "OAuth 2.0 via Clerk",
|
||||
"issuer": os.getenv("CLERK_ISSUER", "https://clerk.accounts.dev"),
|
||||
"issuer": CLERK_ISSUER,
|
||||
"providers": ["google"],
|
||||
"flow": "authorization_code"
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
# OAuth 2.0 Authorization Server Metadata proxy (for MCP clients that can't reach Clerk directly)
|
||||
# MCP Auth Toolkit expects this to be under /mcp/.well-known/oauth-authorization-server
|
||||
@app.get("/mcp/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server():
|
||||
"""OAuth 2.0 Authorization Server Metadata proxy to Clerk - MCP Auth Toolkit standard location"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
# OAuth 2.0 Authorization Server Metadata - MCP standard location
|
||||
@app.get("/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server_root():
|
||||
"""OAuth 2.0 Authorization Server Metadata - root level for compatibility"""
|
||||
return {
|
||||
"issuer": BASE_URL, # Use BASE_URL as issuer for MCP integration
|
||||
"authorization_endpoint": f"{BASE_URL}/auth/login",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
@@ -314,15 +319,15 @@ async def oauth_authorization_server():
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
}
|
||||
|
||||
# Claude AI MCP specific endpoint format
|
||||
# Claude AI MCP specific endpoint format - suffix versions
|
||||
@app.get("/.well-known/oauth-authorization-server/mcp")
|
||||
async def oauth_authorization_server_mcp_suffix():
|
||||
"""OAuth 2.0 Authorization Server Metadata - Claude AI MCP specific format"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
return {
|
||||
"issuer": BASE_URL, # Use BASE_URL as issuer for MCP integration
|
||||
"authorization_endpoint": f"{BASE_URL}/auth/login",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
@@ -336,12 +341,12 @@ async def oauth_authorization_server_mcp_suffix():
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
}
|
||||
|
||||
@app.get("/.well-known/oauth-protected-resource/mcp")
|
||||
async def oauth_protected_resource_mcp_suffix():
|
||||
"""OAuth 2.0 Protected Resource Metadata - Claude AI MCP specific format"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [
|
||||
BASE_URL
|
||||
@@ -350,78 +355,13 @@ async def oauth_protected_resource_mcp_suffix():
|
||||
"bearer_methods_supported": ["header"],
|
||||
"resource_documentation": f"{BASE_URL}/mcp",
|
||||
"resource_policy_uri": f"{BASE_URL}/privacy"
|
||||
})
|
||||
|
||||
# Keep root level for compatibility with some MCP clients
|
||||
@app.get("/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server_root():
|
||||
"""OAuth 2.0 Authorization Server Metadata proxy to Clerk - root level for compatibility"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"token_endpoint_auth_methods_supported": ["client_secret_basic", "none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"subject_types_supported": ["public"],
|
||||
"id_token_signing_alg_values_supported": ["RS256"],
|
||||
"claims_supported": ["sub", "iss", "aud", "exp", "iat", "email", "name"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
|
||||
# MCP endpoint info for GET requests (ChatGPT compatibility)
|
||||
@app.get("/mcp")
|
||||
async def mcp_info():
|
||||
"""MCP endpoint information for discovery"""
|
||||
return JSONResponse({
|
||||
"mcp_server": True,
|
||||
"name": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"protocol": "mcp/1.0",
|
||||
"transport": ["http"],
|
||||
"authentication_required": True,
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": "https://yargimcp.com/sign-in?redirect_url=https://api.yargimcp.com/auth/mcp-callback",
|
||||
"token_url": f"{BASE_URL}/auth/mcp-token",
|
||||
"scopes": ["read", "search"],
|
||||
"provider": "clerk"
|
||||
},
|
||||
"endpoints": {
|
||||
"mcp_protocol": "/mcp",
|
||||
"discovery": "/mcp/discovery",
|
||||
"well_known": "/.well-known/mcp",
|
||||
"health": "/health",
|
||||
"oauth_login": "/auth/login"
|
||||
},
|
||||
"capabilities": {
|
||||
"tools": True,
|
||||
"resources": True,
|
||||
"prompts": False
|
||||
},
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"usage": {
|
||||
"note": "This is an MCP server. Use POST to /mcp/ with proper MCP protocol headers.",
|
||||
"headers_required": [
|
||||
"Content-Type: application/json",
|
||||
"Accept: application/json",
|
||||
"Authorization: Bearer <token>",
|
||||
"X-Session-ID: <session-id>"
|
||||
]
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
# OAuth 2.0 Protected Resource Metadata (RFC 9728) - MCP Spec Required
|
||||
@app.get("/.well-known/oauth-protected-resource")
|
||||
async def oauth_protected_resource():
|
||||
"""OAuth 2.0 Protected Resource Metadata as required by MCP spec"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [
|
||||
BASE_URL
|
||||
@@ -430,13 +370,13 @@ async def oauth_protected_resource():
|
||||
"bearer_methods_supported": ["header"],
|
||||
"resource_documentation": f"{BASE_URL}/mcp",
|
||||
"resource_policy_uri": f"{BASE_URL}/privacy"
|
||||
})
|
||||
}
|
||||
|
||||
# Standard well-known discovery endpoint
|
||||
@app.get("/.well-known/mcp")
|
||||
async def well_known_mcp():
|
||||
"""Standard MCP discovery endpoint"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"mcp_server": {
|
||||
"name": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
@@ -449,13 +389,13 @@ async def well_known_mcp():
|
||||
"capabilities": ["tools", "resources"],
|
||||
"tools_count": len(mcp_server._tool_manager._tools)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
# MCP Discovery endpoint for ChatGPT integration
|
||||
@app.get("/mcp/discovery")
|
||||
async def mcp_discovery():
|
||||
"""MCP Discovery endpoint for ChatGPT and other MCP clients"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"name": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"version": "0.1.0",
|
||||
@@ -465,7 +405,7 @@ async def mcp_discovery():
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": "/auth/login",
|
||||
"token_url": "/auth/callback",
|
||||
"token_url": "/token",
|
||||
"scopes": ["read", "search"],
|
||||
"provider": "clerk"
|
||||
},
|
||||
@@ -479,7 +419,7 @@ async def mcp_discovery():
|
||||
"url": BASE_URL,
|
||||
"email": "support@yargi-mcp.dev"
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
# FastAPI status endpoint
|
||||
@app.get("/status")
|
||||
@@ -492,85 +432,54 @@ async def status():
|
||||
"description": tool.description[:100] + "..." if len(tool.description) > 100 else tool.description
|
||||
})
|
||||
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "operational",
|
||||
"tools": tools,
|
||||
"total_tools": len(tools),
|
||||
"transport": "streamable_http",
|
||||
"architecture": "FastAPI wrapper + MCP Starlette sub-app",
|
||||
"auth_status": "enabled" if os.getenv("ENABLE_AUTH", "false").lower() == "true" else "disabled"
|
||||
})
|
||||
}
|
||||
|
||||
# Note: JWT token validation is now handled entirely by Clerk
|
||||
# All authentication flows use Clerk JWT tokens directly
|
||||
|
||||
async def validate_clerk_session(request: Request, clerk_token: str = None) -> str:
|
||||
"""Validate Clerk session from cookies or JWT token and return user_id"""
|
||||
logger.info(f"Validating Clerk session - token provided: {bool(clerk_token)}")
|
||||
# Simplified OAuth session validation for callback endpoints only
|
||||
async def validate_clerk_session_for_oauth(request: Request, clerk_token: str = None) -> str:
|
||||
"""Validate Clerk session for OAuth callback endpoints only (not for MCP endpoints)"""
|
||||
|
||||
try:
|
||||
# Try to import Clerk SDK
|
||||
from clerk_backend_api import Clerk
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
# Use Clerk SDK if available
|
||||
if not CLERK_SDK_AVAILABLE:
|
||||
raise ImportError("Clerk SDK not available")
|
||||
clerk = Clerk(bearer_auth=CLERK_SECRET_KEY)
|
||||
|
||||
# Try JWT token first (from URL parameter)
|
||||
if clerk_token:
|
||||
logger.info("Validating Clerk JWT token from URL parameter")
|
||||
try:
|
||||
# Extract session_id from JWT token and verify with Clerk
|
||||
import jwt
|
||||
decoded_token = jwt.decode(clerk_token, options={"verify_signature": False})
|
||||
session_id = decoded_token.get("sid") # Use standard JWT 'sid' claim
|
||||
|
||||
if session_id:
|
||||
# Verify with Clerk using session_id
|
||||
session = clerk.sessions.verify(session_id=session_id, token=clerk_token)
|
||||
user_id = session.user_id if session else None
|
||||
|
||||
if user_id:
|
||||
logger.info(f"JWT token validation successful - user_id: {user_id}")
|
||||
return user_id
|
||||
else:
|
||||
logger.error("JWT token validation failed - no user_id in session")
|
||||
else:
|
||||
logger.error("No session_id found in JWT token")
|
||||
return "oauth_user_from_token"
|
||||
except Exception as e:
|
||||
logger.error(f"JWT token validation failed: {str(e)}")
|
||||
# Fall through to cookie validation
|
||||
|
||||
pass
|
||||
|
||||
# Fallback to cookie validation
|
||||
logger.info("Attempting cookie-based session validation")
|
||||
clerk_session = request.cookies.get("__session")
|
||||
if not clerk_session:
|
||||
logger.error("No Clerk session cookie found")
|
||||
raise HTTPException(status_code=401, detail="No Clerk session found")
|
||||
|
||||
|
||||
# Validate session with Clerk
|
||||
session = clerk.sessions.verify_session(clerk_session)
|
||||
logger.info(f"Cookie session validation successful - user_id: {session.user_id}")
|
||||
return session.user_id
|
||||
|
||||
except ImportError:
|
||||
# Fallback for development without Clerk SDK
|
||||
logger.warning("Clerk SDK not available - using development fallback")
|
||||
return "dev_user_123"
|
||||
except Exception as e:
|
||||
logger.error(f"Session validation failed: {str(e)}")
|
||||
raise HTTPException(status_code=401, detail=f"Session validation failed: {str(e)}")
|
||||
raise HTTPException(status_code=401, detail=f"OAuth session validation failed: {str(e)}")
|
||||
|
||||
# MCP OAuth Callback Endpoint
|
||||
@app.get("/auth/mcp-callback")
|
||||
async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
"""Handle OAuth callback for MCP token generation"""
|
||||
logger.info(f"MCP OAuth callback - clerk_token provided: {bool(clerk_token)}")
|
||||
|
||||
try:
|
||||
# Validate Clerk session with JWT token support
|
||||
user_id = await validate_clerk_session(request, clerk_token)
|
||||
logger.info(f"User authenticated successfully - user_id: {user_id}")
|
||||
|
||||
# Use the Clerk JWT token directly (no need to generate custom token)
|
||||
logger.info("User authenticated successfully via Clerk")
|
||||
user_id = await validate_clerk_session_for_oauth(request, clerk_token)
|
||||
|
||||
# Return success response
|
||||
return HTMLResponse(f"""
|
||||
@@ -606,7 +515,6 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
""")
|
||||
|
||||
except HTTPException as e:
|
||||
logger.error(f"MCP OAuth callback failed: {e.detail}")
|
||||
return HTMLResponse(f"""
|
||||
<html>
|
||||
<head>
|
||||
@@ -632,7 +540,6 @@ async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
</html>
|
||||
""", status_code=e.status_code)
|
||||
except Exception as e:
|
||||
logger.error(f"Unexpected error in MCP OAuth callback: {str(e)}")
|
||||
return HTMLResponse(f"""
|
||||
<html>
|
||||
<head>
|
||||
@@ -657,22 +564,26 @@ async def mcp_token_endpoint(request: Request):
|
||||
"""OAuth2 token endpoint for MCP clients - returns Clerk JWT token info"""
|
||||
try:
|
||||
# Validate Clerk session
|
||||
user_id = await validate_clerk_session(request)
|
||||
user_id = await validate_clerk_session_for_oauth(request)
|
||||
|
||||
return JSONResponse({
|
||||
return {
|
||||
"message": "Use your Clerk JWT token directly with Bearer authentication",
|
||||
"token_type": "Bearer",
|
||||
"scope": "yargi.read",
|
||||
"user_id": user_id,
|
||||
"instructions": "Include 'Authorization: Bearer YOUR_CLERK_JWT_TOKEN' in your requests"
|
||||
})
|
||||
}
|
||||
except HTTPException as e:
|
||||
return JSONResponse(
|
||||
status_code=e.status_code,
|
||||
content={"error": "invalid_request", "error_description": e.detail}
|
||||
)
|
||||
|
||||
# Note: Only HTTP transport supported - SSE transport deprecated
|
||||
# Mount MCP app at /mcp/ with trailing slash
|
||||
app.mount("/mcp/", mcp_app)
|
||||
|
||||
# Set the lifespan context after mounting
|
||||
app.router.lifespan_context = mcp_app.lifespan
|
||||
|
||||
# Export for uvicorn
|
||||
__all__ = ["app"]
|
||||
@@ -102,8 +102,22 @@ class BedestenApiClient:
|
||||
response_json = response.json()
|
||||
doc_response = BedestenDocumentResponse(**response_json)
|
||||
|
||||
# Decode base64 content
|
||||
content_bytes = base64.b64decode(doc_response.data.content)
|
||||
# Add null safety checks for document data
|
||||
if not hasattr(doc_response, 'data') or doc_response.data is None:
|
||||
raise ValueError("Document response does not contain data")
|
||||
|
||||
if not hasattr(doc_response.data, 'content') or doc_response.data.content is None:
|
||||
raise ValueError("Document data does not contain content")
|
||||
|
||||
if not hasattr(doc_response.data, 'mimeType') or doc_response.data.mimeType is None:
|
||||
raise ValueError("Document data does not contain mimeType")
|
||||
|
||||
# Decode base64 content with error handling
|
||||
try:
|
||||
content_bytes = base64.b64decode(doc_response.data.content)
|
||||
except Exception as e:
|
||||
raise ValueError(f"Failed to decode base64 content: {str(e)}")
|
||||
|
||||
mime_type = doc_response.data.mimeType
|
||||
|
||||
logger.info(f"BedestenApiClient: Document mime type: {mime_type}")
|
||||
|
||||
@@ -21,7 +21,7 @@ class BedestenSearchData(BaseModel):
|
||||
pageSize: int = Field(..., description="Results per page (1-10)")
|
||||
pageNumber: int = Field(..., description="Page number (1-indexed)")
|
||||
itemTypeList: List[str] = Field(..., description="Court type filter (YARGITAYKARARI/DANISTAYKARAR/YERELHUKUK/ISTINAFHUKUK/KYB)")
|
||||
phrase: str = Field(..., description="Search phrase (use \"exact phrase\" for precise matching)")
|
||||
phrase: str = Field(..., description="Search phrase. Supports: 'word', \"exact phrase\", +required, -exclude, AND/OR/NOT operators. No wildcards or regex.")
|
||||
birimAdi: BirimAdiEnum = Field("ALL", description="""
|
||||
Chamber filter (optional). Abbreviated values with Turkish names:
|
||||
• Yargıtay: H1-H23 (1-23. Hukuk Dairesi), C1-C23 (1-23. Ceza Dairesi), HGK (Hukuk Genel Kurulu), CGK (Ceza Genel Kurulu), BGK (Büyük Genel Kurulu), HBK (Hukuk Daireleri Başkanlar Kurulu), CBK (Ceza Daireleri Başkanlar Kurulu)
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
# fly.toml app configuration file for yargi-mcp-noauth
|
||||
#
|
||||
# See https://fly.io/docs/reference/configuration/ for information about how to use this file.
|
||||
#
|
||||
|
||||
app = 'yargi-mcp-free'
|
||||
primary_region = 'fra'
|
||||
|
||||
[env]
|
||||
ENABLE_AUTH = "false"
|
||||
HOST = "0.0.0.0"
|
||||
PORT = "8000"
|
||||
LOG_LEVEL = "info"
|
||||
|
||||
[build]
|
||||
|
||||
[http_service]
|
||||
internal_port = 8000
|
||||
force_https = true
|
||||
auto_stop_machines = 'off'
|
||||
auto_start_machines = true
|
||||
min_machines_running = 1
|
||||
processes = ['app']
|
||||
|
||||
# Enable connection persistence for MCP sessions
|
||||
[http_service.concurrency]
|
||||
type = "connections"
|
||||
hard_limit = 100
|
||||
soft_limit = 80
|
||||
|
||||
[[vm]]
|
||||
memory = '1gb'
|
||||
cpu_kind = 'shared'
|
||||
cpus = 1
|
||||
|
||||
[deploy]
|
||||
strategy = "immediate"
|
||||
|
||||
[processes]
|
||||
app = "python asgi_app.py"
|
||||
|
||||
[checks.http_health] # keep MCP /health live
|
||||
type = "http"
|
||||
interval = "30s"
|
||||
timeout = "10s"
|
||||
path = "/health"
|
||||
@@ -17,10 +17,16 @@ LOG_LEVEL = "info"
|
||||
[http_service]
|
||||
internal_port = 8000
|
||||
force_https = true
|
||||
auto_stop_machines = 'stop'
|
||||
auto_stop_machines = 'off'
|
||||
auto_start_machines = true
|
||||
min_machines_running = 0
|
||||
min_machines_running = 1
|
||||
processes = ['app']
|
||||
|
||||
# Enable connection persistence for MCP sessions
|
||||
[http_service.concurrency]
|
||||
type = "connections"
|
||||
hard_limit = 100
|
||||
soft_limit = 80
|
||||
|
||||
[[vm]]
|
||||
memory = '1gb'
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,475 @@
|
||||
# kik_mcp_module/client_v2.py
|
||||
|
||||
import httpx
|
||||
import logging
|
||||
import uuid
|
||||
import ssl
|
||||
import os
|
||||
from typing import Optional
|
||||
from datetime import datetime
|
||||
|
||||
# Cryptography imports for AES-256-CBC encryption of document IDs
|
||||
try:
|
||||
from cryptography.hazmat.primitives.ciphers import Cipher, algorithms, modes
|
||||
from cryptography.hazmat.backends import default_backend
|
||||
HAS_CRYPTOGRAPHY = True
|
||||
except ImportError:
|
||||
HAS_CRYPTOGRAPHY = False
|
||||
|
||||
from .models_v2 import (
|
||||
KikV2DecisionType, KikV2SearchPayload, KikV2SearchPayloadDk, KikV2SearchPayloadMk,
|
||||
KikV2RequestData, KikV2QueryRequest, KikV2KeyValuePair,
|
||||
KikV2SearchResponse, KikV2SearchResponseDk, KikV2SearchResponseMk,
|
||||
KikV2SearchResult, KikV2CompactDecision, KikV2DocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class KikV2ApiClient:
|
||||
"""
|
||||
New KIK v2 API Client for https://ekapv2.kik.gov.tr
|
||||
|
||||
This client uses the modern JSON-based API endpoint that provides
|
||||
better structured data compared to the legacy form-based API.
|
||||
"""
|
||||
|
||||
BASE_URL = "https://ekapv2.kik.gov.tr"
|
||||
|
||||
# Endpoint mappings for different decision types
|
||||
ENDPOINTS = {
|
||||
KikV2DecisionType.UYUSMAZLIK: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlari",
|
||||
KikV2DecisionType.DUZENLEYICI: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariDk",
|
||||
KikV2DecisionType.MAHKEME: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariMk"
|
||||
}
|
||||
|
||||
# AES-256-CBC encryption key for document ID encryption (reverse engineered from ekapv2.kik.gov.tr Angular app)
|
||||
# This key is used to encrypt numeric gundemMaddesiId values to 64-character hex hashes for document URLs
|
||||
DOCUMENT_ID_ENCRYPTION_KEY = bytes([
|
||||
236, 193, 164, 43, 12, 135, 121, 170, 4, 244, 123, 219, 82, 158, 124, 174,
|
||||
174, 228, 219, 174, 208, 104, 174, 120, 32, 76, 250, 4, 143, 159, 211, 176
|
||||
])
|
||||
|
||||
@staticmethod
|
||||
def encrypt_document_id(numeric_id: str) -> str:
|
||||
"""
|
||||
Encrypt a numeric KİK gundemMaddesiId to the 64-character hex hash
|
||||
used in document URLs.
|
||||
|
||||
Algorithm: AES-256-CBC with PKCS7 padding
|
||||
Output format: IV (16 bytes hex) + Ciphertext (16 bytes hex) = 64 chars
|
||||
|
||||
Args:
|
||||
numeric_id: The numeric document ID from search results (e.g., "177280")
|
||||
|
||||
Returns:
|
||||
64-character hex string for use in document URL KararId parameter
|
||||
"""
|
||||
if not HAS_CRYPTOGRAPHY:
|
||||
raise ImportError("cryptography library required for document ID encryption")
|
||||
|
||||
# Generate random IV (16 bytes)
|
||||
iv = os.urandom(16)
|
||||
|
||||
# Create AES-CBC cipher with the encryption key
|
||||
cipher = Cipher(
|
||||
algorithms.AES(KikV2ApiClient.DOCUMENT_ID_ENCRYPTION_KEY),
|
||||
modes.CBC(iv),
|
||||
backend=default_backend()
|
||||
)
|
||||
encryptor = cipher.encryptor()
|
||||
|
||||
# Encode plaintext and apply PKCS7 padding
|
||||
plaintext = numeric_id.encode('utf-8')
|
||||
block_size = 16
|
||||
padding_len = block_size - (len(plaintext) % block_size)
|
||||
padded_plaintext = plaintext + bytes([padding_len] * padding_len)
|
||||
|
||||
# Encrypt
|
||||
ciphertext = encryptor.update(padded_plaintext) + encryptor.finalize()
|
||||
|
||||
# Return IV + ciphertext as lowercase hex (64 characters total)
|
||||
return iv.hex() + ciphertext.hex()
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
# Create SSL context with legacy server support
|
||||
ssl_context = ssl.create_default_context()
|
||||
ssl_context.check_hostname = False
|
||||
ssl_context.verify_mode = ssl.CERT_NONE
|
||||
|
||||
# Enable legacy server connect option for older SSL implementations (Python 3.12+)
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
|
||||
# Set broader cipher suite support including legacy ciphers
|
||||
ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
verify=ssl_context,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Language": "tr",
|
||||
"Content-Type": "application/json",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": f"{self.BASE_URL}/sorgulamalar/kurul-kararlari",
|
||||
"Sec-Fetch-Dest": "empty",
|
||||
"Sec-Fetch-Mode": "cors",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36",
|
||||
"api-version": "v1",
|
||||
"sec-ch-ua": '"Not;A=Brand";v="99", "Google Chrome";v="139", "Chromium";v="139"',
|
||||
"sec-ch-ua-mobile": "?0",
|
||||
"sec-ch-ua-platform": '"macOS"'
|
||||
},
|
||||
timeout=request_timeout
|
||||
)
|
||||
|
||||
# Generate security headers (these might need to be updated based on API requirements)
|
||||
self.security_headers = self._generate_security_headers()
|
||||
|
||||
def _generate_security_headers(self) -> dict:
|
||||
"""
|
||||
Generate the custom security headers required by KIK v2 API.
|
||||
These headers appear to be for request validation/encryption.
|
||||
"""
|
||||
# Generate a random GUID for each session
|
||||
request_guid = str(uuid.uuid4())
|
||||
|
||||
# These are example values - in a real implementation, these might need
|
||||
# to be calculated based on the request content or session
|
||||
return {
|
||||
"X-Custom-Request-Guid": request_guid,
|
||||
"X-Custom-Request-R8id": "hwnOjsN8qdgtDw70x3sKkxab0rj2bQ8Uph4+C+oU+9AMmQqRN3eMOEEeet748DOf",
|
||||
"X-Custom-Request-Siv": "p2IQRTitF8z7I39nBjdAqA==",
|
||||
"X-Custom-Request-Ts": "1vB3Wwrt8YQ5U6t3XAzZ+Q=="
|
||||
}
|
||||
|
||||
def _build_search_payload(self,
|
||||
decision_type: KikV2DecisionType,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = ""):
|
||||
"""Build the search payload for KIK v2 API."""
|
||||
|
||||
key_value_pairs = []
|
||||
|
||||
# Add non-empty search criteria
|
||||
if karar_metni:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=karar_metni))
|
||||
|
||||
if karar_no:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararNo", value=karar_no))
|
||||
|
||||
if basvuran:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BasvuranAdi", value=basvuran))
|
||||
|
||||
if idare_adi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="IdareAdi", value=idare_adi))
|
||||
|
||||
if baslangic_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BaslangicTarihi", value=baslangic_tarihi))
|
||||
|
||||
if bitis_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BitisTarihi", value=bitis_tarihi))
|
||||
|
||||
# If no search criteria provided, use a generic search
|
||||
if not key_value_pairs:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=""))
|
||||
|
||||
query_request = KikV2QueryRequest(keyValueOfstringanyType=key_value_pairs)
|
||||
request_data = KikV2RequestData(keyValuePairs=query_request)
|
||||
|
||||
# Return appropriate payload based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
return KikV2SearchPayload(sorgulaKurulKararlari=request_data)
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
return KikV2SearchPayloadDk(sorgulaKurulKararlariDk=request_data)
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
return KikV2SearchPayloadMk(sorgulaKurulKararlariMk=request_data)
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
async def search_decisions(self,
|
||||
decision_type: KikV2DecisionType = KikV2DecisionType.UYUSMAZLIK,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = "") -> KikV2SearchResult:
|
||||
"""
|
||||
Search KIK decisions using the v2 API.
|
||||
|
||||
Args:
|
||||
decision_type: Type of decision to search (uyusmazlik/duzenleyici/mahkeme)
|
||||
karar_metni: Decision text search
|
||||
karar_no: Decision number (e.g., "2025/UH.II-1801")
|
||||
basvuran: Applicant name
|
||||
idare_adi: Administration name
|
||||
baslangic_tarihi: Start date (YYYY-MM-DD format)
|
||||
bitis_tarihi: End date (YYYY-MM-DD format)
|
||||
|
||||
Returns:
|
||||
KikV2SearchResult with compact decision list
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Searching {decision_type.value} decisions with criteria - karar_metni: '{karar_metni}', karar_no: '{karar_no}', basvuran: '{basvuran}'")
|
||||
|
||||
try:
|
||||
# Build request payload
|
||||
payload = self._build_search_payload(
|
||||
decision_type=decision_type,
|
||||
karar_metni=karar_metni,
|
||||
karar_no=karar_no,
|
||||
basvuran=basvuran,
|
||||
idare_adi=idare_adi,
|
||||
baslangic_tarihi=baslangic_tarihi,
|
||||
bitis_tarihi=bitis_tarihi
|
||||
)
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Get the appropriate endpoint for this decision type
|
||||
endpoint = self.ENDPOINTS[decision_type]
|
||||
|
||||
# Make API request
|
||||
response = await self.http_client.post(
|
||||
endpoint,
|
||||
json=payload.model_dump(),
|
||||
headers=headers
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
response_data = response.json()
|
||||
|
||||
logger.debug(f"KikV2ApiClient: Raw API response structure: {type(response_data)}")
|
||||
|
||||
# Parse the API response based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
api_response = KikV2SearchResponse(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariResponse.SorgulaKurulKararlariResult
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
api_response = KikV2SearchResponseDk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariDkResponse.SorgulaKurulKararlariDkResult
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
api_response = KikV2SearchResponseMk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariMkResponse.SorgulaKurulKararlariMkResult
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
# Check for API errors
|
||||
if result_data.hataKodu and result_data.hataKodu != "0":
|
||||
logger.warning(f"KikV2ApiClient: API returned error - Code: {result_data.hataKodu}, Message: {result_data.hataMesaji}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code=result_data.hataKodu,
|
||||
error_message=result_data.hataMesaji
|
||||
)
|
||||
|
||||
# Convert to compact format
|
||||
compact_decisions = []
|
||||
total_count = 0
|
||||
|
||||
for decision_group in result_data.KurulKararTutanakDetayListesi:
|
||||
for decision_detail in decision_group.KurulKararTutanakDetayi:
|
||||
compact_decision = KikV2CompactDecision(
|
||||
kararNo=decision_detail.kararNo,
|
||||
kararTarihi=decision_detail.kararTarihi,
|
||||
basvuran=decision_detail.basvuran,
|
||||
idareAdi=decision_detail.idareAdi,
|
||||
basvuruKonusu=decision_detail.basvuruKonusu,
|
||||
gundemMaddesiId=decision_detail.gundemMaddesiId,
|
||||
decision_type=decision_type.value
|
||||
)
|
||||
compact_decisions.append(compact_decision)
|
||||
total_count += 1
|
||||
|
||||
logger.info(f"KikV2ApiClient: Found {total_count} decisions")
|
||||
|
||||
return KikV2SearchResult(
|
||||
decisions=compact_decisions,
|
||||
total_records=total_count,
|
||||
page=1,
|
||||
error_code="0",
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"KikV2ApiClient: HTTP error during search: {e.response.status_code} - {e.response.text}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="HTTP_ERROR",
|
||||
error_message=f"HTTP {e.response.status_code}: {e.response.text}"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Unexpected error during search: {str(e)}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="UNEXPECTED_ERROR",
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def get_document_markdown(self, document_id: str) -> KikV2DocumentMarkdown:
|
||||
"""
|
||||
Get KİK decision document content in Markdown format.
|
||||
|
||||
This method uses a two-step process:
|
||||
1. Call GetSorgulamaUrl endpoint to get the actual document URL
|
||||
2. Use httpx to fetch the document content
|
||||
|
||||
Args:
|
||||
document_id: The gundemMaddesiId from search results
|
||||
|
||||
Returns:
|
||||
KikV2DocumentMarkdown with document content converted to Markdown
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Getting document for ID: {document_id}")
|
||||
|
||||
if not document_id or not document_id.strip():
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Document ID is required"
|
||||
)
|
||||
|
||||
try:
|
||||
# Step 1: Get the actual document URL using GetSorgulamaUrl endpoint
|
||||
logger.info(f"KikV2ApiClient: Step 1 - Getting document URL for ID: {document_id}")
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Call GetSorgulamaUrl to get the real document URL
|
||||
url_payload = {"sorguSayfaTipi": 2} # As shown in curl example
|
||||
|
||||
url_response = await self.http_client.post(
|
||||
"/b_ihalearaclari/api/KurulKararlari/GetSorgulamaUrl",
|
||||
json=url_payload,
|
||||
headers=headers
|
||||
)
|
||||
|
||||
url_response.raise_for_status()
|
||||
url_data = url_response.json()
|
||||
|
||||
# Get the base document URL from API response
|
||||
base_document_url = url_data.get("sorgulamaUrl", "")
|
||||
if not base_document_url:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Could not get document URL from GetSorgulamaUrl API"
|
||||
)
|
||||
|
||||
# If document_id is numeric, encrypt it to get the KararId hash
|
||||
# The web interface uses AES-256-CBC encrypted hashes for document URLs
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID {document_id} to hash: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt document ID, using as-is: {enc_error}")
|
||||
|
||||
# Construct full document URL with the encrypted KararId
|
||||
document_url = f"{base_document_url}?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Retrieved document URL: {document_url}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error getting document URL for ID {document_id}: {str(e)}")
|
||||
# Fallback to old method if GetSorgulamaUrl fails
|
||||
# Also encrypt numeric IDs in fallback path
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID in fallback: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt in fallback: {enc_error}")
|
||||
document_url = f"https://ekap.kik.gov.tr/EKAP/Vatandas/KurulKararGoster.aspx?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Falling back to direct URL: {document_url}")
|
||||
|
||||
try:
|
||||
# Step 2: Use httpx to get the document content
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Using httpx to retrieve document from: {document_url}")
|
||||
|
||||
# Create a separate httpx client for document retrieval with HTML headers
|
||||
doc_ssl_context = ssl.create_default_context()
|
||||
doc_ssl_context.check_hostname = False
|
||||
doc_ssl_context.verify_mode = ssl.CERT_NONE
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
doc_ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
doc_ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
verify=doc_ssl_context,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "tr,en-US;q=0.5",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=60.0,
|
||||
follow_redirects=True
|
||||
) as doc_client:
|
||||
response = await doc_client.get(document_url)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
logger.info(f"KikV2ApiClient: Retrieved content via httpx, length: {len(html_content)}")
|
||||
|
||||
# Convert HTML to Markdown using MarkItDown with BytesIO
|
||||
try:
|
||||
from markitdown import MarkItDown
|
||||
from io import BytesIO
|
||||
|
||||
md = MarkItDown()
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = BytesIO(html_bytes)
|
||||
|
||||
result = md.convert_stream(html_stream, file_extension=".html")
|
||||
markdown_content = result.text_content
|
||||
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content=markdown_content,
|
||||
source_url=document_url,
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except ImportError:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="MarkItDown library not available",
|
||||
source_url=document_url,
|
||||
error_message="MarkItDown library not installed"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error retrieving document {document_id}: {str(e)}")
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url=document_url,
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("KikV2ApiClient: HTTP client session closed.")
|
||||
@@ -1,74 +0,0 @@
|
||||
# kik_mcp_module/models.py
|
||||
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
import base64 # Base64 encoding/decoding için
|
||||
|
||||
class KikKararTipi(str, Enum):
|
||||
"""Enum for KIK (Public Procurement Authority) Decision Types."""
|
||||
UYUSMAZLIK = "rbUyusmazlik"
|
||||
DUZENLEYICI = "rbDuzenleyici"
|
||||
MAHKEME = "rbMahkeme"
|
||||
|
||||
class KikSearchRequest(BaseModel):
|
||||
"""Model for KIK Decision search criteria."""
|
||||
karar_tipi: KikKararTipi = Field(KikKararTipi.UYUSMAZLIK, description="Type")
|
||||
karar_no: str = Field("", description="No")
|
||||
karar_tarihi_baslangic: str = Field("", description="Start", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
karar_tarihi_bitis: str = Field("", description="End", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
resmi_gazete_sayisi: str = Field("", description="Gazette")
|
||||
resmi_gazete_tarihi: str = Field("", description="Date", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
basvuru_konusu_ihale: str = Field("", description="Subject")
|
||||
basvuru_sahibi: str = Field("", description="Applicant")
|
||||
ihaleyi_yapan_idare: str = Field("", description="Entity")
|
||||
yil: str = Field("", description="Year")
|
||||
karar_metni: str = Field("", description="Text")
|
||||
page: int = Field(1, ge=1, description="Page")
|
||||
|
||||
class KikDecisionEntry(BaseModel):
|
||||
"""Represents a single decision entry from KIK search results."""
|
||||
preview_event_target: str = Field(..., description="Event target")
|
||||
karar_no_str: str = Field(..., alias="kararNo", description="Decision number")
|
||||
karar_tipi: KikKararTipi = Field(..., description="Decision type")
|
||||
|
||||
karar_tarihi_str: str = Field(..., alias="kararTarihi", description="Date")
|
||||
idare_str: str = Field("", alias="idare", description="Entity")
|
||||
basvuru_sahibi_str: str = Field("", alias="basvuruSahibi", description="Applicant")
|
||||
ihale_konusu_str: str = Field("", alias="ihaleKonusu", description="Subject")
|
||||
|
||||
@computed_field
|
||||
@property
|
||||
def karar_id(self) -> str:
|
||||
"""
|
||||
A Base64 encoded unique ID for the decision, combining decision type and number.
|
||||
Format before encoding: "{karar_tipi.value}|{karar_no_str}"
|
||||
"""
|
||||
combined_key = f"{self.karar_tipi.value}|{self.karar_no_str}"
|
||||
return base64.b64encode(combined_key.encode('utf-8')).decode('utf-8')
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikSearchResult(BaseModel):
|
||||
"""Model for KIK search results."""
|
||||
decisions: List[KikDecisionEntry]
|
||||
total_records: int = 0
|
||||
current_page: int = 1
|
||||
|
||||
class KikDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
KIK decision document, with Markdown content potentially paginated.
|
||||
"""
|
||||
retrieved_with_karar_id: Optional[str] = Field(None, description="Request ID")
|
||||
retrieved_karar_no: Optional[str] = Field(None, description="Decision number")
|
||||
retrieved_karar_tipi: Optional[KikKararTipi] = Field(None, description="Decision type")
|
||||
|
||||
karar_id_param_from_url: Optional[str] = Field(None, alias="kararIdParam", description="Internal ID")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Content")
|
||||
source_url: Optional[str] = Field(None, description="Source URL")
|
||||
error_message: Optional[str] = Field(None, description="Error")
|
||||
current_page: int = Field(1, description="Page")
|
||||
total_pages: int = Field(1, description="Total pages")
|
||||
is_paginated: bool = Field(False, description="Paginated")
|
||||
full_content_char_count: Optional[int] = Field(None, description="Char count")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
@@ -0,0 +1,147 @@
|
||||
# kik_mcp_module/models_v2.py
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
|
||||
# New KIK v2 API Models
|
||||
|
||||
class KikV2DecisionType(str, Enum):
|
||||
"""KIK v2 Decision Types with corresponding endpoints."""
|
||||
UYUSMAZLIK = "uyusmazlik" # Disputes - GetKurulKararlari
|
||||
DUZENLEYICI = "duzenleyici" # Regulatory - GetKurulKararlariDk
|
||||
MAHKEME = "mahkeme" # Court - GetKurulKararlariMk
|
||||
|
||||
class KikV2SearchRequest(BaseModel):
|
||||
"""Model for KIK v2 API search request."""
|
||||
KararMetni: str = Field("", description="Decision text search query")
|
||||
KararNo: str = Field("", description="Decision number (e.g., '2025/UH.II-1801')")
|
||||
BasvuranAdi: str = Field("", description="Applicant name")
|
||||
IdareAdi: str = Field("", description="Administration name")
|
||||
BaslangicTarihi: str = Field("", description="Start date (YYYY-MM-DD)")
|
||||
BitisTarihi: str = Field("", description="End date (YYYY-MM-DD)")
|
||||
|
||||
class KikV2KeyValuePair(BaseModel):
|
||||
"""Key-value pair for KIK v2 API request."""
|
||||
key: str
|
||||
value: str
|
||||
|
||||
class KikV2QueryRequest(BaseModel):
|
||||
"""Nested query structure for KIK v2 API."""
|
||||
keyValueOfstringanyType: List[KikV2KeyValuePair]
|
||||
|
||||
class KikV2RequestData(BaseModel):
|
||||
"""Main request data structure for KIK v2 API."""
|
||||
keyValuePairs: KikV2QueryRequest
|
||||
|
||||
# Request Payloads for different decision types
|
||||
class KikV2SearchPayload(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Uyuşmazlık (Disputes)."""
|
||||
sorgulaKurulKararlari: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadDk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Düzenleyici (Regulatory)."""
|
||||
sorgulaKurulKararlariDk: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadMk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Mahkeme (Court)."""
|
||||
sorgulaKurulKararlariMk: KikV2RequestData
|
||||
|
||||
# Response Models
|
||||
|
||||
class KikV2DecisionDetail(BaseModel):
|
||||
"""Individual decision detail from KIK v2 API response."""
|
||||
resmiGazeteMukerrerSayi: str = Field("", description="Official Gazette duplicate number")
|
||||
itiraz: str = Field("", description="Objection")
|
||||
yayinlanmaTarihi: str = Field("", description="Publication date")
|
||||
idareAdi: str = Field("", description="Administration name")
|
||||
uzmanTCKN: str = Field("", description="Expert TCKN")
|
||||
resmiGazeteTarihi: str = Field("", description="Official Gazette date")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
kararTurKod: str = Field("", description="Decision type code")
|
||||
kararTurAciklama: str = Field("", description="Decision type description")
|
||||
karar: str = Field("", description="Decision text")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
resmiGazeteSayisi: str = Field("", description="Official Gazette number")
|
||||
inceleme: str = Field("", description="Review")
|
||||
basvuruTarihi: str = Field("", description="Application date")
|
||||
kararNitelikKod: str = Field("", description="Decision nature code")
|
||||
resmiGazeteMukerrer: str = Field("", description="Official Gazette duplicate")
|
||||
basvuruSayisi: str = Field("", description="Application number")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
kararNitelik: str = Field("", description="Decision nature")
|
||||
uyusmazlikKararNo: str = Field("", description="Dispute decision number")
|
||||
kurulNo: str = Field("", description="Board number")
|
||||
gundemMaddesiSiraNo: str = Field("", description="Agenda item sequence")
|
||||
kararTarihi: str = Field("", description="Decision date (ISO format)")
|
||||
dosyaBirimKodu: str = Field("", description="File unit code")
|
||||
gundemMaddesiId: str = Field("", description="Agenda item ID")
|
||||
|
||||
class KikV2DecisionGroup(BaseModel):
|
||||
"""Group of decision details."""
|
||||
KurulKararTutanakDetayi: List[KikV2DecisionDetail] = Field(alias="kurulKararTutanakDetayi")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultData(BaseModel):
|
||||
"""Search result data structure."""
|
||||
hataKodu: str = Field("", description="Error code")
|
||||
hataMesaji: str = Field("", description="Error message")
|
||||
KurulKararTutanakDetayListesi: List[KikV2DecisionGroup]
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultWrapper(BaseModel):
|
||||
"""Wrapper for search result."""
|
||||
SorgulaKurulKararlariResult: KikV2SearchResultData
|
||||
|
||||
# Base Response Models
|
||||
class KikV2SearchResponse(BaseModel):
|
||||
"""Complete KIK v2 API search response for Uyuşmazlık (Disputes)."""
|
||||
SorgulaKurulKararlariResponse: KikV2SearchResultWrapper
|
||||
|
||||
# Düzenleyici Kararlar (Regulatory Decisions) Response Models
|
||||
class KikV2SearchResultWrapperDk(BaseModel):
|
||||
"""Wrapper for regulatory decisions search result."""
|
||||
SorgulaKurulKararlariDkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseDk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Düzenleyici (Regulatory) decisions."""
|
||||
SorgulaKurulKararlariDkResponse: KikV2SearchResultWrapperDk
|
||||
|
||||
# Mahkeme Kararlar (Court Decisions) Response Models
|
||||
class KikV2SearchResultWrapperMk(BaseModel):
|
||||
"""Wrapper for court decisions search result."""
|
||||
SorgulaKurulKararlariMkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseMk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Mahkeme (Court) decisions."""
|
||||
SorgulaKurulKararlariMkResponse: KikV2SearchResultWrapperMk
|
||||
|
||||
# Simplified Models for MCP Tools
|
||||
|
||||
class KikV2CompactDecision(BaseModel):
|
||||
"""Compact decision format for MCP tool responses."""
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
kararTarihi: str = Field("", description="Decision date")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
idareAdi: str = Field("", description="Administration")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
gundemMaddesiId: str = Field("", description="Document ID for retrieval")
|
||||
decision_type: str = Field("", description="Decision type (uyusmazlik/duzenleyici/mahkeme)")
|
||||
|
||||
class KikV2SearchResult(BaseModel):
|
||||
"""Compact search results for MCP tools."""
|
||||
decisions: List[KikV2CompactDecision]
|
||||
total_records: int = Field(0, description="Total number of decisions found")
|
||||
page: int = Field(1, description="Current page number")
|
||||
error_code: str = Field("", description="API error code")
|
||||
error_message: str = Field("", description="API error message")
|
||||
|
||||
class KikV2DocumentMarkdown(BaseModel):
|
||||
"""Document content in Markdown format."""
|
||||
document_id: str = Field("", description="Document ID")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
markdown_content: str = Field("", description="Decision content in Markdown")
|
||||
source_url: str = Field("", description="Source URL")
|
||||
error_message: str = Field("", description="Error message if retrieval failed")
|
||||
+630
-1034
File diff suppressed because it is too large
Load Diff
+7
-5
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "yargi-mcp"
|
||||
version = "0.1.5"
|
||||
version = "0.2.0"
|
||||
description = "MCP Server For Turkish Legal Databases"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
@@ -25,12 +25,12 @@ dependencies = [
|
||||
"markitdown[pdf]>=0.1.1",
|
||||
"pydantic>=2.11.4",
|
||||
"aiohttp>=3.11.18",
|
||||
"playwright>=1.52.0",
|
||||
"fastmcp>=2.10.5",
|
||||
"pypdf>=5.5.0",
|
||||
"fastapi>=0.115.14",
|
||||
"PyJWT>=2.8.0",
|
||||
"tiktoken>=0.5.0",
|
||||
"cryptography>=44.0.0",
|
||||
"openai>=1.0.0",
|
||||
"numpy>=1.24.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
@@ -50,6 +50,8 @@ saas = [
|
||||
"clerk-backend-api>=3.0.0",
|
||||
"stripe>=9.1.0",
|
||||
"upstash-redis>=1.1.0",
|
||||
"tiktoken>=0.5.0",
|
||||
"PyJWT>=2.8.0",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
@@ -59,7 +61,7 @@ yargi-mcp = "mcp_server_main:main"
|
||||
py-modules = ["mcp_server_main", "mcp_auth_factory", "mcp_auth_http_adapter", "asgi_app", "fastapi_app", "starlette_app", "run_asgi", "stripe_webhook"]
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["*_mcp_module", "mcp_auth"]
|
||||
include = ["*_mcp_module", "mcp_auth", "semantic_search"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=65.0", "wheel"]
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
fastmcp
|
||||
httpx
|
||||
beautifulsoup4
|
||||
markitdown[pdf]
|
||||
pydantic
|
||||
aiohttp
|
||||
playwright
|
||||
pypdf
|
||||
fastapi>=0.115.14
|
||||
uvicorn[standard]>=0.30.0
|
||||
starlette>=0.37.0
|
||||
@@ -0,0 +1,7 @@
|
||||
# semantic_search/__init__.py
|
||||
|
||||
from .embedder import OpenRouterEmbedder, is_openrouter_available
|
||||
from .vector_store import VectorStore
|
||||
from .processor import DocumentProcessor
|
||||
|
||||
__all__ = ['OpenRouterEmbedder', 'is_openrouter_available', 'VectorStore', 'DocumentProcessor']
|
||||
@@ -0,0 +1,154 @@
|
||||
# semantic_search/embedder.py
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import List, Optional
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def is_openrouter_available() -> bool:
|
||||
"""Check if OpenRouter API key is available."""
|
||||
return bool(os.getenv("OPENROUTER_API_KEY"))
|
||||
|
||||
|
||||
class OpenRouterEmbedder:
|
||||
"""
|
||||
Embedder using OpenRouter API with Google's Gemini Embedding model.
|
||||
Requires OPENROUTER_API_KEY environment variable.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
"""
|
||||
Initialize OpenRouter Embedder.
|
||||
|
||||
Raises:
|
||||
ValueError: If OPENROUTER_API_KEY is not set
|
||||
ImportError: If openai package is not installed
|
||||
"""
|
||||
api_key = os.getenv("OPENROUTER_API_KEY")
|
||||
if not api_key:
|
||||
raise ValueError("OPENROUTER_API_KEY environment variable is not set")
|
||||
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError:
|
||||
raise ImportError("openai package is required. Install with: pip install openai")
|
||||
|
||||
self.client = OpenAI(
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
api_key=api_key,
|
||||
)
|
||||
self.model = "google/gemini-embedding-001"
|
||||
self.dimension = 3072
|
||||
|
||||
logger.info(f"OpenRouter Embedder initialized with model: {self.model}")
|
||||
|
||||
def encode_query(self, query: str, task: str = "search result") -> np.ndarray:
|
||||
"""
|
||||
Encode a search query.
|
||||
|
||||
Args:
|
||||
query: The search query text
|
||||
task: Task type for prompt template
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (3072 dimensions)
|
||||
"""
|
||||
# Apply query prompt template
|
||||
text = f"task: {task} | query: {query}"
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=text,
|
||||
encoding_format="float",
|
||||
extra_headers={
|
||||
"HTTP-Referer": "https://yargimcp.com",
|
||||
"X-Title": "Yargi MCP Server",
|
||||
}
|
||||
)
|
||||
|
||||
embedding = np.array(response.data[0].embedding, dtype=np.float32)
|
||||
|
||||
# L2 normalize for cosine similarity
|
||||
norm = np.linalg.norm(embedding)
|
||||
if norm > 0:
|
||||
embedding = embedding / norm
|
||||
|
||||
logger.debug(f"Encoded query: {query[:50]}... -> shape: {embedding.shape}")
|
||||
return embedding
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode query: {e}")
|
||||
raise
|
||||
|
||||
def encode_documents(self, documents: List[str], titles: Optional[List[str]] = None) -> np.ndarray:
|
||||
"""
|
||||
Encode multiple documents with batch API call.
|
||||
|
||||
Args:
|
||||
documents: List of document texts
|
||||
titles: Optional list of document titles
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (N x 3072 dimensions)
|
||||
"""
|
||||
if not documents:
|
||||
return np.array([])
|
||||
|
||||
# Apply document prompt template
|
||||
texts = []
|
||||
for i, doc in enumerate(documents):
|
||||
title = titles[i] if titles and i < len(titles) else "none"
|
||||
text = f"title: {title} | text: {doc}"
|
||||
texts.append(text)
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=texts,
|
||||
encoding_format="float",
|
||||
extra_headers={
|
||||
"HTTP-Referer": "https://yargimcp.com",
|
||||
"X-Title": "Yargi MCP Server",
|
||||
}
|
||||
)
|
||||
|
||||
# Extract embeddings in order
|
||||
embeddings = np.array(
|
||||
[d.embedding for d in sorted(response.data, key=lambda x: x.index)],
|
||||
dtype=np.float32
|
||||
)
|
||||
|
||||
# L2 normalize each embedding for cosine similarity
|
||||
norms = np.linalg.norm(embeddings, axis=1, keepdims=True)
|
||||
embeddings = embeddings / (norms + 1e-8)
|
||||
|
||||
logger.info(f"Encoded {len(documents)} documents -> shape: {embeddings.shape}")
|
||||
return embeddings
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode documents: {e}")
|
||||
raise
|
||||
|
||||
def compute_similarity(self, query_embedding: np.ndarray, document_embeddings: np.ndarray) -> np.ndarray:
|
||||
"""
|
||||
Compute cosine similarity between query and documents.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding (3072,)
|
||||
document_embeddings: Document embeddings (N x 3072)
|
||||
|
||||
Returns:
|
||||
Similarity scores (N,)
|
||||
"""
|
||||
# Ensure query is 2D for matrix multiplication
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarity (embeddings are already normalized)
|
||||
similarities = np.dot(document_embeddings, query_embedding.T).squeeze()
|
||||
|
||||
return similarities
|
||||
@@ -0,0 +1,305 @@
|
||||
# semantic_search/processor.py
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List, Dict, Any, Optional
|
||||
from dataclasses import dataclass
|
||||
import hashlib
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class DocumentChunk:
|
||||
"""Represents a chunk of a document."""
|
||||
chunk_id: str
|
||||
document_id: str
|
||||
text: str
|
||||
metadata: Dict[str, Any]
|
||||
chunk_index: int
|
||||
total_chunks: int
|
||||
|
||||
class DocumentProcessor:
|
||||
"""
|
||||
Processes legal documents for semantic search.
|
||||
Handles chunking, cleaning, and metadata extraction.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
chunk_size: int = 1000,
|
||||
chunk_overlap: int = 200,
|
||||
min_chunk_size: int = 100):
|
||||
"""
|
||||
Initialize document processor.
|
||||
|
||||
Args:
|
||||
chunk_size: Target size for each chunk in characters
|
||||
chunk_overlap: Number of overlapping characters between chunks
|
||||
min_chunk_size: Minimum chunk size to keep
|
||||
"""
|
||||
self.chunk_size = chunk_size
|
||||
self.chunk_overlap = chunk_overlap
|
||||
self.min_chunk_size = min_chunk_size
|
||||
|
||||
logger.info(f"Initialized DocumentProcessor (chunk_size={chunk_size}, overlap={chunk_overlap})")
|
||||
|
||||
def process_document(self,
|
||||
document_id: str,
|
||||
text: str,
|
||||
metadata: Optional[Dict[str, Any]] = None) -> List[DocumentChunk]:
|
||||
"""
|
||||
Process a single document into chunks.
|
||||
|
||||
Args:
|
||||
document_id: Unique document identifier
|
||||
text: Document text content
|
||||
metadata: Optional document metadata
|
||||
|
||||
Returns:
|
||||
List of document chunks
|
||||
"""
|
||||
if not text or len(text.strip()) < self.min_chunk_size:
|
||||
logger.warning(f"Document {document_id} too short to process")
|
||||
return []
|
||||
|
||||
# Clean text
|
||||
cleaned_text = self._clean_text(text)
|
||||
|
||||
# Extract metadata from text if not provided
|
||||
if metadata is None:
|
||||
metadata = {}
|
||||
|
||||
# Add extracted metadata
|
||||
extracted_metadata = self._extract_metadata(cleaned_text)
|
||||
metadata.update(extracted_metadata)
|
||||
|
||||
# Create chunks
|
||||
chunks = self._create_chunks(cleaned_text)
|
||||
|
||||
# Create DocumentChunk objects
|
||||
document_chunks = []
|
||||
for i, chunk_text in enumerate(chunks):
|
||||
chunk_id = self._generate_chunk_id(document_id, i)
|
||||
|
||||
chunk = DocumentChunk(
|
||||
chunk_id=chunk_id,
|
||||
document_id=document_id,
|
||||
text=chunk_text,
|
||||
metadata={
|
||||
**metadata,
|
||||
'chunk_index': i,
|
||||
'total_chunks': len(chunks)
|
||||
},
|
||||
chunk_index=i,
|
||||
total_chunks=len(chunks)
|
||||
)
|
||||
document_chunks.append(chunk)
|
||||
|
||||
logger.info(f"Processed document {document_id} into {len(chunks)} chunks")
|
||||
return document_chunks
|
||||
|
||||
def _clean_text(self, text: str) -> str:
|
||||
"""
|
||||
Clean and normalize text for processing.
|
||||
|
||||
Args:
|
||||
text: Raw text
|
||||
|
||||
Returns:
|
||||
Cleaned text
|
||||
"""
|
||||
# Remove excessive whitespace
|
||||
text = re.sub(r'\s+', ' ', text)
|
||||
|
||||
# Remove special characters but keep Turkish characters
|
||||
# Keep: letters, numbers, spaces, and common punctuation
|
||||
text = re.sub(r'[^\w\s\.\,\;\:\!\?\-\(\)\"\'ÇĞIİÖŞÜçğıiöşü]', ' ', text)
|
||||
|
||||
# Remove multiple spaces
|
||||
text = re.sub(r' +', ' ', text)
|
||||
|
||||
# Trim
|
||||
text = text.strip()
|
||||
|
||||
return text
|
||||
|
||||
def _extract_metadata(self, text: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Extract metadata from legal document text.
|
||||
|
||||
Args:
|
||||
text: Document text
|
||||
|
||||
Returns:
|
||||
Extracted metadata
|
||||
"""
|
||||
metadata = {}
|
||||
|
||||
# Extract case numbers (Esas/Karar)
|
||||
esas_pattern = r'E(?:sas)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
karar_pattern = r'K(?:arar)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
|
||||
esas_match = re.search(esas_pattern, text[:500]) # Look in first 500 chars
|
||||
if esas_match:
|
||||
metadata['esas_no'] = f"E.{esas_match.group(1)}/{esas_match.group(2)}"
|
||||
|
||||
karar_match = re.search(karar_pattern, text[:500])
|
||||
if karar_match:
|
||||
metadata['karar_no'] = f"K.{karar_match.group(1)}/{karar_match.group(2)}"
|
||||
|
||||
# Extract dates (DD.MM.YYYY or DD/MM/YYYY format)
|
||||
date_pattern = r'(\d{1,2})[\.\/](\d{1,2})[\.\/](\d{4})'
|
||||
dates = re.findall(date_pattern, text[:1000]) # Look in first 1000 chars
|
||||
if dates:
|
||||
# Take the first date as decision date
|
||||
day, month, year = dates[0]
|
||||
metadata['karar_tarihi'] = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
|
||||
# Extract court/chamber name
|
||||
chamber_patterns = [
|
||||
r'(\d+)\.\s*Hukuk\s+Dairesi',
|
||||
r'(\d+)\.\s*Ceza\s+Dairesi',
|
||||
r'Hukuk\s+Genel\s+Kurulu',
|
||||
r'Ceza\s+Genel\s+Kurulu',
|
||||
r'(\d+)\.\s*Daire'
|
||||
]
|
||||
|
||||
for pattern in chamber_patterns:
|
||||
match = re.search(pattern, text[:500], re.IGNORECASE)
|
||||
if match:
|
||||
metadata['chamber'] = match.group(0)
|
||||
break
|
||||
|
||||
return metadata
|
||||
|
||||
def _create_chunks(self, text: str) -> List[str]:
|
||||
"""
|
||||
Create overlapping chunks from text.
|
||||
|
||||
Args:
|
||||
text: Cleaned document text
|
||||
|
||||
Returns:
|
||||
List of text chunks
|
||||
"""
|
||||
chunks = []
|
||||
|
||||
# Split by sentences for better semantic coherence
|
||||
sentences = self._split_sentences(text)
|
||||
|
||||
current_chunk = []
|
||||
current_size = 0
|
||||
|
||||
for sentence in sentences:
|
||||
sentence_size = len(sentence)
|
||||
|
||||
# If adding this sentence exceeds chunk size
|
||||
if current_size + sentence_size > self.chunk_size and current_chunk:
|
||||
# Save current chunk
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
chunks.append(chunk_text)
|
||||
|
||||
# Create overlap for next chunk
|
||||
overlap_size = 0
|
||||
overlap_sentences = []
|
||||
|
||||
# Add sentences from the end until we reach overlap size
|
||||
for sent in reversed(current_chunk):
|
||||
overlap_size += len(sent)
|
||||
overlap_sentences.insert(0, sent)
|
||||
if overlap_size >= self.chunk_overlap:
|
||||
break
|
||||
|
||||
# Start new chunk with overlap
|
||||
current_chunk = overlap_sentences
|
||||
current_size = sum(len(s) for s in current_chunk)
|
||||
|
||||
# Add sentence to current chunk
|
||||
current_chunk.append(sentence)
|
||||
current_size += sentence_size
|
||||
|
||||
# Add final chunk if not empty
|
||||
if current_chunk:
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
if len(chunk_text) >= self.min_chunk_size:
|
||||
chunks.append(chunk_text)
|
||||
|
||||
return chunks
|
||||
|
||||
def _split_sentences(self, text: str) -> List[str]:
|
||||
"""
|
||||
Split text into sentences.
|
||||
|
||||
Args:
|
||||
text: Text to split
|
||||
|
||||
Returns:
|
||||
List of sentences
|
||||
"""
|
||||
# Simple sentence splitting for Turkish text
|
||||
# Split on period, question mark, exclamation, but not on abbreviations
|
||||
|
||||
# Common Turkish abbreviations to preserve
|
||||
abbreviations = ['Dr', 'Prof', 'Av', 'Md', 'Yrd', 'Doç', 'No', 'S', 'vs', 'vb', 'bkz']
|
||||
|
||||
# Replace abbreviations temporarily
|
||||
temp_text = text
|
||||
replacements = {}
|
||||
for i, abbr in enumerate(abbreviations):
|
||||
placeholder = f"__ABBR{i}__"
|
||||
temp_text = temp_text.replace(f"{abbr}.", placeholder)
|
||||
replacements[placeholder] = f"{abbr}."
|
||||
|
||||
# Split sentences
|
||||
sentence_endings = re.compile(r'[.!?]+')
|
||||
sentences = sentence_endings.split(temp_text)
|
||||
|
||||
# Restore abbreviations and clean
|
||||
cleaned_sentences = []
|
||||
for sentence in sentences:
|
||||
# Restore abbreviations
|
||||
for placeholder, original in replacements.items():
|
||||
sentence = sentence.replace(placeholder, original)
|
||||
|
||||
# Clean and add if not empty
|
||||
sentence = sentence.strip()
|
||||
if sentence and len(sentence) > 10: # Minimum sentence length
|
||||
cleaned_sentences.append(sentence)
|
||||
|
||||
return cleaned_sentences
|
||||
|
||||
def _generate_chunk_id(self, document_id: str, chunk_index: int) -> str:
|
||||
"""
|
||||
Generate unique chunk ID.
|
||||
|
||||
Args:
|
||||
document_id: Parent document ID
|
||||
chunk_index: Index of chunk in document
|
||||
|
||||
Returns:
|
||||
Unique chunk ID
|
||||
"""
|
||||
chunk_string = f"{document_id}_chunk_{chunk_index}"
|
||||
chunk_hash = hashlib.md5(chunk_string.encode()).hexdigest()[:8]
|
||||
return f"{document_id}_c{chunk_index}_{chunk_hash}"
|
||||
|
||||
def combine_chunks(self, chunks: List[DocumentChunk]) -> str:
|
||||
"""
|
||||
Combine chunks back into full document text.
|
||||
|
||||
Args:
|
||||
chunks: List of document chunks
|
||||
|
||||
Returns:
|
||||
Combined text
|
||||
"""
|
||||
if not chunks:
|
||||
return ""
|
||||
|
||||
# Sort by chunk index
|
||||
sorted_chunks = sorted(chunks, key=lambda x: x.chunk_index)
|
||||
|
||||
# For overlapping chunks, we need to be careful about duplication
|
||||
# Simple approach: just concatenate with space
|
||||
combined = " ".join([chunk.text for chunk in sorted_chunks])
|
||||
|
||||
return combined
|
||||
@@ -0,0 +1,235 @@
|
||||
# semantic_search/vector_store.py
|
||||
|
||||
import logging
|
||||
import numpy as np
|
||||
from typing import List, Dict, Any, Tuple, Optional
|
||||
from dataclasses import dataclass
|
||||
import json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class Document:
|
||||
"""Represents a document with its embedding and metadata."""
|
||||
id: str
|
||||
text: str
|
||||
embedding: np.ndarray
|
||||
metadata: Dict[str, Any]
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Convert to dictionary (excluding embedding for serialization)."""
|
||||
return {
|
||||
'id': self.id,
|
||||
'text': self.text,
|
||||
'metadata': self.metadata
|
||||
}
|
||||
|
||||
class VectorStore:
|
||||
"""
|
||||
In-memory vector storage with similarity search capabilities.
|
||||
Future versions can use Faiss, ChromaDB, or other vector databases.
|
||||
"""
|
||||
|
||||
def __init__(self, dimension: int = 768):
|
||||
"""
|
||||
Initialize vector store.
|
||||
|
||||
Args:
|
||||
dimension: Embedding dimension size
|
||||
"""
|
||||
self.dimension = dimension
|
||||
self.documents: List[Document] = []
|
||||
self.embeddings: Optional[np.ndarray] = None
|
||||
self.index_built = False
|
||||
|
||||
logger.info(f"Initialized VectorStore with dimension: {dimension}")
|
||||
|
||||
def add_documents(self,
|
||||
ids: List[str],
|
||||
texts: List[str],
|
||||
embeddings: np.ndarray,
|
||||
metadata: Optional[List[Dict[str, Any]]] = None) -> int:
|
||||
"""
|
||||
Add documents to the vector store.
|
||||
|
||||
Args:
|
||||
ids: Document IDs
|
||||
texts: Document texts
|
||||
embeddings: Document embeddings (N x dimension)
|
||||
metadata: Optional metadata for each document
|
||||
|
||||
Returns:
|
||||
Number of documents added
|
||||
"""
|
||||
if len(ids) != len(texts) or len(ids) != embeddings.shape[0]:
|
||||
raise ValueError("Mismatched lengths for ids, texts, and embeddings")
|
||||
|
||||
if metadata and len(metadata) != len(ids):
|
||||
raise ValueError("Metadata length doesn't match document count")
|
||||
|
||||
# Add documents
|
||||
for i in range(len(ids)):
|
||||
doc = Document(
|
||||
id=ids[i],
|
||||
text=texts[i],
|
||||
embedding=embeddings[i],
|
||||
metadata=metadata[i] if metadata else {}
|
||||
)
|
||||
self.documents.append(doc)
|
||||
|
||||
# Rebuild index
|
||||
self._build_index()
|
||||
|
||||
logger.info(f"Added {len(ids)} documents to vector store. Total: {len(self.documents)}")
|
||||
return len(ids)
|
||||
|
||||
def _build_index(self):
|
||||
"""Build or rebuild the embedding index."""
|
||||
if not self.documents:
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
return
|
||||
|
||||
# Stack all embeddings into a single array
|
||||
self.embeddings = np.vstack([doc.embedding for doc in self.documents])
|
||||
self.index_built = True
|
||||
|
||||
logger.debug(f"Built index with shape: {self.embeddings.shape}")
|
||||
|
||||
def search(self,
|
||||
query_embedding: np.ndarray,
|
||||
top_k: int = 10,
|
||||
threshold: Optional[float] = None) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Search for similar documents using cosine similarity.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
top_k: Number of results to return
|
||||
threshold: Optional similarity threshold (0-1)
|
||||
|
||||
Returns:
|
||||
List of (Document, similarity_score) tuples
|
||||
"""
|
||||
if not self.index_built or self.embeddings is None:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Ensure query is 2D
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarities (assuming normalized embeddings)
|
||||
similarities = np.dot(self.embeddings, query_embedding.T).squeeze()
|
||||
|
||||
# Apply threshold if specified
|
||||
if threshold is not None:
|
||||
valid_indices = np.where(similarities >= threshold)[0]
|
||||
if len(valid_indices) == 0:
|
||||
logger.info(f"No documents above threshold {threshold}")
|
||||
return []
|
||||
similarities = similarities[valid_indices]
|
||||
valid_docs = [self.documents[i] for i in valid_indices]
|
||||
else:
|
||||
valid_docs = self.documents
|
||||
|
||||
# Get top-k indices
|
||||
top_k = min(top_k, len(valid_docs))
|
||||
if top_k == 0:
|
||||
return []
|
||||
|
||||
# Use argpartition for efficiency with large arrays
|
||||
if len(similarities) > top_k:
|
||||
top_indices = np.argpartition(similarities, -top_k)[-top_k:]
|
||||
top_indices = top_indices[np.argsort(similarities[top_indices])[::-1]]
|
||||
else:
|
||||
top_indices = np.argsort(similarities)[::-1]
|
||||
|
||||
# Create results
|
||||
results = []
|
||||
for idx in top_indices:
|
||||
doc = valid_docs[idx] if threshold else self.documents[idx]
|
||||
score = float(similarities[idx])
|
||||
results.append((doc, score))
|
||||
|
||||
logger.info(f"Search returned {len(results)} results (top_k={top_k})")
|
||||
return results
|
||||
|
||||
def hybrid_search(self,
|
||||
query_embedding: np.ndarray,
|
||||
keyword_scores: Dict[str, float],
|
||||
top_k: int = 10,
|
||||
alpha: float = 0.5) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Hybrid search combining vector similarity and keyword scores.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
keyword_scores: Document ID to keyword relevance score mapping
|
||||
top_k: Number of results to return
|
||||
alpha: Weight for vector similarity (1-alpha for keyword score)
|
||||
|
||||
Returns:
|
||||
List of (Document, combined_score) tuples
|
||||
"""
|
||||
if not self.index_built:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Get vector similarities
|
||||
vector_results = self.search(query_embedding, top_k=len(self.documents))
|
||||
|
||||
# Combine scores
|
||||
combined_scores = []
|
||||
for doc, vector_score in vector_results:
|
||||
keyword_score = keyword_scores.get(doc.id, 0.0)
|
||||
# Normalize keyword score to 0-1 range if needed
|
||||
if keyword_score > 1.0:
|
||||
keyword_score = keyword_score / max(keyword_scores.values())
|
||||
|
||||
combined_score = alpha * vector_score + (1 - alpha) * keyword_score
|
||||
combined_scores.append((doc, combined_score))
|
||||
|
||||
# Sort by combined score and return top-k
|
||||
combined_scores.sort(key=lambda x: x[1], reverse=True)
|
||||
results = combined_scores[:top_k]
|
||||
|
||||
logger.info(f"Hybrid search returned {len(results)} results")
|
||||
return results
|
||||
|
||||
def clear(self):
|
||||
"""Clear all documents from the store."""
|
||||
self.documents = []
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
logger.info("Cleared vector store")
|
||||
|
||||
def size(self) -> int:
|
||||
"""Get number of documents in store."""
|
||||
return len(self.documents)
|
||||
|
||||
def get_by_id(self, doc_id: str) -> Optional[Document]:
|
||||
"""Get document by ID."""
|
||||
for doc in self.documents:
|
||||
if doc.id == doc_id:
|
||||
return doc
|
||||
return None
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""Get statistics about the vector store."""
|
||||
stats = {
|
||||
'num_documents': len(self.documents),
|
||||
'dimension': self.dimension,
|
||||
'index_built': self.index_built,
|
||||
'memory_usage_mb': 0
|
||||
}
|
||||
|
||||
if self.embeddings is not None:
|
||||
# Estimate memory usage
|
||||
memory_bytes = self.embeddings.nbytes
|
||||
for doc in self.documents:
|
||||
memory_bytes += len(doc.text.encode('utf-8'))
|
||||
memory_bytes += len(json.dumps(doc.metadata).encode('utf-8'))
|
||||
stats['memory_usage_mb'] = memory_bytes / (1024 * 1024)
|
||||
|
||||
return stats
|
||||
@@ -1,7 +1,6 @@
|
||||
# uyusmazlik_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
import aiohttp
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Union, Tuple
|
||||
import logging
|
||||
@@ -9,7 +8,7 @@ import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
from urllib.parse import urljoin, urlencode # urlencode for aiohttp form data
|
||||
from urllib.parse import urljoin
|
||||
|
||||
from .models import (
|
||||
UyusmazlikSearchRequest,
|
||||
@@ -56,17 +55,21 @@ class UyusmazlikApiClient:
|
||||
# Individual documents are fetched by their full URLs obtained from search results.
|
||||
|
||||
def __init__(self, request_timeout: float = 30.0):
|
||||
self.request_timeout = request_timeout # Store timeout for aiohttp and httpx
|
||||
# Headers for aiohttp search. httpx for docs will create its own.
|
||||
self.default_aiohttp_search_headers = {
|
||||
"Accept": "*/*", # Mimicking browser headers provided by user
|
||||
"Accept-Encoding": "gzip, deflate, br, zstd",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"X-Requested-With": "XMLHttpRequest",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": self.BASE_URL + "/",
|
||||
|
||||
}
|
||||
self.request_timeout = request_timeout
|
||||
# Create shared httpx client for all requests
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "*/*",
|
||||
"Accept-Encoding": "gzip, deflate, br, zstd",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"X-Requested-With": "XMLHttpRequest",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": self.BASE_URL + "/",
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=False
|
||||
)
|
||||
|
||||
|
||||
async def search_decisions(
|
||||
@@ -107,32 +110,36 @@ class UyusmazlikApiClient:
|
||||
add_to_form_data("Hepsi", params.hepsi)
|
||||
add_to_form_data("Herhangibirisi", params.herhangi_birisi)
|
||||
add_to_form_data("NotHepsi", params.not_hepsi)
|
||||
# X-Requested-With is handled by default_aiohttp_search_headers
|
||||
|
||||
search_url = urljoin(self.BASE_URL, self.SEARCH_ENDPOINT)
|
||||
# For aiohttp, data for application/x-www-form-urlencoded should be a dict or str.
|
||||
# Using urlencode for list of tuples.
|
||||
encoded_form_payload = urlencode(form_data_list, encoding='UTF-8')
|
||||
# Convert form data to dict for httpx
|
||||
form_data_dict = {}
|
||||
for key, value in form_data_list:
|
||||
if key in form_data_dict:
|
||||
# Handle multiple values (like KararSonucuList)
|
||||
if not isinstance(form_data_dict[key], list):
|
||||
form_data_dict[key] = [form_data_dict[key]]
|
||||
form_data_dict[key].append(value)
|
||||
else:
|
||||
form_data_dict[key] = value
|
||||
|
||||
logger.info(f"UyusmazlikApiClient (aiohttp): Performing search to {search_url} with form_data: {encoded_form_payload}")
|
||||
logger.info(f"UyusmazlikApiClient (httpx): Performing search to {self.SEARCH_ENDPOINT} with form_data: {form_data_dict}")
|
||||
|
||||
html_content = ""
|
||||
aiohttp_headers = self.default_aiohttp_search_headers.copy()
|
||||
aiohttp_headers["Content-Type"] = "application/x-www-form-urlencoded; charset=UTF-8"
|
||||
|
||||
try:
|
||||
# Create a new session for each call for simplicity with aiohttp here
|
||||
async with aiohttp.ClientSession(headers=aiohttp_headers) as session:
|
||||
async with session.post(search_url, data=encoded_form_payload, timeout=self.request_timeout) as response:
|
||||
response.raise_for_status() # Raises ClientResponseError for 400-599
|
||||
html_content = await response.text(encoding='utf-8') # Ensure correct encoding
|
||||
logger.debug("UyusmazlikApiClient (aiohttp): Received HTML response for search.")
|
||||
# Use shared httpx client
|
||||
response = await self.http_client.post(
|
||||
self.SEARCH_ENDPOINT,
|
||||
data=form_data_dict,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded; charset=UTF-8"}
|
||||
)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
logger.debug("UyusmazlikApiClient (httpx): Received HTML response for search.")
|
||||
|
||||
except aiohttp.ClientError as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): HTTP client error during search: {e}")
|
||||
except httpx.HTTPError as e:
|
||||
logger.error(f"UyusmazlikApiClient (httpx): HTTP client error during search: {e}")
|
||||
raise # Re-raise to be handled by the MCP tool
|
||||
except Exception as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): Error processing search request: {e}")
|
||||
logger.error(f"UyusmazlikApiClient (httpx): Error processing search request: {e}")
|
||||
raise
|
||||
|
||||
# --- HTML Parsing (remains the same as previous version) ---
|
||||
@@ -217,7 +224,6 @@ class UyusmazlikApiClient:
|
||||
try:
|
||||
# Using a new httpx.AsyncClient instance for this GET request for simplicity
|
||||
async with httpx.AsyncClient(verify=False, timeout=self.request_timeout) as doc_fetch_client:
|
||||
|
||||
get_response = await doc_fetch_client.get(document_url, headers={"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"})
|
||||
get_response.raise_for_status()
|
||||
html_content_from_api = get_response.text
|
||||
@@ -236,5 +242,9 @@ class UyusmazlikApiClient:
|
||||
raise
|
||||
|
||||
async def close_client_session(self):
|
||||
|
||||
logger.info("UyusmazlikApiClient: No persistent client session from __init__ to close.")
|
||||
"""Close the shared httpx client session."""
|
||||
if hasattr(self, 'http_client') and self.http_client:
|
||||
await self.http_client.aclose()
|
||||
logger.info("UyusmazlikApiClient: HTTP client session closed.")
|
||||
else:
|
||||
logger.info("UyusmazlikApiClient: No persistent client session from __init__ to close.")
|
||||
Reference in New Issue
Block a user