Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6eb86d0a9a | ||
|
|
0e51ca432a | ||
|
|
08a19fb83c | ||
|
|
15402b4423 | ||
|
|
1b483a6fcf | ||
|
|
cc055103fe | ||
|
|
2c3347643d | ||
|
|
fadc3b0bc0 | ||
|
|
6bbc656dc6 | ||
|
|
a062237474 | ||
|
|
b69eda77af | ||
|
|
3927dcee8f | ||
|
|
3768104679 | ||
|
|
aa580ffafc | ||
|
|
931eb3ca8f | ||
|
|
d258ad2375 | ||
|
|
061887f870 | ||
|
|
ac611f840c | ||
|
|
c938f10ba2 | ||
|
|
1356c4d020 | ||
|
|
5392435c7a | ||
|
|
96a5a538b2 | ||
|
|
26aa3dacc6 | ||
|
|
58457b076f | ||
|
|
4521e1de85 | ||
|
|
8f04010c57 | ||
|
|
7ed9c25687 | ||
|
|
4fdc7a3689 | ||
|
|
1538a4c145 | ||
|
|
a24def2e66 | ||
|
|
6b781b61d2 | ||
|
|
fb29146755 | ||
|
|
42731a2c03 | ||
|
|
ae5d590cca | ||
|
|
ee544dc603 | ||
|
|
355f505da9 | ||
|
|
4c06a5926b | ||
|
|
2cec4dccd6 | ||
|
|
5c2e9cc92b | ||
|
|
a66a3f2053 | ||
|
|
a4d9e2e53d | ||
|
|
036e49a928 | ||
|
|
7f78f87508 | ||
|
|
d8805cb93b | ||
|
|
28ff2e39a5 | ||
|
|
fd08637ca2 | ||
|
|
a5e6baeec8 | ||
|
|
8818a7809a | ||
|
|
12d51e3735 | ||
|
|
efe962abf1 | ||
|
|
1d73265f10 | ||
|
|
f1d3b60efb | ||
|
|
93e64bc1fc | ||
|
|
77e2748ade | ||
|
|
da146cf3ec | ||
|
|
e771c5b3c5 | ||
|
|
1223b37adb | ||
|
|
e26f09aced | ||
|
|
ae5bae2f4a | ||
|
|
a2b50951e9 | ||
|
|
91ad04cf09 | ||
|
|
5cec0df785 | ||
|
|
d51f11c7ba | ||
|
|
b207b16ef7 | ||
|
|
f47147ba44 | ||
|
|
4d57a3939f | ||
|
|
50c6963eee | ||
|
|
def7e7d65e | ||
|
|
82a0d13d25 | ||
|
|
18b552ca2f | ||
|
|
1e96b1888e | ||
|
|
815786a09d | ||
|
|
260adb3ac9 | ||
|
|
7164205425 | ||
|
|
3961a23d3a | ||
|
|
25723f070f | ||
|
|
6376037ccf | ||
|
|
d1728ce114 | ||
|
|
69b5da5cef | ||
|
|
91564bf0a1 | ||
|
|
6f94eca33c | ||
|
|
4122790821 | ||
|
|
0f5bae8bb1 | ||
|
|
4d7da0d3ba | ||
|
|
4f48681b09 | ||
|
|
b401bad890 | ||
|
|
0a80bc535b | ||
|
|
b1da034ea9 | ||
|
|
e900bc03dd | ||
|
|
54f81e18f0 | ||
|
|
4e18e792c5 | ||
|
|
4a3edef287 | ||
|
|
e2ca844ab9 | ||
|
|
a49d0859ea | ||
|
|
a7877f34f4 | ||
|
|
364f3761d7 | ||
|
|
673f996f5f | ||
|
|
90a7a23064 | ||
|
|
443657f9e2 | ||
|
|
f5fa0076f8 | ||
|
|
7a346ef3f6 | ||
|
|
217103f0b6 | ||
|
|
1fbcb65031 | ||
|
|
9e40671798 | ||
|
|
6c8a614872 | ||
|
|
861d9e86ef | ||
|
|
c4b5d3608a | ||
|
|
38e0cc032b | ||
|
|
2c1b8c6f9d | ||
|
|
92f04fbab6 | ||
|
|
515347e29c | ||
|
|
ebefe22a4c | ||
|
|
c93244ee10 | ||
|
|
d84f8a2c88 | ||
|
|
ec40b9d6a2 | ||
|
|
daa16cae99 | ||
|
|
c3bc9e17eb | ||
|
|
c9344fc538 | ||
|
|
d12ad7d900 | ||
|
|
891751043c | ||
|
|
ba447502fe | ||
|
|
c092a7af45 | ||
|
|
34216a9557 | ||
|
|
753283f0e8 | ||
|
|
611456fd49 | ||
|
|
7a1ff0b9ed | ||
|
|
95620285d9 | ||
|
|
8a148899e3 | ||
|
|
fa6c448afa | ||
|
|
cc363e7a6d | ||
|
|
f96c1a2e44 | ||
|
|
856ecdf13d | ||
|
|
f7dac9363a | ||
|
|
b32aabf541 | ||
|
|
945ffe6267 | ||
|
|
17f9b109a2 | ||
|
|
2a24ce03ad | ||
|
|
e34d81be26 | ||
|
|
54e8d61f83 | ||
|
|
35382136c5 | ||
|
|
59d81d4b03 | ||
|
|
06a319c0ee | ||
|
|
a48b003121 | ||
|
|
1620d7a9b0 | ||
|
|
c10069bcc7 | ||
|
|
c0fe7e1305 | ||
|
|
f492109eb7 | ||
|
|
35f3b739d4 | ||
|
|
8f1ca6b854 | ||
|
|
71218996fd | ||
|
|
6bac51dc19 | ||
|
|
b898cad4f4 | ||
|
|
3d172c508a | ||
|
|
48efe74dc0 | ||
|
|
8ae5772c2c | ||
|
|
87cf4fd46d | ||
|
|
d590702272 | ||
|
|
1ebe8847fb | ||
|
|
cb318faeba | ||
|
|
83f54a86a8 | ||
|
|
f5f0f99678 | ||
|
|
b0d7151ba1 | ||
|
|
de9e337163 | ||
|
|
a7ebb8b27e | ||
|
|
01e58ab5ed | ||
|
|
7081bbbd91 | ||
|
|
ca89c9480d | ||
|
|
b751e94847 | ||
|
|
3ccb52e719 | ||
|
|
e8b92e347c | ||
|
|
e65b42abc8 | ||
|
|
71096ae67d | ||
|
|
9fb23dadae | ||
|
|
d77781709a | ||
|
|
c059ec30a7 | ||
|
|
6842ad207d | ||
|
|
bbc56d9409 | ||
|
|
66b5e2631e | ||
|
|
a17853b918 | ||
|
|
0a302bb13b | ||
|
|
28e9a47464 | ||
|
|
6c1efece31 | ||
|
|
eb9441a6f3 | ||
|
|
d006dc8a55 | ||
|
|
55cee6933d | ||
|
|
2f375d74e5 | ||
|
|
82856b25e9 | ||
|
|
8464aedceb | ||
|
|
64c3a2c138 | ||
|
|
318bedd4c5 | ||
|
|
91895f6c1a | ||
|
|
b86c18c842 | ||
|
|
5509380cef | ||
|
|
1c818756e4 | ||
|
|
34ac65dc02 | ||
|
|
e6b7e645ce | ||
|
|
b4f8faf5eb |
@@ -181,6 +181,3 @@ site
|
||||
|
||||
# Production logs
|
||||
**/logs/*.log.*
|
||||
**/Dockerfile
|
||||
**/Dockerfile
|
||||
fly.toml
|
||||
|
||||
@@ -56,6 +56,9 @@ LOG_LEVEL=info
|
||||
# Base URL for the application (used for OAuth callbacks and API URLs)
|
||||
BASE_URL=http://localhost:8000
|
||||
|
||||
# JWT Secret for MCP token generation
|
||||
JWT_SECRET_KEY=your_jwt_secret_key_here
|
||||
|
||||
# =============================================================================
|
||||
# MCP SERVER SETTINGS
|
||||
# =============================================================================
|
||||
@@ -67,6 +70,50 @@ BASE_URL=http://localhost:8000
|
||||
# MAX_REQUESTS_PER_MINUTE=60
|
||||
# BURST_CAPACITY=20
|
||||
|
||||
# =============================================================================
|
||||
# SEMANTIC SEARCH SETTINGS (Optional)
|
||||
# =============================================================================
|
||||
|
||||
# Embedding provider for the semantic_search tool.
|
||||
# Pick exactly one of: OpenRouter (hosted) or Local (your own server).
|
||||
|
||||
# --- Option A: OpenRouter (hosted, default) -----------------------------------
|
||||
# Get your API key from: https://openrouter.ai/keys
|
||||
# If neither this nor EMBEDDING_PROVIDER=local is set, semantic search is off.
|
||||
OPENROUTER_API_KEY=sk-or-v1-your_openrouter_api_key_here
|
||||
|
||||
# Optional: override the OpenRouter embedding model and dimension.
|
||||
# Defaults: google/gemini-embedding-001 at 3072 dims (paid on OpenRouter).
|
||||
# Pick any model from https://openrouter.ai/models?modality=embedding
|
||||
# and set the dimension to that model's output size — they must match.
|
||||
# OPENROUTER_EMBEDDING_MODEL=google/gemini-embedding-001
|
||||
# OPENROUTER_EMBEDDING_DIMENSION=3072
|
||||
|
||||
# --- Option B: Local OpenAI-compatible server (no API key required) ----------
|
||||
# Recommended for Turkish: intfloat/multilingual-e5-large served by HuggingFace
|
||||
# Text Embeddings Inference (TEI). One-line setup:
|
||||
#
|
||||
# docker run -p 8080:80 ghcr.io/huggingface/text-embeddings-inference:latest \
|
||||
# --model-id intfloat/multilingual-e5-large
|
||||
#
|
||||
# Then uncomment the block below. Other model families work too — set
|
||||
# EMBEDDING_PROMPT_STYLE to match: e5 / gemini / raw.
|
||||
#
|
||||
# EMBEDDING_PROVIDER=local
|
||||
# LOCAL_EMBEDDING_BASE_URL=http://localhost:8080/v1
|
||||
# LOCAL_EMBEDDING_MODEL=intfloat/multilingual-e5-large
|
||||
# LOCAL_EMBEDDING_DIMENSION=1024
|
||||
# EMBEDDING_PROMPT_STYLE=e5
|
||||
# LOCAL_EMBEDDING_API_KEY= # most local servers ignore this
|
||||
#
|
||||
# Ollama fallback (if you prefer Ollama and don't need top Turkish quality):
|
||||
# ollama serve && ollama pull nomic-embed-text
|
||||
# EMBEDDING_PROVIDER=local
|
||||
# LOCAL_EMBEDDING_BASE_URL=http://localhost:11434/v1
|
||||
# LOCAL_EMBEDDING_MODEL=nomic-embed-text
|
||||
# LOCAL_EMBEDDING_DIMENSION=768
|
||||
# EMBEDDING_PROMPT_STYLE=raw # nomic uses its own search_query/search_document
|
||||
|
||||
# =============================================================================
|
||||
# USAGE INSTRUCTIONS
|
||||
# =============================================================================
|
||||
|
||||
+27
@@ -1,3 +1,6 @@
|
||||
# Serena
|
||||
.serena/
|
||||
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
@@ -190,3 +193,27 @@ GEMINI.md
|
||||
fly.toml
|
||||
scripts/deploy-flyio.sh
|
||||
docs/DEPLOYMENT_FLYIO.md
|
||||
setup_jwt_template.py
|
||||
mcp_server_main.py.backup
|
||||
mcp_overhead_content.json
|
||||
ANTHROPIC_TEST_README.md
|
||||
extract_mcp_overhead.py
|
||||
mcp_overhead_content.txt
|
||||
mcp_overhead_summary.txt
|
||||
run_http_server.py
|
||||
run_local_test.py
|
||||
|
||||
# MCP overhead analysis files
|
||||
mcp_overhead_*.json
|
||||
mcp_overhead_*.txt
|
||||
mcp_test_results_*.json
|
||||
mcp_quick_test_*.json
|
||||
|
||||
# General text files (temporary notes, etc)
|
||||
*.txt
|
||||
analyze_playwright_mcp.py
|
||||
measure_mcp_directly.py
|
||||
playwright_mcp_overhead.json
|
||||
simple_test.py
|
||||
analyze_anayasa_html.py
|
||||
CLAUDE.md
|
||||
|
||||
+44
-19
@@ -1,29 +1,54 @@
|
||||
# -------- BASE IMAGE (includes Chromium & deps) ----------------------------
|
||||
FROM mcr.microsoft.com/playwright/python:v1.52.0-noble
|
||||
# Use Python 3.12 slim image
|
||||
FROM python:3.12-slim
|
||||
|
||||
# -------- Runtime setup ----------------------------------------------------
|
||||
# Set working directory
|
||||
WORKDIR /app
|
||||
|
||||
# Copy dependency manifests first for layer-cache
|
||||
COPY pyproject.toml poetry.lock* requirements*.txt* ./
|
||||
# Install system dependencies (gcc/g++ kept in case any wheel falls back to source build)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc \
|
||||
g++ \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Fast, deterministic install with `uv`
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system --no-cache-dir .[asgi,saas]
|
||||
# Copy project metadata first for better Docker layer caching
|
||||
COPY pyproject.toml ./
|
||||
COPY README.md ./
|
||||
|
||||
# Copy application source
|
||||
COPY . .
|
||||
# Copy entry points
|
||||
COPY app.py ./
|
||||
COPY asgi_app.py ./
|
||||
COPY mcp_server_main.py ./
|
||||
|
||||
# -------- Environment ------------------------------------------------------
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
ENV ENABLE_AUTH=true
|
||||
ENV PORT=8000
|
||||
# Copy MCP modules and shared packages
|
||||
COPY anayasa_mcp_module ./anayasa_mcp_module
|
||||
COPY bddk_mcp_module ./bddk_mcp_module
|
||||
COPY bedesten_mcp_module ./bedesten_mcp_module
|
||||
COPY btk_mcp_module ./btk_mcp_module
|
||||
COPY danistay_mcp_module ./danistay_mcp_module
|
||||
COPY emsal_mcp_module ./emsal_mcp_module
|
||||
COPY gib_mcp_module ./gib_mcp_module
|
||||
COPY kik_mcp_module ./kik_mcp_module
|
||||
COPY kvkk_mcp_module ./kvkk_mcp_module
|
||||
COPY rekabet_mcp_module ./rekabet_mcp_module
|
||||
COPY sayistay_mcp_module ./sayistay_mcp_module
|
||||
COPY sigorta_tahkim_mcp_module ./sigorta_tahkim_mcp_module
|
||||
COPY uyusmazlik_mcp_module ./uyusmazlik_mcp_module
|
||||
COPY yargitay_mcp_module ./yargitay_mcp_module
|
||||
COPY semantic_search ./semantic_search
|
||||
|
||||
# -------- Health check -----------------------------------------------------
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=10s --retries=3 \
|
||||
CMD python -c "import httpx, os, sys; r=httpx.get(f'http://localhost:{os.getenv(\"PORT\",\"8000\")}/health'); sys.exit(0 if r.status_code==200 else 1)"
|
||||
# Install the package with ASGI extras (uvicorn + starlette)
|
||||
RUN pip install --no-cache-dir -e ".[asgi]"
|
||||
|
||||
# Expose port
|
||||
EXPOSE 8000
|
||||
|
||||
# -------- Entrypoint -------------------------------------------------------
|
||||
CMD ["uvicorn", "asgi_app:app", "--host", "0.0.0.0", "--port", "8000", "--proxy-headers"]
|
||||
# Set environment variables
|
||||
ENV PORT=8000
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
|
||||
# Health check
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
|
||||
CMD python -c "import httpx; httpx.get('http://localhost:8000/health', timeout=5)" || exit 1
|
||||
|
||||
# Run the ASGI application
|
||||
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8000"]
|
||||
|
||||
@@ -1,13 +1,110 @@
|
||||
# Yargı MCP: Türk Hukuk Kaynakları için MCP Sunucusu
|
||||
|
||||
> ## ✨ Profesyonel Sürüm Hazır: Yargı MCP Pro
|
||||
>
|
||||
> **Mevzuat ve içtihatı tek bir MCP sunucusunda birleştiren** profesyonel sürüm yayında:
|
||||
>
|
||||
> 👉 **https://yargi.betaspacestudio.com**
|
||||
|
||||
> ## 🚨 SUNUCU YENİ ADRESE TAŞINDI
|
||||
>
|
||||
> **Yeni Remote MCP adresi:** `https://yargimcp.surucu.dev/mcp`
|
||||
>
|
||||
> **Eski adres** (`https://yargimcp.fastmcp.app/mcp`) **artık kullanım dışıdır** — yalnızca taşındığını bildiren bir uyarı tool'u döner.
|
||||
>
|
||||
> **Yapmanız gereken:** MCP istemcinizdeki (Claude Desktop, 5ire, Google Antigravity, ChatGPT vb.) sunucu URL'sini yukarıdaki yeni adresle güncelleyin.
|
||||
|
||||
## Word'den UDF'ye profesyonel dönüşüm için yeni uygulamam [udfcevir.com](https://udfcevir.com) adresinde!
|
||||
|
||||
[](https://www.star-history.com/#saidsurucu/yargi-mcp&Date)
|
||||
|
||||
Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kararlar, Uyuşmazlık Mahkemesi, Anayasa Mahkemesi - Norm Denetimi ile Bireysel Başvuru Kararları, Kamu İhale Kurulu Kararları, Rekabet Kurumu Kararları ve Sayıştay Kararları) erişimi kolaylaştıran bir [FastMCP](https://gofastmcp.com/) sunucusu oluşturur. Bu sayede, bu kaynaklardan veri arama ve belge getirme işlemleri, Model Context Protocol (MCP) destekleyen LLM (Büyük Dil Modeli) uygulamaları (örneğin Claude Desktop veya [5ire](https://5ire.app)) ve diğer istemciler tarafından araç (tool) olarak kullanılabilir hale gelir.
|
||||
Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kararlar, Uyuşmazlık Mahkemesi, Anayasa Mahkemesi - Norm Denetimi ile Bireysel Başvuru Kararları, Kamu İhale Kurulu Kararları, Rekabet Kurumu Kararları, Sayıştay Kararları, KVKK Kararları, BDDK Kararları, BTK Kararları, GİB Özelgeleri ve Sigorta Tahkim Komisyonu Kararları) erişimi kolaylaştıran bir [FastMCP](https://gofastmcp.com/) sunucusu oluşturur. Bu sayede, bu kaynaklardan veri arama ve belge getirme işlemleri, Model Context Protocol (MCP) destekleyen LLM (Büyük Dil Modeli) uygulamaları (örneğin Claude Desktop veya [5ire](https://5ire.app)) ve diğer istemciler tarafından araç (tool) olarak kullanılabilir hale gelir.
|
||||
|
||||
---
|
||||
|
||||
## 🚀 5 Dakikada Başla (Remote MCP)
|
||||
|
||||
### ✅ Kurulum Gerektirmez! Hemen Kullan!
|
||||
|
||||
🔗 **Remote MCP Adresi:** `https://yargimcp.surucu.dev/mcp`
|
||||
|
||||
> ⚠️ **Eski adres** `https://yargimcp.fastmcp.app/mcp` **artık kullanım dışıdır** — yalnızca taşındığını bildiren bir uyarı tool'u döner. Lütfen yukarıdaki yeni adresi kullanın.
|
||||
|
||||
### Claude Desktop ile Kullanım (Ücretli abonelik gerekir)
|
||||
|
||||
1. **Claude Desktop'ı açın**
|
||||
2. **Settings → Connectors → Add Custom Connector**
|
||||
3. **Bilgileri girin:**
|
||||
- **Name:** `Yargı MCP`
|
||||
- **URL:** `https://yargimcp.surucu.dev/mcp`
|
||||
4. **Add** butonuna tıklayın
|
||||
5. **Hemen kullanmaya başlayın!** 🎉
|
||||
|
||||
### Google Antigravity ile Kullanım (Lokal `uv` Kurulumu — Kopyala-Yapıştır)
|
||||
|
||||
> **Ön Gereksinimler:** Bilgisayarınızda **Python**, **`uv`** ([kurulum](https://docs.astral.sh/uv/getting-started/installation/)) ve **Node.js** ([indir](https://nodejs.org/en/download)) kurulu olmalı. (Node.js yalnızca aşağıdaki kurulum komutunu çalıştırmak için gerekir; MCP'yi `uvx` çalıştırır.)
|
||||
|
||||
Aşağıdaki **bloğun tamamını** terminale yapıştırın. Komut, Antigravity'nin okuduğu `~/.gemini/config/mcp_config.json` dosyasını sizin yerinize oluşturur/günceller (varsa diğer sunucularınız korunur):
|
||||
|
||||
**macOS / Linux** (Terminal):
|
||||
|
||||
```bash
|
||||
node - <<'YARGI'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=path.join(os.homedir(),".gemini","config"),file=path.join(dir,"mcp_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
YARGI
|
||||
```
|
||||
|
||||
**Windows** (PowerShell):
|
||||
|
||||
```powershell
|
||||
@'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=path.join(os.homedir(),".gemini","config"),file=path.join(dir,"mcp_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
'@ | node -
|
||||
```
|
||||
|
||||
Komut `yargi-mcp eklendi -> ...` çıktısını verdiğinde kurulum tamamlanmıştır. Antigravity'yi (açıksa kapatıp) yeniden başlatın; `yargi-mcp` araçları otomatik yüklenir.
|
||||
|
||||
> 💡 **İpucu:** Lokal kurulumda hukuk kaynaklarına erişim doğrudan bilgisayarınızda `uvx yargi-mcp` ile çalışır; uzaktan sunucuya ihtiyaç duymaz.
|
||||
|
||||
### Remote MCP Sorun Giderme
|
||||
|
||||
`https://yargimcp.surucu.dev/mcp` bir web sayfası değil, Streamable HTTP MCP uç noktasıdır. Tarayıcıda açınca veya düz `curl` ile GET isteği atınca `406 Not Acceptable` ve `Client must accept text/event-stream` benzeri bir yanıt görmek normaldir; bu, sunucunun kapalı olduğu anlamına gelmez. MCP istemcisi `Accept: application/json, text/event-stream` başlığıyla JSON-RPC isteği göndermelidir.
|
||||
|
||||
Hızlı sağlık kontrolü için tarayıcıda şu adresleri açabilirsiniz:
|
||||
|
||||
- `https://yargimcp.surucu.dev/health` — servis sağlık durumu
|
||||
|
||||
Claude.ai veya başka bir istemci "araç yok" gibi davranırsa:
|
||||
|
||||
1. Connector'ı kaldırıp yeniden ekleyin.
|
||||
2. URL olarak önce `https://yargimcp.surucu.dev/mcp` deneyin; istemciniz yönlendirmeleri takip etmiyorsa `https://yargimcp.surucu.dev/mcp/` deneyin.
|
||||
3. Eski `https://yargimcp.fastmcp.app/mcp` adresinin istemci ayarlarında veya önbellekte kalmadığından emin olun.
|
||||
4. İstemcinin remote/Streamable HTTP MCP desteklediğini ve `text/event-stream` kabul ettiğini kontrol edin.
|
||||
|
||||
---
|
||||
|
||||

|
||||
|
||||
🎯 **Temel Özellikler**
|
||||
|
||||
🚀 **YÜKSEK PERFORMANS OPTİMİZASYONU:** Bu MCP sunucusu **%61.8 token azaltma** ile optimize edilmiştir (8,692 token tasarrufu). Claude AI ile daha hızlı yanıt süreleri ve daha verimli etkileşim sağlar.
|
||||
|
||||
* Çeşitli Türk hukuk veritabanlarına programatik erişim için standart bir MCP arayüzü.
|
||||
* **Kapsamlı Mahkeme Daire/Kurul Filtreleme:** 79 farklı daire/kurul filtreleme seçeneği
|
||||
* **Dual/Triple API Desteği:** Her mahkeme için birden fazla API kaynağı ile maksimum kapsama
|
||||
@@ -26,12 +123,18 @@ Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kara
|
||||
* **KİK (Kamu İhale Kurulu):** Çeşitli kriterlerle Kurul kararlarını arama; uzun karar metinlerini (varsayılan 5.000 karakterlik) sayfalanmış Markdown formatında getirme.
|
||||
* **Rekabet Kurumu:** Çeşitli kriterlerle Kurul kararlarını arama; karar metinlerini Markdown formatında getirme.
|
||||
* **Sayıştay:** 3 karar türü ile kapsamlı denetim kararlarına erişim + **8 Daire Filtreleme** + **Tarih Aralığı & İçerik Arama** (Genel Kurul yorumlayıcı kararları, Temyiz Kurulu itiraz kararları, Daire ilk derece denetim kararları)
|
||||
* **KVKK (Kişisel Verilerin Korunması Kurulu):** Brave Search API ile veri koruma kararlarını arama; uzun karar metinlerini (5.000 karakterlik) sayfalanmış Markdown formatında getirme + **Türkçe Arama** + **Site Hedeflemeli Arama** (kvkk.gov.tr kararları)
|
||||
* **BDDK (Bankacılık Düzenleme ve Denetleme Kurumu):** Bankacılık düzenleme kararlarını arama; karar metinlerini Markdown formatında getirme + **Optimized Search** + **"Karar Sayısı" Targeting** + **Spesifik URL Filtreleme** (bddk.org.tr/Mevzuat/DokumanGetir)
|
||||
* **BTK (Bilgi Teknolojileri ve İletişim Kurumu):** Kurul Kararlarını arama (anahtar kelime + karar no + karar tarihi + yayın tarihi + ilgili birim filtreleri); karar PDF'lerini (5.000 karakterlik) sayfalanmış Markdown formatında getirme (btk.gov.tr)
|
||||
* **GİB (Gelir İdaresi Başkanlığı) Özelgeleri:** Resmi vergi özelgelerini arama (18.000+ özelge: KDV, Kurumlar, Gelir, ÖTV, Damga vb.); tam metni sayfalanmış Markdown formatında getirme + **Keyword + Özelge No + Kanun No + Tarih Aralığı** + **Otomatik ISO 8601 Dönüşümü** + **Metadata Başlık Bloğu**
|
||||
* **Sigorta Tahkim Komisyonu:** Hakem Karar Dergisi (64 sayı, 2010-2025) içindeki sigorta tahkim kararlarını arama; dergi PDF'lerini Markdown formatında getirme + **Sayı İçi Karar Arama** + **Türkçe Büyük/Küçük Harf Desteği** + **Relevance Scoring**
|
||||
|
||||
* Karar metinlerinin daha kolay işlenebilmesi için Markdown formatına çevrilmesi.
|
||||
* Claude Desktop uygulaması ile `fastmcp install` komutu kullanılarak kolay entegrasyon.
|
||||
* Yargı MCP artık [5ire](https://5ire.app) gibi Claude Desktop haricindeki MCP istemcilerini de destekliyor!
|
||||
---
|
||||
🚀 **Claude Haricindeki Modellerle Kullanmak İçin Çok Kolay Kurulum (Örnek: 5ire için)**
|
||||
<details>
|
||||
<summary>🚀 <strong>Claude Haricindeki Modellerle Kullanmak İçin Çok Kolay Kurulum (Örnek: 5ire için)</strong></summary>
|
||||
|
||||
Bu bölüm, Yargı MCP aracını 5ire gibi Claude Desktop dışındaki MCP istemcileriyle kullanmak isteyenler içindir.
|
||||
|
||||
@@ -48,39 +151,81 @@ Bu bölüm, Yargı MCP aracını 5ire gibi Claude Desktop dışındaki MCP istem
|
||||
* **Name:** `Yargı MCP`
|
||||
* **Command:**
|
||||
```
|
||||
uvx --from git+https://github.com/saidsurucu/yargi-mcp yargi-mcp
|
||||
uvx yargi-mcp
|
||||
```
|
||||
* **Save** butonuna basarak kaydedin.
|
||||

|
||||
* Şimdi **Tools** altında **Yargı MCP**'yi görüyor olmalısınız. Üstüne geldiğinizde sağda çıkan butona tıklayıp etkinleştirin (yeşil ışık yanmalı).
|
||||
* Artık Yargı MCP ile konuşabilirsiniz.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
⚙️ **Claude Desktop Manuel Kurulumu**
|
||||
<details>
|
||||
<summary>⚙️ <strong>Claude Desktop Lokal Kurulumu (Kopyala-Yapıştır)</strong></summary>
|
||||
|
||||
> **Ön Gereksinimler:** Bilgisayarınızda **Python**, **`uv`** ([kurulum](https://docs.astral.sh/uv/getting-started/installation/)), **Node.js** ([indir](https://nodejs.org/en/download)) ve (Windows için) Microsoft Visual C++ Redistributable kurulu olmalı. (Node.js yalnızca aşağıdaki kurulum komutunu çalıştırmak için gerekir; MCP'yi `uvx` çalıştırır.)
|
||||
|
||||
1. **Ön Gereksinimler:** Python, `uv`, (Windows için) Microsoft Visual C++ Redistributable'ın sisteminizde kurulu olduğundan emin olun. Detaylı bilgi için yukarıdaki "5ire için Kurulum" bölümündeki ilgili adımlara bakabilirsiniz.
|
||||
2. Claude Desktop **Settings -> Developer -> Edit Config**.
|
||||
3. Açılan `claude_desktop_config.json` dosyasına `mcpServers` altına ekleyin:
|
||||
Aşağıdaki **bloğun tamamını** terminale yapıştırın. Komut, Claude Desktop'ın `claude_desktop_config.json` dosyasını sizin yerinize oluşturur/günceller (varsa diğer sunucularınız korunur):
|
||||
|
||||
**macOS / Linux** (Terminal):
|
||||
|
||||
```bash
|
||||
node - <<'YARGI'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=process.platform==="darwin"
|
||||
? path.join(os.homedir(),"Library","Application Support","Claude")
|
||||
: path.join(os.homedir(),".config","Claude");
|
||||
const file=path.join(dir,"claude_desktop_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
YARGI
|
||||
```
|
||||
|
||||
**Windows** (PowerShell):
|
||||
|
||||
```powershell
|
||||
@'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=path.join(process.env.APPDATA||path.join(os.homedir(),"AppData","Roaming"),"Claude");
|
||||
const file=path.join(dir,"claude_desktop_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
'@ | node -
|
||||
```
|
||||
|
||||
Komut `yargi-mcp eklendi -> ...` çıktısını verdiğinde kurulum tamamlanmıştır. **Claude Desktop'ı tamamen kapatıp yeniden başlatın**; `yargi-mcp` araçları otomatik yüklenir.
|
||||
|
||||
---
|
||||
|
||||
**Manuel alternatif:** Claude Desktop **Settings → Developer → Edit Config** menüsünden `claude_desktop_config.json` dosyasını açıp `mcpServers` altına ekleyebilirsiniz:
|
||||
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
// ... (varsa diğer sunucularınız) ...
|
||||
"Yargı MCP": {
|
||||
"yargi-mcp": {
|
||||
"command": "uvx",
|
||||
"args": [
|
||||
"--from", "git+https://github.com/saidsurucu/yargi-mcp",
|
||||
"yargi-mcp"
|
||||
]
|
||||
"args": ["yargi-mcp"]
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
4. Claude Desktop'ı kapatıp yeniden başlatın.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
🌟 **Gemini CLI ile Kullanım**
|
||||
<details>
|
||||
<summary>🌟 <strong>Gemini CLI ile Kullanım</strong></summary>
|
||||
|
||||
Yargı MCP'yi Gemini CLI ile kullanmak için:
|
||||
|
||||
@@ -101,8 +246,6 @@ Yargı MCP'yi Gemini CLI ile kullanmak için:
|
||||
"yargi_mcp": {
|
||||
"command": "uvx",
|
||||
"args": [
|
||||
"--from",
|
||||
"git+https://github.com/saidsurucu/yargi-mcp",
|
||||
"yargi-mcp"
|
||||
]
|
||||
}
|
||||
@@ -123,81 +266,195 @@ Yargı MCP'yi Gemini CLI ile kullanmak için:
|
||||
- "Danıştay'ın imar planı iptaline ilişkin kararlarını bul"
|
||||
- "Anayasa Mahkemesi'nin ifade özgürlüğü kararlarını getir"
|
||||
|
||||
🛠️ **Kullanılabilir Araçlar (MCP Tools)**
|
||||
</details>
|
||||
|
||||
Bu FastMCP sunucusu aşağıdaki temel araçları sunar:
|
||||
---
|
||||
<details>
|
||||
<summary>🧠 <strong>Semantik Arama (Opsiyonel)</strong></summary>
|
||||
|
||||
### **Yargıtay Araçları (Dual API + 52 Daire Filtreleme)**
|
||||
* **Ana API:**
|
||||
* `search_yargitay_detailed(arananKelime, birimYrgKurulDaire, ...)`: Yargıtay kararlarını detaylı kriterlerle arar. **52 daire/kurul seçeneği** (Hukuk/Ceza Daireleri 1-23, Genel Kurullar, Başkanlar Kurulu)
|
||||
* `get_yargitay_document_markdown(id: str)`: Belirli bir Yargıtay kararının metnini Markdown formatında getirir.
|
||||
* **Bedesten API (Alternatif):**
|
||||
* `search_yargitay_bedesten(phrase, birimAdi, kararTarihiStart, kararTarihiEnd, ...)`: Bedesten API ile Yargıtay kararlarını arar. **Aynı 52 daire filtreleme** + **Tarih Filtreleme** + **Kesin Cümle Arama** (`"\"mülkiyet kararı\""`)
|
||||
* `get_yargitay_bedesten_document_markdown(documentId: str)`: Bedesten'den karar metni (HTML/PDF → Markdown)
|
||||
Yargı MCP, **semantik arama** özelliği ile kararları anlamsal olarak sıralayabilir. Opsiyoneldir; iki yoldan biri yapılandırıldığında otomatik etkinleşir:
|
||||
|
||||
### **Danıştay Araçları (Triple API + 27 Daire Filtreleme)**
|
||||
* **Ana API'lar:**
|
||||
* `search_danistay_by_keyword(andKelimeler, orKelimeler, ...)`: Danıştay kararlarını anahtar kelimelerle arar.
|
||||
* `search_danistay_detailed(daire, esasYil, ...)`: Danıştay kararlarını detaylı kriterlerle arar.
|
||||
* `get_danistay_document_markdown(id: str)`: Belirli bir Danıştay kararının metnini Markdown formatında getirir.
|
||||
* **Bedesten API (Alternatif):**
|
||||
* `search_danistay_bedesten(phrase, birimAdi, kararTarihiStart, kararTarihiEnd, ...)`: Bedesten API ile Danıştay kararlarını arar. **27 daire/kurul seçeneği** + **Tarih Filtreleme** + **Kesin Cümle Arama** (`"\"idari işlem\""`) (1-17. Daireler, Vergi/İdare Kurulları, Askeri Mahkemeler)
|
||||
* `get_danistay_bedesten_document_markdown(documentId: str)`: Bedesten'den karar metni
|
||||
- **Yerel** (önerilen, ücretsiz): kendi makinenizdeki OpenAI-uyumlu embedding sunucusu (HuggingFace TEI, llama.cpp, Ollama, vLLM, LM Studio…)
|
||||
- **Hosted**: OpenRouter API anahtarı
|
||||
|
||||
### **Diğer Mahkemeler (Bedesten API + Gelişmiş Arama)**
|
||||
* **Yerel Hukuk Mahkemeleri:**
|
||||
* `search_yerel_hukuk_bedesten(phrase, kararTarihiStart, kararTarihiEnd, ...)`: Yerel hukuk mahkemesi kararlarını arar + **Tarih & Kesin Cümle Arama** (`"\"sözleşme ihlali\""`)
|
||||
* `get_yerel_hukuk_bedesten_document_markdown(documentId: str)`: Karar metni
|
||||
* **İstinaf Hukuk Mahkemeleri:**
|
||||
* `search_istinaf_hukuk_bedesten(phrase, kararTarihiStart, kararTarihiEnd, ...)`: İstinaf mahkemesi kararlarını arar + **Tarih & Kesin Cümle Arama** (`"\"temyiz incelemesi\""`)
|
||||
* `get_istinaf_hukuk_bedesten_document_markdown(documentId: str)`: Karar metni
|
||||
* **Kanun Yararına Bozma (KYB):**
|
||||
* `search_kyb_bedesten(phrase, kararTarihiStart, kararTarihiEnd, ...)`: Olağanüstü kanun yolu kararlarını arar + **Tarih & Kesin Cümle Arama** (`"\"kanun yararına bozma\""`)
|
||||
* `get_kyb_bedesten_document_markdown(documentId: str)`: Karar metni
|
||||
### Semantik Arama Nasıl Çalışır?
|
||||
1. `initial_keyword` ile Bedesten API'den 100 karar çekilir
|
||||
2. `query` ile bu kararlar embedding modeli kullanılarak anlamsal olarak sıralanır
|
||||
3. En alakalı kararlar döndürülür
|
||||
|
||||
* **Emsal Karar Araçları:**
|
||||
* `search_emsal_detailed_decisions(search_query: EmsalSearchRequest) -> CompactEmsalSearchResult`: Emsal (UYAP) kararlarını detaylı kriterlerle arar.
|
||||
* `get_emsal_document_markdown(id: str) -> EmsalDocumentMarkdown`: Belirli bir Emsal kararının metnini Markdown formatında getirir.
|
||||
### Önerilen Türkçe Kurulumu (Yerel — `multilingual-e5-large`)
|
||||
|
||||
* **Uyuşmazlık Mahkemesi Araçları:**
|
||||
* `search_uyusmazlik_decisions(search_params: UyusmazlikSearchRequest) -> UyusmazlikSearchResponse`: Uyuşmazlık Mahkemesi kararlarını çeşitli form kriterleriyle arar.
|
||||
* `get_uyusmazlik_document_markdown_from_url(document_url: HttpUrl) -> UyusmazlikDocumentMarkdown`: Bir Uyuşmazlık kararını tam URL'sinden alıp Markdown formatında getirir.
|
||||
`intfloat/multilingual-e5-large` Türkçe için kıyas ettiğimiz açık kaynak modeller arasında en iyilerinden. HuggingFace'in **Text Embeddings Inference (TEI)** sunucusuyla tek komutta ayağa kalkar ve OpenAI-uyumlu API sunar:
|
||||
|
||||
* **Anayasa Mahkemesi (Norm Denetimi) Araçları:**
|
||||
* `search_anayasa_norm_denetimi_decisions(search_query: AnayasaNormDenetimiSearchRequest) -> AnayasaSearchResult`: AYM Norm Denetimi kararlarını kapsamlı kriterlerle arar.
|
||||
* `get_anayasa_norm_denetimi_document_markdown(document_url: str, page_number: Optional[int] = 1) -> AnayasaDocumentMarkdown`: Belirli bir AYM Norm Denetimi kararını URL'sinden alır ve 5.000 karakterlik sayfalanmış Markdown içeriğini getirir.
|
||||
```bash
|
||||
docker run -p 8080:80 ghcr.io/huggingface/text-embeddings-inference:latest \
|
||||
--model-id intfloat/multilingual-e5-large
|
||||
```
|
||||
|
||||
* **Anayasa Mahkemesi (Bireysel Başvuru) Araçları:**
|
||||
* `search_anayasa_bireysel_basvuru_report(search_query: AnayasaBireyselReportSearchRequest) -> AnayasaBireyselReportSearchResult`: AYM Bireysel Başvuru "Karar Arama Raporu" oluşturur.
|
||||
* `get_anayasa_bireysel_basvuru_document_markdown(document_url_path: str, page_number: Optional[int] = 1) -> AnayasaBireyselBasvuruDocumentMarkdown`: Belirli bir AYM Bireysel Başvuru kararını URL path'inden alır ve 5.000 karakterlik sayfalanmış Markdown içeriğini getirir.
|
||||
Sonra Yargı MCP'ye şu env vars'ları geçirin:
|
||||
|
||||
* **KİK (Kamu İhale Kurulu) Araçları:**
|
||||
* `search_kik_decisions(search_query: KikSearchRequest) -> KikSearchResult`: KİK (Kamu İhale Kurulu) kararlarını arar.
|
||||
* `get_kik_document_markdown(karar_id: str, page_number: Optional[int] = 1) -> KikDocumentMarkdown`: Belirli bir KİK kararını, Base64 ile encode edilmiş `karar_id`'sini kullanarak alır ve 5.000 karakterlik sayfalanmış Markdown içeriğini getirir.
|
||||
* **Rekabet Kurumu Araçları:**
|
||||
```bash
|
||||
EMBEDDING_PROVIDER=local
|
||||
LOCAL_EMBEDDING_BASE_URL=http://localhost:8080/v1
|
||||
LOCAL_EMBEDDING_MODEL=intfloat/multilingual-e5-large
|
||||
LOCAL_EMBEDDING_DIMENSION=1024
|
||||
EMBEDDING_PROMPT_STYLE=e5
|
||||
```
|
||||
|
||||
> ⚠️ **Önemli:** `EMBEDDING_PROMPT_STYLE=e5` şart — e5 modelleri `query:` / `passage:` öneki bekleyecek şekilde eğitilmiştir; yanlış önek sessizce kaliteyi düşürür.
|
||||
|
||||
#### Claude Desktop örneği (yerel TEI)
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"Yargı MCP": {
|
||||
"command": "uvx",
|
||||
"args": ["yargi-mcp"],
|
||||
"env": {
|
||||
"EMBEDDING_PROVIDER": "local",
|
||||
"LOCAL_EMBEDDING_BASE_URL": "http://localhost:8080/v1",
|
||||
"LOCAL_EMBEDDING_MODEL": "intfloat/multilingual-e5-large",
|
||||
"LOCAL_EMBEDDING_DIMENSION": "1024",
|
||||
"EMBEDDING_PROMPT_STYLE": "e5"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Alternatif 1: Ollama (yerel, daha hafif kurulum)
|
||||
|
||||
```bash
|
||||
ollama serve
|
||||
ollama pull nomic-embed-text # 768 dim, İngilizce ağırlıklı
|
||||
```
|
||||
|
||||
```bash
|
||||
EMBEDDING_PROVIDER=local
|
||||
LOCAL_EMBEDDING_BASE_URL=http://localhost:11434/v1
|
||||
LOCAL_EMBEDDING_MODEL=nomic-embed-text
|
||||
LOCAL_EMBEDDING_DIMENSION=768
|
||||
EMBEDDING_PROMPT_STYLE=raw
|
||||
```
|
||||
|
||||
> Ollama kütüphanesinde `multilingual-e5-large` doğrudan yok; Türkçe için TEI yolu daha doğru sonuç verir.
|
||||
|
||||
### Alternatif 2: OpenRouter (hosted)
|
||||
|
||||
```bash
|
||||
OPENROUTER_API_KEY=sk-or-v1-xxx...
|
||||
# İsteğe bağlı — varsayılan google/gemini-embedding-001 (3072 dim, ÜCRETLİ)
|
||||
# OPENROUTER_EMBEDDING_MODEL=...
|
||||
# OPENROUTER_EMBEDDING_DIMENSION=...
|
||||
# EMBEDDING_PROMPT_STYLE=gemini # varsayılan
|
||||
```
|
||||
|
||||
API anahtarınızı [openrouter.ai/keys](https://openrouter.ai/keys) adresinden alın. Varsayılan model `google/gemini-embedding-001` artık ücretli — ücretsiz bir model seçerseniz `OPENROUTER_EMBEDDING_MODEL`, `OPENROUTER_EMBEDDING_DIMENSION` ve uygun `EMBEDDING_PROMPT_STYLE` değerlerini birlikte ayarlayın.
|
||||
|
||||
### Yapılandırma Referansı
|
||||
|
||||
| Env Var | Açıklama | Örnek |
|
||||
|---|---|---|
|
||||
| `EMBEDDING_PROVIDER` | `local` ise yerel sunucu, boş ise OpenRouter | `local` |
|
||||
| `EMBEDDING_PROMPT_STYLE` | `gemini` / `e5` / `raw` — modelin beklediği önek | `e5` |
|
||||
| `LOCAL_EMBEDDING_BASE_URL` | Yerel sunucunun OpenAI-uyumlu URL'i | `http://localhost:8080/v1` |
|
||||
| `LOCAL_EMBEDDING_MODEL` | Model adı | `intfloat/multilingual-e5-large` |
|
||||
| `LOCAL_EMBEDDING_DIMENSION` | Modelin çıktı boyutu (mutlaka eşleşmeli) | `1024` |
|
||||
| `OPENROUTER_API_KEY` | OpenRouter anahtarı (sadece hosted için) | `sk-or-v1-…` |
|
||||
| `OPENROUTER_EMBEDDING_MODEL` | OpenRouter model id'si | `google/gemini-embedding-001` |
|
||||
| `OPENROUTER_EMBEDDING_DIMENSION` | OpenRouter modelinin çıktı boyutu | `3072` |
|
||||
|
||||
> 💡 **Not:** Hiçbir embedding sağlayıcı yapılandırılmazsa semantik arama aracı görünmez, diğer 28 araç normal şekilde çalışır.
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary>🛠️ <strong>Kullanılabilir Araçlar (MCP Tools)</strong></summary>
|
||||
|
||||
Bu FastMCP sunucusu **26 aktif MCP aracı** + **1 opsiyonel semantik arama aracı** sunar (token verimliliği için optimize edilmiş):
|
||||
|
||||
### **Yargıtay Araçları (Birleşik Bedesten API - Token Optimized)**
|
||||
*Not: Yargıtay araçları token verimliliği için birleşik Bedesten API'ye entegre edilmiştir*
|
||||
|
||||
### **Danıştay Araçları (Birleşik Bedesten API - Token Optimized)**
|
||||
*Not: Danıştay araçları token verimliliği için birleşik Bedesten API'ye entegre edilmiştir*
|
||||
|
||||
### **Birleşik Bedesten API Araçları (5 Mahkeme) - 🚀 TOKEN OPTİMİZE**
|
||||
1. `search_bedesten_unified(phrase, court_types, birimAdi, kararTarihiStart, kararTarihiEnd, ...)`: **5 mahkeme türünü** birleşik arama (Yargıtay, Danıştay, Yerel Hukuk, İstinaf Hukuk, KYB) + **79 daire filtreleme** + **Tarih & Kesin Cümle Arama**
|
||||
2. `get_bedesten_document_markdown(documentId: str)`: Bedesten API'den herhangi bir belgeyi Markdown formatında getirir (HTML/PDF → Markdown)
|
||||
|
||||
### **Emsal Karar Araçları (UYAP)**
|
||||
3. `search_emsal_detailed_decisions(keyword, ...)`: Emsal (UYAP) kararlarını detaylı kriterlerle arar.
|
||||
4. `get_emsal_document_markdown(id: str)`: Belirli bir Emsal kararının metnini Markdown formatında getirir.
|
||||
|
||||
### **Uyuşmazlık Mahkemesi Araçları**
|
||||
5. `search_uyusmazlik_decisions(icerik, ...)`: Uyuşmazlık Mahkemesi kararlarını çeşitli form kriterleriyle arar.
|
||||
6. `get_uyusmazlik_document_markdown_from_url(document_url)`: Bir Uyuşmazlık kararını tam URL'sinden alıp Markdown formatında getirir.
|
||||
|
||||
### **Anayasa Mahkemesi Araçları (Birleşik API) - 🚀 TOKEN OPTİMİZE**
|
||||
7. `search_anayasa_unified(decision_type, keywords_all, ...)`: AYM kararlarını birleşik arama (Norm Denetimi + Bireysel Başvuru) - **4 araç → 2 araç optimizasyonu**
|
||||
8. `get_anayasa_document_unified(document_url, page_number)`: AYM kararlarını birleşik belge getirme - **sayfalanmış Markdown** içeriği
|
||||
|
||||
### **KİK (Kamu İhale Kurulu) Araçları**
|
||||
9. `search_kik_v2_decisions(decision_type, karar_metni, karar_no, basvuran, idare_adi, baslangic_tarihi, bitis_tarihi)`: KİK v2 API ile uyuşmazlık, düzenleyici ve mahkeme kararlarını arar.
|
||||
10. `get_kik_v2_document_markdown(gundemMaddesiId)`: Arama sonucundaki `gundemMaddesiId` ile KİK karar metnini Markdown formatında getirir.
|
||||
### **Rekabet Kurumu Araçları**
|
||||
* `search_rekabet_kurumu_decisions(KararTuru: Literal[...], ...) -> RekabetSearchResult`: Rekabet Kurumu kararlarını arar. `KararTuru` için kullanıcı dostu isimler kullanılır (örn: "Birleşme ve Devralma").
|
||||
* `get_rekabet_kurumu_document(karar_id: str, page_number: Optional[int] = 1) -> RekabetDocument`: Belirli bir Rekabet Kurumu kararını `karar_id` ile alır. Kararın PDF formatındaki orijinalinden istenen sayfayı ayıklar ve Markdown formatında döndürür.
|
||||
|
||||
|
||||
---
|
||||
|
||||
* **Sayıştay Araçları (3 Karar Türü + 8 Daire Filtreleme):**
|
||||
* `search_sayistay_genel_kurul(karar_no, karar_tarih_baslangic, karar_tamami, ...)`: Sayıştay Genel Kurul (yorumlayıcı) kararlarını arar. **Tarih aralığı** (2006-2024) + **İçerik arama** (400 karakter)
|
||||
* `search_sayistay_temyiz_kurulu(ilam_dairesi, kamu_idaresi_turu, temyiz_karar, ...)`: Temyiz Kurulu (itiraz) kararlarını arar. **8 Daire filtreleme** + **Kurum türü** + **Konu sınıflandırması**
|
||||
* `search_sayistay_daire(yargilama_dairesi, web_karar_metni, hesap_yili, ...)`: Daire (ilk derece denetim) kararlarını arar. **8 Daire filtreleme** + **Hesap yılı** + **İçerik arama**
|
||||
* `get_sayistay_genel_kurul_document_markdown(decision_id: str)`: Genel Kurul kararının tam metnini Markdown formatında getirir
|
||||
* `get_sayistay_temyiz_kurulu_document_markdown(decision_id: str)`: Temyiz Kurulu kararının tam metnini Markdown formatında getirir
|
||||
* `get_sayistay_daire_document_markdown(decision_id: str)`: Daire kararının tam metnini Markdown formatında getirir
|
||||
* **Sayıştay Araçları (Birleşik API, 3 Karar Türü + 8 Daire Filtreleme):**
|
||||
* `search_sayistay_unified(decision_type, start, length, ...)`: `genel_kurul`, `temyiz_kurulu` veya `daire` kararlarını tek araçla arar. `length` 1-100 aralığındadır.
|
||||
* `get_sayistay_document_unified(decision_id, decision_type)`: Birleşik arama sonucundaki karar ID'si ve karar türüyle tam metni Markdown formatında getirir.
|
||||
|
||||
* **KVKK Araçları (Brave Search API + Türkçe Arama):**
|
||||
* `search_kvkk_decisions(keywords, page)`: KVKK (Kişisel Verilerin Korunması Kurulu) kararlarını Brave Search API ile arar. **Türkçe arama** + **Site hedeflemeli** (`site:kvkk.gov.tr "karar özeti"`) + **Sayfalama desteği**. Sonuç sayısı sunucuda 10 olarak sabitlenmiştir.
|
||||
* `get_kvkk_document_markdown(decision_url: str, page_number: Optional[int] = 1)`: KVKK kararının tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa)
|
||||
|
||||
### BDDK Araçları
|
||||
* `search_bddk_decisions(keywords, page)`: BDDK (Bankacılık Düzenleme ve Denetleme Kurumu) kararlarını arar. **"Karar Sayısı" targeting** + **Spesifik URL filtreleme** (`bddk.org.tr/Mevzuat/DokumanGetir`) + **Optimized search**
|
||||
* `get_bddk_document_markdown(document_id: str, page_number: Optional[int] = 1)`: BDDK kararının tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa)
|
||||
|
||||
### BTK (Bilgi Teknolojileri ve İletişim Kurumu) Araçları (Resmi BTK JSON API)
|
||||
* `search_btk_decisions(keywords, decision_no, decision_date, publication_date, relevant_unit, page, pageSize)`: BTK Kurul Kararlarını arar. **Anahtar kelime + Karar No** (ör. `2026/DK-THD/91`) **+ Karar Tarihi + Yayın Tarihi + İlgili Birim** filtreleri + **Sayfalama** (`pageSize` 1-50)
|
||||
* `get_btk_document_markdown(pdf_url: str, page_number: int = 1)`: BTK kararının PDF'ini indirip **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa). `pdf_url`, `search_btk_decisions` sonucundaki `pdf_url` alanından alınır (`btk.gov.tr`)
|
||||
|
||||
### GİB (Gelir İdaresi Başkanlığı) Özelge Araçları (Resmi GİB JSON API)
|
||||
* `search_gib_ozelge(keywords, ozelgeNo, kanunNo, ozelgeStartDate, ozelgeEndDate, page, pageSize)`: GİB özelgelerini (Türk Gelir İdaresi Başkanlığı vergi özelgeleri) arar — **18.000+ özelge** (KDV, Kurumlar, Gelir, ÖTV, Damga, VUK vb.). **Keyword + Özelge No + Kanun No + Tarih Aralığı** + **Otomatik ISO 8601 Dönüşümü** (`YYYY-MM-DD` girdileri otomatik olarak full ISO 8601'e çevrilir)
|
||||
* `get_gib_ozelge_document_markdown(ozelge_id: int, page_number: int = 1)`: Belirli bir özelgenin tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa) + **Metadata başlık bloğu** (Başlık, Sayı, Tarih, Kanun, Kaynak URL)
|
||||
|
||||
### Sigorta Tahkim Komisyonu Araçları (Tavily Search API + PDF)
|
||||
* `search_sigorta_tahkim_decisions(keywords, page)`: Sigorta Tahkim Komisyonu kararlarını Tavily Search API ile arar. **Site hedeflemeli** (`sigortatahkim.org`) + **Sayfalama desteği**. Sonuç sayısı sunucuda 10 olarak sabitlenmiştir.
|
||||
* `get_sigorta_tahkim_document_markdown(issue_number: str, page_number: int)`: Hakem Karar Dergisi sayısının PDF'ini indirip **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa). 64 sayı (2010-2025)
|
||||
* `search_within_sigorta_tahkim_issue(issue_number: str, keyword: str, max_results: int)`: Belirli bir dergi sayısı içindeki kararları anahtar kelime ile arar. **Türkçe İ/I desteği** + **Relevance scoring** + **Excerpt** ile sonuç
|
||||
|
||||
### Yardımcı ve Uyumluluk Araçları
|
||||
* `check_government_servers_health()`: Yargı kaynaklarının erişilebilirliğini kontrol eder.
|
||||
* `search(query)`: ChatGPT Deep Research uyumluluğu için Bedesten destekli kaynaklarda arama yapar.
|
||||
* `fetch(id)`: ChatGPT Deep Research uyumluluğu için tek bir Bedesten belge ID'sinin tam metnini getirir.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
|
||||
### **📊 Kapsamlı İstatistikler**
|
||||
- **Toplam Mahkeme/Kurum:** 12 farklı hukuki kurum
|
||||
- **Toplam MCP Tool:** 36+ arama ve belge getirme aracı
|
||||
<details>
|
||||
<summary>📊 <strong>Kapsamlı İstatistikler & Optimizasyon Başarıları</strong></summary>
|
||||
|
||||
🚀 **TOKEN OPTİMİZASYON BAŞARISI:**
|
||||
- **%61.8 Token Azaltma:** 14,061 → 5,369 tokens (8,692 token tasarrufu)
|
||||
- **Hedef Aşım:** 10,000 token hedefini 4,631 token aştık
|
||||
- **Daha Hızlı Yanıt:** Claude AI ile optimize edilmiş etkileşim
|
||||
- **Korunan İşlevsellik:** %100 özellik desteği devam ediyor
|
||||
|
||||
**GENEL İSTATİSTİKLER:**
|
||||
- **Toplam Mahkeme/Kurum:** 16 farklı hukuki kurum (BTK, GİB Özelgeleri ve Sigorta Tahkim Komisyonu dahil)
|
||||
- **Toplam MCP Tool:** 28 aktif araç + 1 opsiyonel semantik arama aracı
|
||||
- **Daire/Kurul Filtreleme:** 87 farklı seçenek (52 Yargıtay + 27 Danıştay + 8 Sayıştay)
|
||||
- **Tarih Filtreleme:** 5 Bedesten API aracında ISO 8601 formatında tam tarih aralığı desteği
|
||||
- **Kesin Cümle Arama:** 5 Bedesten API aracında çift tırnak ile tam cümle arama (`"\"mülkiyet kararı\""` formatı)
|
||||
- **Tarih Filtreleme:** Birleşik Bedesten API aracında ISO 8601 formatında tam tarih aralığı desteği
|
||||
- **Kesin Cümle Arama:** Birleşik Bedesten API aracında çift tırnak ile tam cümle arama (`"\"mülkiyet kararı\""` formatı)
|
||||
- **Birleşik API:** 10 ayrı Bedesten aracı → 2 birleşik araç (search_bedesten_unified + get_bedesten_document_markdown)
|
||||
- **API Kaynağı:** Dual/Triple API desteği ile maksimum kapsama
|
||||
- **Tam Türk Adalet Sistemi:** Yerel mahkemelerden en yüksek mahkemelere kadar
|
||||
|
||||
@@ -222,9 +479,19 @@ Bedesten API Bedesten API Dual/Triple API Norm+Bireysel API
|
||||
- Kesin arama: `"\"mülkiyet kararı\""` (tam cümle olarak)
|
||||
- Daha kesin sonuçlar için hukuki terimler ve kavramlar
|
||||
|
||||
**🔧 OPTİMİZASYON DETAYLARI:**
|
||||
- **Anayasa Mahkemesi:** 4 araç → 2 birleşik araç (search_anayasa_unified + get_anayasa_document_unified)
|
||||
- **Yargıtay & Danıştay:** Ana API araçları birleşik Bedesten API'ye entegre edildi
|
||||
- **Sayıştay:** 6 araç → 2 birleşik araç (search_sayistay_unified + get_sayistay_document_unified)
|
||||
- **Parameter Optimizasyonu:** pageSize parametreleri optimize edildi
|
||||
- **Açıklama Optimizasyonu:** Uzun açıklamalar kısaltıldı (örn: KIK karar_metni)
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
|
||||
🌐 **Web Service / ASGI Deployment**
|
||||
<details>
|
||||
<summary>🌐 <strong>Web Service / ASGI Deployment</strong></summary>
|
||||
|
||||
Yargı MCP artık web servisi olarak da çalıştırılabilir! ASGI desteği sayesinde:
|
||||
|
||||
@@ -246,6 +513,8 @@ uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
|
||||
Detaylı deployment rehberi için: [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md)
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
|
||||
📜 **Lisans**
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
"""
|
||||
Analyze KİK v2 hash generation by examining JavaScript code patterns
|
||||
and trying to reverse engineer the hash generation logic.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import hashlib
|
||||
import hmac
|
||||
import base64
|
||||
from fastmcp import Client
|
||||
from mcp_server_main import app
|
||||
|
||||
def analyze_webpack_hash_patterns():
|
||||
"""
|
||||
Analyze the webpack JavaScript code you provided to find hash generation patterns
|
||||
"""
|
||||
print("🔍 Analyzing webpack hash generation patterns...")
|
||||
|
||||
# From the JavaScript code, I can see several hash/ID generation patterns:
|
||||
hash_patterns = {
|
||||
# Webpack chunk system hashes (from the JS code)
|
||||
"webpack_chunks": {
|
||||
315: "d9a9486a4f5ba326",
|
||||
531: "cd8fb385c88033ae",
|
||||
671: "04c48b287646627a",
|
||||
856: "682c9a7b87351f90",
|
||||
1017: "9de022378fc275f6",
|
||||
# ... many more from the __webpack_require__.u function
|
||||
},
|
||||
|
||||
# Symbol generation from Zone.js
|
||||
"zone_symbols": [
|
||||
"__zone_symbol__",
|
||||
"__Zone_symbol_prefix",
|
||||
"Zone.__symbol__"
|
||||
],
|
||||
|
||||
# Angular module federation patterns
|
||||
"module_federation": [
|
||||
"__webpack_modules__",
|
||||
"__webpack_module_cache__",
|
||||
"__webpack_require__"
|
||||
]
|
||||
}
|
||||
|
||||
# The target hash format
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"🎯 Target hash: {target_hash}")
|
||||
print(f" Length: {len(target_hash)} characters")
|
||||
print(f" Format: {'SHA256' if len(target_hash) == 64 else 'Other'} (64 chars = SHA256)")
|
||||
|
||||
return hash_patterns
|
||||
|
||||
def test_webpack_style_hashing(data_dict):
|
||||
"""Test webpack-style hash generation methods"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try various webpack-style hash methods
|
||||
hashes[f"webpack_md5_{key}"] = hashlib.md5(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha1_{key}"] = hashlib.sha1(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha256_{key}"] = hashlib.sha256(test_string.encode()).hexdigest()
|
||||
|
||||
# Try with various prefixes/suffixes (common in webpack)
|
||||
prefixed = f"__webpack__{test_string}"
|
||||
hashes[f"webpack_prefixed_sha256_{key}"] = hashlib.sha256(prefixed.encode()).hexdigest()
|
||||
|
||||
# Try with module federation style
|
||||
module_style = f"shell:{test_string}"
|
||||
hashes[f"module_fed_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
# Try JSON stringified
|
||||
json_style = json.dumps({"id": value, "type": "decision"}, separators=(',', ':'))
|
||||
hashes[f"json_sha256_{key}"] = hashlib.sha256(json_style.encode()).hexdigest()
|
||||
|
||||
# Try with timestamp or sequence
|
||||
with_seq = f"{test_string}_0"
|
||||
hashes[f"seq_sha256_{key}"] = hashlib.sha256(with_seq.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_angular_routing_hashes(data_dict):
|
||||
"""Test Angular routing/state management hash generation"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
# Angular often uses route parameters for hash generation
|
||||
route_style = f"/kurul-kararlari/{value}"
|
||||
hashes[f"route_sha256_{key}"] = hashlib.sha256(route_style.encode()).hexdigest()
|
||||
|
||||
# Component state style
|
||||
state_style = f"KurulKararGoster_{value}"
|
||||
hashes[f"state_sha256_{key}"] = hashlib.sha256(state_style.encode()).hexdigest()
|
||||
|
||||
# Angular module style
|
||||
module_style = f"kik.kurul.karar.{value}"
|
||||
hashes[f"module_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_base64_encoding_variants(data_dict):
|
||||
"""Test various base64 and encoding variants"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try base64 encoding then hashing
|
||||
b64_encoded = base64.b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64_sha256_{key}"] = hashlib.sha256(b64_encoded.encode()).hexdigest()
|
||||
|
||||
# Try URL-safe base64
|
||||
b64_url = base64.urlsafe_b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64url_sha256_{key}"] = hashlib.sha256(b64_url.encode()).hexdigest()
|
||||
|
||||
# Try hex encoding
|
||||
hex_encoded = test_string.encode().hex()
|
||||
hashes[f"hex_sha256_{key}"] = hashlib.sha256(hex_encoded.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
async def test_hash_generation_comprehensive():
|
||||
print("🔐 Comprehensive KİK document hash generation analysis...")
|
||||
print("=" * 70)
|
||||
|
||||
# First analyze the webpack patterns
|
||||
webpack_patterns = analyze_webpack_hash_patterns()
|
||||
|
||||
client = Client(app)
|
||||
|
||||
async with client:
|
||||
print("✅ MCP client connected")
|
||||
|
||||
# Get sample decisions
|
||||
print("\n📊 Getting sample decisions for hash analysis...")
|
||||
search_result = await client.call_tool("search_kik_v2_decisions", {
|
||||
"decision_type": "uyusmazlik",
|
||||
"karar_metni": "2024"
|
||||
})
|
||||
|
||||
if hasattr(search_result, 'content') and search_result.content:
|
||||
search_data = json.loads(search_result.content[0].text)
|
||||
decisions = search_data.get('decisions', [])
|
||||
|
||||
if decisions:
|
||||
print(f"✅ Found {len(decisions)} decisions")
|
||||
|
||||
# Test with first decision
|
||||
sample_decision = decisions[0]
|
||||
print(f"\n📋 Sample decision for hash analysis:")
|
||||
for key, value in sample_decision.items():
|
||||
print(f" {key}: {value}")
|
||||
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"\n🎯 Target hash to match: {target_hash}")
|
||||
|
||||
all_hashes = {}
|
||||
|
||||
# Test different hash generation methods
|
||||
print(f"\n🔨 Testing webpack-style hashing...")
|
||||
webpack_hashes = test_webpack_style_hashing(sample_decision)
|
||||
all_hashes.update(webpack_hashes)
|
||||
|
||||
print(f"🔨 Testing Angular routing hashes...")
|
||||
angular_hashes = test_angular_routing_hashes(sample_decision)
|
||||
all_hashes.update(angular_hashes)
|
||||
|
||||
print(f"🔨 Testing base64 encoding variants...")
|
||||
b64_hashes = test_base64_encoding_variants(sample_decision)
|
||||
all_hashes.update(b64_hashes)
|
||||
|
||||
# Check for matches
|
||||
print(f"\n🎯 Checking for hash matches...")
|
||||
matches_found = []
|
||||
partial_matches = []
|
||||
|
||||
for hash_name, hash_value in all_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
matches_found.append((hash_name, hash_value))
|
||||
print(f" 🎉 EXACT MATCH FOUND: {hash_name}")
|
||||
elif hash_value[:8] == target_hash[:8]: # First 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (first 8): {hash_name} -> {hash_value[:16]}...")
|
||||
elif hash_value[-8:] == target_hash[-8:]: # Last 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (last 8): {hash_name} -> ...{hash_value[-16:]}")
|
||||
|
||||
if not matches_found and not partial_matches:
|
||||
print(f" ❌ No matches found")
|
||||
print(f"\n📝 Sample generated hashes (first 10):")
|
||||
for i, (hash_name, hash_value) in enumerate(list(all_hashes.items())[:10]):
|
||||
print(f" {hash_name}: {hash_value}")
|
||||
|
||||
# Try combinations with other decisions
|
||||
print(f"\n🔄 Testing hash combinations with multiple decisions...")
|
||||
if len(decisions) > 1:
|
||||
for i, decision in enumerate(decisions[1:3]): # Test 2 more
|
||||
print(f"\n Testing decision {i+2}: {decision.get('kararNo')}")
|
||||
decision_hashes = test_webpack_style_hashing(decision)
|
||||
|
||||
for hash_name, hash_value in decision_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
print(f" 🎉 MATCH FOUND in decision {i+2}: {hash_name}")
|
||||
matches_found.append((f"decision_{i+2}_{hash_name}", hash_value))
|
||||
|
||||
# Try composite hashes (combining multiple fields)
|
||||
print(f"\n🔗 Testing composite hash generation...")
|
||||
composite_tests = [
|
||||
f"{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararNo')}",
|
||||
f"{sample_decision.get('kararNo')}_{sample_decision.get('kararTarihi')}",
|
||||
f"uyusmazlik_{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararTarihi')}",
|
||||
json.dumps(sample_decision, separators=(',', ':'), sort_keys=True),
|
||||
f"{sample_decision.get('basvuran')}_{sample_decision.get('gundemMaddesiId')}",
|
||||
]
|
||||
|
||||
for i, composite_str in enumerate(composite_tests):
|
||||
composite_hash = hashlib.sha256(composite_str.encode()).hexdigest()
|
||||
if composite_hash == target_hash:
|
||||
print(f" 🎉 COMPOSITE MATCH FOUND: test_{i} -> {composite_str[:50]}...")
|
||||
matches_found.append((f"composite_{i}", composite_hash))
|
||||
|
||||
print(f"\n🎯 Hash analysis completed!")
|
||||
print(f" Total matches found: {len(matches_found)}")
|
||||
print(f" Partial matches: {len(partial_matches)}")
|
||||
|
||||
else:
|
||||
print("❌ No decisions found")
|
||||
else:
|
||||
print("❌ Search failed")
|
||||
|
||||
print("=" * 70)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_hash_generation_comprehensive())
|
||||
@@ -0,0 +1,201 @@
|
||||
# anayasa_mcp_module/api_client.py
|
||||
# Low-level client for the new Anayasa Mahkemesi "Kararlar Bilgi Bankası" (KBB) JSON API.
|
||||
#
|
||||
# Both the Norm Denetimi host (normkararlarbilgibankasi.anayasa.gov.tr) and the
|
||||
# Bireysel Başvuru host (kararlarbilgibankasi.anayasa.gov.tr) share the SAME
|
||||
# backend, exposed at POST /api/core/public/search. The request differs only by
|
||||
# the "kararTipi" discriminator:
|
||||
#
|
||||
# {"kararTipi": "NormDenetimi", "query": "mülkiyet", "page": 1, "size": 10}
|
||||
# -> {"total": N, "page": 1, "data": [...summary records...], "page_size": 10}
|
||||
#
|
||||
# {"kararTipi": "NormDenetimi", "id": "<uuid>", "page": 1, "size": 1}
|
||||
# -> data[0] additionally includes "icerik" = full decision HTML
|
||||
#
|
||||
# The previous HTML-scraping endpoints (/Ara, /ND/.., /BB/..) were retired when
|
||||
# the sites were rebuilt as a single-page app; they now return HTTP 404.
|
||||
|
||||
import base64
|
||||
import html as html_module
|
||||
import io
|
||||
import logging
|
||||
import re
|
||||
from typing import Any, Dict, Optional, Tuple
|
||||
from urllib.parse import urlparse, parse_qs, quote
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from markitdown import MarkItDown
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Markdown pagination chunk size (characters), shared across AYM document tools.
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
|
||||
def strip_html_text(value: Optional[str]) -> str:
|
||||
"""Return plain text from a possibly-HTML field (e.g. kararKonusu)."""
|
||||
if not value:
|
||||
return ""
|
||||
text = BeautifulSoup(html_module.unescape(value), "html.parser").get_text(" ", strip=True)
|
||||
return re.sub(r"\s+", " ", text).strip()
|
||||
|
||||
|
||||
def convert_icerik_to_markdown(icerik_html: Optional[str]) -> Optional[str]:
|
||||
"""Convert the "icerik" decision HTML returned by the KBB API to Markdown.
|
||||
|
||||
The icerik field is a self-contained HTML fragment (the rendered decision
|
||||
body). Scripts/styles are stripped before handing it to MarkItDown.
|
||||
"""
|
||||
if not icerik_html:
|
||||
return None
|
||||
|
||||
processed_html = html_module.unescape(icerik_html)
|
||||
soup = BeautifulSoup(processed_html, "html.parser")
|
||||
for tag in soup.find_all(["script", "style"]):
|
||||
tag.decompose()
|
||||
|
||||
body = soup.find("body")
|
||||
html_fragment = str(body) if body else str(soup)
|
||||
if not html_fragment.strip().lower().startswith(("<html", "<!doctype")):
|
||||
html_fragment = f'<html><head><meta charset="UTF-8"></head><body>{html_fragment}</body></html>'
|
||||
|
||||
try:
|
||||
html_stream = io.BytesIO(html_fragment.encode("utf-8"))
|
||||
conversion_result = MarkItDown().convert(html_stream)
|
||||
return conversion_result.text_content
|
||||
except Exception as e: # pragma: no cover - defensive
|
||||
logger.error("AnayasaApiClient: MarkItDown conversion error: %s", e)
|
||||
return None
|
||||
|
||||
# kararTipi discriminator values accepted by the API.
|
||||
KARAR_TIPI_NORM = "NormDenetimi"
|
||||
KARAR_TIPI_BIREYSEL = "BireyselBasvuru"
|
||||
|
||||
NORM_HOST = "https://normkararlarbilgibankasi.anayasa.gov.tr"
|
||||
BIREYSEL_HOST = "https://kararlarbilgibankasi.anayasa.gov.tr"
|
||||
SEARCH_PATH = "/api/core/public/search"
|
||||
|
||||
# Map kararTipi -> the host whose SPA can display the decision (cosmetic only;
|
||||
# either host's API answers for any kararTipi).
|
||||
_HOST_FOR_TIPI = {
|
||||
KARAR_TIPI_NORM: NORM_HOST,
|
||||
KARAR_TIPI_BIREYSEL: BIREYSEL_HOST,
|
||||
}
|
||||
|
||||
|
||||
def encode_document_token(uuid: str) -> str:
|
||||
"""Encode a raw decision UUID into the base64url token the SPA uses in its URLs.
|
||||
|
||||
The SPA addresses decisions as base64url("kbb:" + uuid) (no padding).
|
||||
"""
|
||||
raw = f"kbb:{uuid}".encode("utf-8")
|
||||
return base64.urlsafe_b64encode(raw).decode("ascii").rstrip("=")
|
||||
|
||||
|
||||
def decode_document_token(token: str) -> Optional[str]:
|
||||
"""Decode a base64url SPA token back into the raw decision UUID.
|
||||
|
||||
Returns None if the token is not a valid "kbb:<uuid>" token.
|
||||
"""
|
||||
try:
|
||||
padded = token + "=" * (-len(token) % 4)
|
||||
decoded = base64.urlsafe_b64decode(padded.encode("ascii")).decode("utf-8")
|
||||
except Exception:
|
||||
return None
|
||||
if decoded.startswith("kbb:"):
|
||||
return decoded[len("kbb:"):]
|
||||
return None
|
||||
|
||||
|
||||
def build_document_url(karar_tipi: str, uuid: str) -> str:
|
||||
"""Build a clickable SPA URL for a decision, used as its document_url."""
|
||||
host = _HOST_FOR_TIPI.get(karar_tipi, BIREYSEL_HOST)
|
||||
token = encode_document_token(uuid)
|
||||
return f"{host}/kbb/pages/search/{karar_tipi}?id={quote(token)}&type={karar_tipi}"
|
||||
|
||||
|
||||
def parse_document_url(document_url: str) -> Tuple[Optional[str], Optional[str]]:
|
||||
"""Extract (karar_tipi, uuid) from a document URL.
|
||||
|
||||
Handles the new SPA URLs (?id=<token>&type=<kararTipi>) and is lenient about
|
||||
older /ND/ and /BB/ style paths so historical references still resolve.
|
||||
Returns (None, None) if neither the type nor id can be determined.
|
||||
"""
|
||||
parsed = urlparse(document_url)
|
||||
qs = parse_qs(parsed.query)
|
||||
|
||||
karar_tipi = None
|
||||
type_param = qs.get("type", [None])[0]
|
||||
path = parsed.path or ""
|
||||
if type_param in (KARAR_TIPI_NORM, KARAR_TIPI_BIREYSEL):
|
||||
karar_tipi = type_param
|
||||
elif "/ND/" in path or "NormDenetimi" in path:
|
||||
karar_tipi = KARAR_TIPI_NORM
|
||||
elif "/BB/" in path or "BireyselBasvuru" in path:
|
||||
karar_tipi = KARAR_TIPI_BIREYSEL
|
||||
|
||||
uuid = None
|
||||
id_param = qs.get("id", [None])[0]
|
||||
if id_param:
|
||||
# The id may be the raw uuid or the base64url SPA token.
|
||||
uuid = decode_document_token(id_param) or id_param
|
||||
|
||||
return karar_tipi, uuid
|
||||
|
||||
|
||||
class AnayasaApiClient:
|
||||
"""Thin async wrapper around the KBB /api/core/public/search endpoint."""
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Content-Type": "application/json",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True,
|
||||
)
|
||||
|
||||
def _search_url(self, karar_tipi: str) -> str:
|
||||
host = _HOST_FOR_TIPI.get(karar_tipi, BIREYSEL_HOST)
|
||||
return f"{host}{SEARCH_PATH}"
|
||||
|
||||
async def search(
|
||||
self,
|
||||
karar_tipi: str,
|
||||
query: str = "",
|
||||
page: int = 1,
|
||||
size: int = 10,
|
||||
) -> Dict[str, Any]:
|
||||
"""Run a list search and return the parsed JSON envelope.
|
||||
|
||||
Envelope shape: {"total": int, "page": int, "data": [..], "page_size": int}.
|
||||
"""
|
||||
body: Dict[str, Any] = {"kararTipi": karar_tipi, "page": page, "size": size}
|
||||
if query:
|
||||
body["query"] = query
|
||||
logger.info("AnayasaApiClient: search kararTipi=%s query=%r page=%s size=%s",
|
||||
karar_tipi, query, page, size)
|
||||
response = await self.http_client.post(self._search_url(karar_tipi), json=body)
|
||||
response.raise_for_status()
|
||||
return response.json()
|
||||
|
||||
async def get_decision(self, karar_tipi: str, uuid: str) -> Optional[Dict[str, Any]]:
|
||||
"""Fetch a single decision record (including the "icerik" HTML) by UUID."""
|
||||
body = {"kararTipi": karar_tipi, "id": uuid, "page": 1, "size": 1}
|
||||
logger.info("AnayasaApiClient: get_decision kararTipi=%s id=%s", karar_tipi, uuid)
|
||||
response = await self.http_client.post(self._search_url(karar_tipi), json=body)
|
||||
response.raise_for_status()
|
||||
payload = response.json()
|
||||
data = payload.get("data") or []
|
||||
return data[0] if data else None
|
||||
|
||||
async def close(self):
|
||||
if self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("AnayasaApiClient: HTTP client session closed.")
|
||||
@@ -1,24 +1,27 @@
|
||||
# anayasa_mcp_module/bireysel_client.py
|
||||
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
|
||||
# Bireysel Başvuru client backed by the new KBB JSON API (see api_client.py).
|
||||
#
|
||||
# Same backend as Norm Denetimi, distinguished by kararTipi="BireyselBasvuru".
|
||||
# The legacy /Ara report-scraping endpoint was retired and now returns HTTP 404.
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup, Tag
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
import math
|
||||
from typing import List, Optional
|
||||
|
||||
from .api_client import (
|
||||
AnayasaApiClient,
|
||||
KARAR_TIPI_BIREYSEL,
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE,
|
||||
build_document_url,
|
||||
parse_document_url,
|
||||
convert_icerik_to_markdown,
|
||||
strip_html_text,
|
||||
)
|
||||
from .models import (
|
||||
AnayasaBireyselReportSearchRequest,
|
||||
AnayasaBireyselReportDecisionDetail,
|
||||
AnayasaBireyselReportDecisionSummary,
|
||||
AnayasaBireyselReportSearchResult,
|
||||
AnayasaBireyselBasvuruDocumentMarkdown, # Model for Bireysel Başvuru document
|
||||
AnayasaBireyselBasvuruDocumentMarkdown,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -27,330 +30,93 @@ if not logger.hasHandlers():
|
||||
|
||||
|
||||
class AnayasaBireyselBasvuruApiClient:
|
||||
BASE_URL = "https://kararlarbilgibankasi.anayasa.gov.tr"
|
||||
SEARCH_PATH = "/Ara"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000 # Character limit per page
|
||||
"""Bireysel Başvuru search/document client over the KBB JSON API."""
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True
|
||||
)
|
||||
|
||||
def _build_query_params_for_bireysel_report(self, params: AnayasaBireyselReportSearchRequest) -> List[Tuple[str, str]]:
|
||||
query_params: List[Tuple[str, str]] = []
|
||||
query_params.append(("KararBulteni", "1")) # Specific to this report type
|
||||
|
||||
if params.keywords:
|
||||
for kw in params.keywords:
|
||||
query_params.append(("KelimeAra[]", kw))
|
||||
|
||||
if params.page_to_fetch and params.page_to_fetch > 1:
|
||||
query_params.append(("page", str(params.page_to_fetch)))
|
||||
|
||||
return query_params
|
||||
self.api = AnayasaApiClient(request_timeout)
|
||||
|
||||
async def search_bireysel_basvuru_report(
|
||||
self,
|
||||
params: AnayasaBireyselReportSearchRequest
|
||||
params: AnayasaBireyselReportSearchRequest,
|
||||
) -> AnayasaBireyselReportSearchResult:
|
||||
final_query_params = self._build_query_params_for_bireysel_report(params)
|
||||
request_url = self.SEARCH_PATH
|
||||
query = " ".join(t for t in (params.keywords or []) if t).strip()
|
||||
payload = await self.api.search(
|
||||
karar_tipi=KARAR_TIPI_BIREYSEL,
|
||||
query=query,
|
||||
page=params.page_to_fetch,
|
||||
size=getattr(params, "results_per_page", 10),
|
||||
)
|
||||
|
||||
logger.info(f"AnayasaBireyselBasvuruApiClient: Performing Bireysel Başvuru Report search. Path: {request_url}, Params: {final_query_params}")
|
||||
|
||||
try:
|
||||
response = await self.http_client.get(request_url, params=final_query_params)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"AnayasaBireyselBasvuruApiClient: HTTP request error during Bireysel Başvuru Report search: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaBireyselBasvuruApiClient: Error processing Bireysel Başvuru Report search request: {e}")
|
||||
raise
|
||||
|
||||
soup = BeautifulSoup(html_content, 'html.parser')
|
||||
|
||||
total_records = None
|
||||
bulunan_karar_div = soup.find("div", class_="bulunankararsayisi")
|
||||
if bulunan_karar_div:
|
||||
match_records = re.search(r'(\d+)\s*Karar Bulundu', bulunan_karar_div.get_text(strip=True))
|
||||
if match_records:
|
||||
total_records = int(match_records.group(1))
|
||||
|
||||
processed_decisions: List[AnayasaBireyselReportDecisionSummary] = []
|
||||
|
||||
report_content_area = soup.find("div", class_="HaberBulteni")
|
||||
if not report_content_area:
|
||||
logger.warning("HaberBulteni div not found, attempting to parse decision divs from the whole page.")
|
||||
report_content_area = soup
|
||||
|
||||
decision_divs = report_content_area.find_all("div", class_="KararBulteniBirKarar")
|
||||
if not decision_divs:
|
||||
logger.warning("No KararBulteniBirKarar divs found.")
|
||||
|
||||
|
||||
for decision_div in decision_divs:
|
||||
title_tag = decision_div.find("h4")
|
||||
title_text = title_tag.get_text(strip=True) if title_tag and title_tag.strong else (title_tag.get_text(strip=True) if title_tag else None)
|
||||
|
||||
|
||||
alti_cizili_div = decision_div.find("div", class_="AltiCizili")
|
||||
ref_no, dec_type, body, app_date, dec_date, url_path = None, None, None, None, None, None
|
||||
if alti_cizili_div:
|
||||
link_tag = alti_cizili_div.find("a", href=True)
|
||||
if link_tag:
|
||||
ref_no = link_tag.get_text(strip=True)
|
||||
url_path = link_tag['href']
|
||||
|
||||
parts_text = alti_cizili_div.get_text(separator="|", strip=True)
|
||||
parts = [part.strip() for part in parts_text.split("|")]
|
||||
|
||||
# Clean ref_no from the first part if it was extracted from link
|
||||
if ref_no and parts and parts[0].strip().startswith(ref_no):
|
||||
parts[0] = parts[0].replace(ref_no, "").strip()
|
||||
if not parts[0]: parts.pop(0) # Remove empty string if ref_no was the only content
|
||||
|
||||
# Assign parts based on typical order, adjusting for missing ref_no at start
|
||||
current_idx = 0
|
||||
if not ref_no and len(parts) > current_idx and re.match(r"\d+/\d+", parts[current_idx]): # Check if first part is ref_no
|
||||
ref_no = parts[current_idx]
|
||||
current_idx += 1
|
||||
|
||||
dec_type = parts[current_idx] if len(parts) > current_idx else None
|
||||
current_idx += 1
|
||||
body = parts[current_idx] if len(parts) > current_idx else None
|
||||
current_idx += 1
|
||||
|
||||
app_date_raw = parts[current_idx] if len(parts) > current_idx else None
|
||||
current_idx += 1
|
||||
dec_date_raw = parts[current_idx] if len(parts) > current_idx else None
|
||||
|
||||
if app_date_raw and "Başvuru Tarihi :" in app_date_raw:
|
||||
app_date = app_date_raw.replace("Başvuru Tarihi :", "").strip()
|
||||
elif app_date_raw: # If label is missing but format matches
|
||||
app_date_match = re.search(r'(\d{1,2}/\d{1,2}/\d{4})', app_date_raw)
|
||||
if app_date_match: app_date = app_date_match.group(1)
|
||||
|
||||
|
||||
if dec_date_raw and "Karar Tarihi :" in dec_date_raw:
|
||||
dec_date = dec_date_raw.replace("Karar Tarihi :", "").strip()
|
||||
elif dec_date_raw: # If label is missing but format matches
|
||||
dec_date_match = re.search(r'(\d{1,2}/\d{1,2}/\d{4})', dec_date_raw)
|
||||
if dec_date_match: dec_date = dec_date_match.group(1)
|
||||
|
||||
|
||||
subject_div = decision_div.find(lambda tag: tag.name == 'div' and not tag.has_attr('class') and tag.get_text(strip=True).startswith("BAŞVURU KONUSU :"))
|
||||
subject_text = subject_div.get_text(strip=True).replace("BAŞVURU KONUSU :", "").strip() if subject_div else None
|
||||
|
||||
details_list: List[AnayasaBireyselReportDecisionDetail] = []
|
||||
karar_detaylari_div = decision_div.find_next_sibling("div", id="KararDetaylari") # Corrected: was KararDetaylari
|
||||
if karar_detaylari_div:
|
||||
table = karar_detaylari_div.find("table", class_="table")
|
||||
if table and table.find("tbody"):
|
||||
for row in table.find("tbody").find_all("tr"):
|
||||
cells = row.find_all("td")
|
||||
if len(cells) == 4: # Hak, Müdahale İddiası, Sonuç, Giderim
|
||||
details_list.append(AnayasaBireyselReportDecisionDetail(
|
||||
hak=cells[0].get_text(strip=True) or None,
|
||||
mudahale_iddiasi=cells[1].get_text(strip=True) or None,
|
||||
sonuc=cells[2].get_text(strip=True) or None,
|
||||
giderim=cells[3].get_text(strip=True) or None,
|
||||
))
|
||||
|
||||
full_decision_page_url = urljoin(self.BASE_URL, url_path) if url_path else None
|
||||
|
||||
processed_decisions.append(AnayasaBireyselReportDecisionSummary(
|
||||
title=title_text,
|
||||
decision_reference_no=ref_no,
|
||||
decision_page_url=full_decision_page_url,
|
||||
decision_type_summary=dec_type,
|
||||
decision_making_body=body,
|
||||
application_date_summary=app_date,
|
||||
decision_date_summary=dec_date,
|
||||
application_subject_summary=subject_text,
|
||||
details=details_list
|
||||
total_records = int(payload.get("total") or 0)
|
||||
decisions: List[AnayasaBireyselReportDecisionSummary] = []
|
||||
for item in payload.get("data") or []:
|
||||
decisions.append(AnayasaBireyselReportDecisionSummary(
|
||||
title=item.get("basvuruAdi") or "",
|
||||
decision_reference_no=item.get("basvuruNo") or "",
|
||||
decision_page_url=build_document_url(KARAR_TIPI_BIREYSEL, item.get("id", "")),
|
||||
decision_type_summary=item.get("kararTuruBasvuruSonucuLabel") or "",
|
||||
decision_making_body=item.get("kararVerenBirimLabel") or "",
|
||||
application_date_summary=item.get("basvuruTarihi") or "",
|
||||
decision_date_summary=item.get("kararTarihi") or "",
|
||||
application_subject_summary=strip_html_text(item.get("kararKonusu")),
|
||||
details=[],
|
||||
))
|
||||
|
||||
return AnayasaBireyselReportSearchResult(
|
||||
decisions=processed_decisions,
|
||||
decisions=decisions,
|
||||
total_records_found=total_records,
|
||||
retrieved_page_number=params.page_to_fetch
|
||||
retrieved_page_number=params.page_to_fetch,
|
||||
)
|
||||
|
||||
def _convert_html_to_markdown_bireysel(self, full_decision_html_content: str) -> Optional[str]:
|
||||
if not full_decision_html_content:
|
||||
return None
|
||||
|
||||
processed_html = html.unescape(full_decision_html_content)
|
||||
soup = BeautifulSoup(processed_html, "html.parser")
|
||||
html_input_for_markdown = ""
|
||||
|
||||
karar_tab_content = soup.find("div", id="Karar")
|
||||
if karar_tab_content:
|
||||
karar_html_span = karar_tab_content.find("span", class_="kararHtml")
|
||||
if karar_html_span:
|
||||
word_section = karar_html_span.find("div", class_="WordSection1")
|
||||
if word_section:
|
||||
for s in word_section.select('script, style, .item.col-xs-12.col-sm-12, center:has(b)'):
|
||||
s.decompose()
|
||||
html_input_for_markdown = str(word_section)
|
||||
else:
|
||||
logger.warning("AnayasaBireyselBasvuruApiClient: WordSection1 not found in span.kararHtml. Using span.kararHtml content.")
|
||||
for s in karar_html_span.select('script, style, .item.col-xs-12.col-sm-12, center:has(b)'):
|
||||
s.decompose()
|
||||
html_input_for_markdown = str(karar_html_span)
|
||||
else:
|
||||
logger.warning("AnayasaBireyselBasvuruApiClient: span.kararHtml not found in div#Karar. Using div#Karar content.")
|
||||
for s in karar_tab_content.select('script, style, .item.col-xs-12.col-sm-12, center:has(b)'):
|
||||
s.decompose()
|
||||
html_input_for_markdown = str(karar_tab_content)
|
||||
else:
|
||||
logger.warning("AnayasaBireyselBasvuruApiClient: div#Karar (KARAR tab) not found. Trying WordSection1 fallback.")
|
||||
word_section_fallback = soup.find("div", class_="WordSection1")
|
||||
if word_section_fallback:
|
||||
for s in word_section_fallback.select('script, style, .item.col-xs-12.col-sm-12, center:has(b)'):
|
||||
s.decompose()
|
||||
html_input_for_markdown = str(word_section_fallback)
|
||||
else:
|
||||
body_tag = soup.find("body")
|
||||
if body_tag:
|
||||
for s in body_tag.select('script, style, .item.col-xs-12.col-sm-12, center:has(b), .banner, .footer, .yazdirmaalani, .filtreler, .menu, .altmenu, .geri, .arabuton, .temizlebutonu, form#KararGetir, .TabBaslik, #KararDetaylari, .share-button-container'):
|
||||
s.decompose()
|
||||
html_input_for_markdown = str(body_tag)
|
||||
else:
|
||||
html_input_for_markdown = processed_html
|
||||
|
||||
markdown_text = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown()
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_file:
|
||||
if not html_input_for_markdown.strip().lower().startswith(("<html", "<!doctype")):
|
||||
tmp_file.write(f"<html><head><meta charset=\"UTF-8\"></head><body>{html_input_for_markdown}</body></html>")
|
||||
else:
|
||||
tmp_file.write(html_input_for_markdown)
|
||||
temp_file_path = tmp_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
markdown_text = conversion_result.text_content
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaBireyselBasvuruApiClient: MarkItDown conversion error: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
return markdown_text
|
||||
|
||||
async def get_decision_document_as_markdown(
|
||||
self,
|
||||
document_url_path: str, # e.g. /BB/2021/20295
|
||||
page_number: int = 1
|
||||
document_url_path: str,
|
||||
page_number: int = 1,
|
||||
) -> AnayasaBireyselBasvuruDocumentMarkdown:
|
||||
full_url = urljoin(self.BASE_URL, document_url_path)
|
||||
logger.info(f"AnayasaBireyselBasvuruApiClient: Fetching Bireysel Başvuru document for Markdown (page {page_number}) from URL: {full_url}")
|
||||
karar_tipi, uuid = parse_document_url(document_url_path)
|
||||
if karar_tipi is None:
|
||||
karar_tipi = KARAR_TIPI_BIREYSEL
|
||||
|
||||
basvuru_no_from_page = None
|
||||
karar_tarihi_from_page = None
|
||||
basvuru_tarihi_from_page = None
|
||||
karari_veren_birim_from_page = None
|
||||
karar_turu_from_page = None
|
||||
resmi_gazete_info_from_page = None
|
||||
record = await self.api.get_decision(karar_tipi, uuid) if uuid else None
|
||||
|
||||
try:
|
||||
response = await self.http_client.get(full_url)
|
||||
response.raise_for_status()
|
||||
html_content_from_api = response.text
|
||||
|
||||
if not isinstance(html_content_from_api, str) or not html_content_from_api.strip():
|
||||
logger.warning(f"AnayasaBireyselBasvuruApiClient: Received empty HTML from {full_url}.")
|
||||
if not record:
|
||||
logger.warning("AnayasaBireyselBasvuruApiClient: No record for %s", document_url_path)
|
||||
return AnayasaBireyselBasvuruDocumentMarkdown(
|
||||
source_url=full_url, markdown_chunk=None, current_page=page_number, total_pages=0, is_paginated=False
|
||||
source_url=document_url_path, markdown_chunk=None,
|
||||
current_page=page_number, total_pages=0, is_paginated=False,
|
||||
)
|
||||
|
||||
soup = BeautifulSoup(html_content_from_api, 'html.parser')
|
||||
rg_tarihi = record.get("resmiGazeteTarihi") or ""
|
||||
rg_sayisi = record.get("resmiGazeteSayisi")
|
||||
official_gazette = f"{rg_tarihi} / {rg_sayisi}".strip(" /") if (rg_tarihi or rg_sayisi) else None
|
||||
|
||||
meta_desc_tag = soup.find("meta", attrs={"name": "description"})
|
||||
if meta_desc_tag and meta_desc_tag.get("content"):
|
||||
content = meta_desc_tag["content"]
|
||||
bn_match = re.search(r"B\.\s*No:\s*([\d\/]+)", content)
|
||||
if bn_match: basvuru_no_from_page = bn_match.group(1).strip()
|
||||
|
||||
date_match = re.search(r"(\d{1,2}\/\d{1,2}\/\d{4}),\s*§", content)
|
||||
if date_match: karar_tarihi_from_page = date_match.group(1).strip()
|
||||
|
||||
karar_detaylari_tab = soup.find("div", id="KararDetaylari")
|
||||
if karar_detaylari_tab:
|
||||
table = karar_detaylari_tab.find("table", class_="table")
|
||||
if table:
|
||||
rows = table.find_all("tr")
|
||||
for row in rows:
|
||||
cells = row.find_all("td")
|
||||
if len(cells) == 2:
|
||||
key = cells[0].get_text(strip=True)
|
||||
value = cells[1].get_text(strip=True)
|
||||
if "Kararı Veren Birim" in key: karari_veren_birim_from_page = value
|
||||
elif "Karar Türü (Başvuru Sonucu)" in key: karar_turu_from_page = value
|
||||
elif "Başvuru No" in key and not basvuru_no_from_page: basvuru_no_from_page = value
|
||||
elif "Başvuru Tarihi" in key: basvuru_tarihi_from_page = value
|
||||
elif "Karar Tarihi" in key and not karar_tarihi_from_page: karar_tarihi_from_page = value
|
||||
elif "Resmi Gazete Tarih / Sayı" in key: resmi_gazete_info_from_page = value
|
||||
|
||||
full_markdown_content = self._convert_html_to_markdown_bireysel(html_content_from_api)
|
||||
|
||||
if not full_markdown_content:
|
||||
return AnayasaBireyselBasvuruDocumentMarkdown(
|
||||
source_url=full_url,
|
||||
basvuru_no_from_page=basvuru_no_from_page,
|
||||
karar_tarihi_from_page=karar_tarihi_from_page,
|
||||
basvuru_tarihi_from_page=basvuru_tarihi_from_page,
|
||||
karari_veren_birim_from_page=karari_veren_birim_from_page,
|
||||
karar_turu_from_page=karar_turu_from_page,
|
||||
resmi_gazete_info_from_page=resmi_gazete_info_from_page,
|
||||
markdown_chunk=None,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False
|
||||
full_markdown = convert_icerik_to_markdown(record.get("icerik"))
|
||||
common = dict(
|
||||
source_url=document_url_path,
|
||||
basvuru_no_from_page=record.get("basvuruNo"),
|
||||
karar_tarihi_from_page=record.get("kararTarihi"),
|
||||
basvuru_tarihi_from_page=record.get("basvuruTarihi"),
|
||||
karari_veren_birim_from_page=record.get("kararVerenBirimLabel"),
|
||||
karar_turu_from_page=record.get("kararTuruBasvuruSonucuLabel"),
|
||||
resmi_gazete_info_from_page=official_gazette,
|
||||
)
|
||||
|
||||
content_length = len(full_markdown_content)
|
||||
total_pages = math.ceil(content_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
if total_pages == 0: total_pages = 1
|
||||
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
start_index = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_index = start_index + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
markdown_chunk = full_markdown_content[start_index:end_index]
|
||||
|
||||
if not full_markdown:
|
||||
return AnayasaBireyselBasvuruDocumentMarkdown(
|
||||
source_url=full_url,
|
||||
basvuru_no_from_page=basvuru_no_from_page,
|
||||
karar_tarihi_from_page=karar_tarihi_from_page,
|
||||
basvuru_tarihi_from_page=basvuru_tarihi_from_page,
|
||||
karari_veren_birim_from_page=karari_veren_birim_from_page,
|
||||
karar_turu_from_page=karar_turu_from_page,
|
||||
resmi_gazete_info_from_page=resmi_gazete_info_from_page,
|
||||
markdown_chunk=markdown_chunk,
|
||||
current_page=current_page_clamped,
|
||||
total_pages=total_pages,
|
||||
is_paginated=(total_pages > 1)
|
||||
**common, markdown_chunk=None, current_page=page_number,
|
||||
total_pages=0, is_paginated=False,
|
||||
)
|
||||
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"AnayasaBireyselBasvuruApiClient: HTTP error fetching Bireysel Başvuru document from {full_url}: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaBireyselBasvuruApiClient: General error processing Bireysel Başvuru document from {full_url}: {e}")
|
||||
raise
|
||||
total_pages = max(1, math.ceil(len(full_markdown) / DOCUMENT_MARKDOWN_CHUNK_SIZE))
|
||||
current_page = max(1, min(page_number, total_pages))
|
||||
start = (current_page - 1) * DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
chunk = full_markdown[start:start + DOCUMENT_MARKDOWN_CHUNK_SIZE]
|
||||
|
||||
return AnayasaBireyselBasvuruDocumentMarkdown(
|
||||
**common, markdown_chunk=chunk, current_page=current_page,
|
||||
total_pages=total_pages, is_paginated=(total_pages > 1),
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
if hasattr(self, 'http_client') and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
await self.api.close()
|
||||
logger.info("AnayasaBireyselBasvuruApiClient: HTTP client session closed.")
|
||||
+102
-306
@@ -1,354 +1,150 @@
|
||||
# anayasa_mcp_module/client.py
|
||||
# This client is for Norm Denetimi: https://normkararlarbilgibankasi.anayasa.gov.tr
|
||||
# Norm Denetimi client backed by the new KBB JSON API (see api_client.py).
|
||||
#
|
||||
# The Anayasa Mahkemesi sites were rebuilt as a single-page app; the old
|
||||
# HTML-scraping endpoints on normkararlarbilgibankasi.anayasa.gov.tr/Ara now
|
||||
# return HTTP 404. This client maps the rich legacy request model onto the new
|
||||
# free-text "query" search and rebuilds the legacy response models from the JSON
|
||||
# payload so existing tooling keeps working.
|
||||
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
from urllib.parse import urlencode, urljoin, quote
|
||||
from markitdown import MarkItDown
|
||||
import math # For math.ceil for pagination
|
||||
import math
|
||||
from typing import List, Optional
|
||||
|
||||
from .api_client import (
|
||||
AnayasaApiClient,
|
||||
KARAR_TIPI_NORM,
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE,
|
||||
build_document_url,
|
||||
parse_document_url,
|
||||
convert_icerik_to_markdown,
|
||||
strip_html_text,
|
||||
)
|
||||
from .models import (
|
||||
AnayasaNormDenetimiSearchRequest,
|
||||
AnayasaDecisionSummary,
|
||||
AnayasaReviewedNormInfo,
|
||||
AnayasaSearchResult,
|
||||
AnayasaDocumentMarkdown, # Model for Norm Denetimi document
|
||||
AnayasaDocumentMarkdown,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
||||
|
||||
|
||||
def _build_query(params: AnayasaNormDenetimiSearchRequest) -> str:
|
||||
"""Derive the free-text query string the new API expects from the legacy model.
|
||||
|
||||
The new endpoint only supports a single full-text "query" field, so the
|
||||
keyword lists are flattened. Esas/Karar numbers are appended when no keyword
|
||||
is provided so number-based lookups still return results.
|
||||
"""
|
||||
terms: List[str] = []
|
||||
for bucket in (params.keywords_all, params.keywords_any):
|
||||
if bucket:
|
||||
terms.extend(t for t in bucket if t)
|
||||
if not terms:
|
||||
for value in (params.case_number_esas, params.decision_number_karar):
|
||||
if value:
|
||||
terms.append(value)
|
||||
return " ".join(terms).strip()
|
||||
|
||||
|
||||
class AnayasaMahkemesiApiClient:
|
||||
BASE_URL = "https://normkararlarbilgibankasi.anayasa.gov.tr"
|
||||
SEARCH_PATH_SEGMENT = "Ara"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000 # Character limit per page
|
||||
"""Norm Denetimi search/document client over the KBB JSON API."""
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True
|
||||
)
|
||||
|
||||
def _build_search_query_params_for_aym(self, params: AnayasaNormDenetimiSearchRequest) -> List[Tuple[str, str]]:
|
||||
query_params: List[Tuple[str, str]] = []
|
||||
if params.keywords_all:
|
||||
for kw in params.keywords_all: query_params.append(("KelimeAra[]", kw))
|
||||
if params.keywords_any:
|
||||
for kw in params.keywords_any: query_params.append(("HerhangiBirKelimeAra[]", kw))
|
||||
if params.keywords_exclude:
|
||||
for kw in params.keywords_exclude: query_params.append(("BulunmayanKelimeAra[]", kw))
|
||||
if params.period and params.period.value and params.period.value != "ALL": query_params.append(("Donemler_id", params.period.value))
|
||||
if params.case_number_esas: query_params.append(("EsasNo", params.case_number_esas))
|
||||
if params.decision_number_karar: query_params.append(("KararNo", params.decision_number_karar))
|
||||
if params.first_review_date_start: query_params.append(("IlkIncelemeTarihiIlk", params.first_review_date_start))
|
||||
if params.first_review_date_end: query_params.append(("IlkIncelemeTarihiSon", params.first_review_date_end))
|
||||
if params.decision_date_start: query_params.append(("KararTarihiIlk", params.decision_date_start))
|
||||
if params.decision_date_end: query_params.append(("KararTarihiSon", params.decision_date_end))
|
||||
if params.application_type and params.application_type.value and params.application_type.value != "ALL": query_params.append(("BasvuruTurler_id", params.application_type.value))
|
||||
if params.applicant_general_name: query_params.append(("BasvuranGeneller_id", params.applicant_general_name))
|
||||
if params.applicant_specific_name: query_params.append(("BasvuranOzeller_id", params.applicant_specific_name))
|
||||
if params.attending_members_names:
|
||||
for name in params.attending_members_names: query_params.append(("Uyeler_id[]", name))
|
||||
if params.rapporteur_name: query_params.append(("Raportorler_id", params.rapporteur_name))
|
||||
if params.norm_type and params.norm_type.value and params.norm_type.value != "ALL": query_params.append(("NormunTurler_id", params.norm_type.value))
|
||||
if params.norm_id_or_name: query_params.append(("NormunNumarasiAdlar_id", params.norm_id_or_name))
|
||||
if params.norm_article: query_params.append(("NormunMaddeNumarasi", params.norm_article))
|
||||
if params.review_outcomes:
|
||||
for outcome_enum_val in params.review_outcomes:
|
||||
if outcome_enum_val.value and outcome_enum_val.value != "ALL": query_params.append(("IncelemeTuruKararSonuclar_id[]", outcome_enum_val.value))
|
||||
if params.reason_for_final_outcome and params.reason_for_final_outcome.value and params.reason_for_final_outcome.value != "ALL":
|
||||
query_params.append(("KararSonucununGerekcesi", params.reason_for_final_outcome.value))
|
||||
if params.basis_constitution_article_numbers:
|
||||
for article_no in params.basis_constitution_article_numbers: query_params.append(("DayanakHukmu[]", article_no))
|
||||
if params.official_gazette_date_start: query_params.append(("ResmiGazeteTarihiIlk", params.official_gazette_date_start))
|
||||
if params.official_gazette_date_end: query_params.append(("ResmiGazeteTarihiSon", params.official_gazette_date_end))
|
||||
if params.official_gazette_number_start: query_params.append(("ResmiGazeteSayisiIlk", params.official_gazette_number_start))
|
||||
if params.official_gazette_number_end: query_params.append(("ResmiGazeteSayisiSon", params.official_gazette_number_end))
|
||||
if params.has_press_release and params.has_press_release.value and params.has_press_release.value != "ALL": query_params.append(("BasinDuyurusu", params.has_press_release.value))
|
||||
if params.has_dissenting_opinion and params.has_dissenting_opinion.value and params.has_dissenting_opinion.value != "ALL": query_params.append(("KarsiOy", params.has_dissenting_opinion.value))
|
||||
if params.has_different_reasoning and params.has_different_reasoning.value and params.has_different_reasoning.value != "ALL": query_params.append(("FarkliGerekce", params.has_different_reasoning.value))
|
||||
|
||||
if params.page_to_fetch and params.page_to_fetch > 1:
|
||||
query_params.append(("page", str(params.page_to_fetch)))
|
||||
return query_params
|
||||
self.api = AnayasaApiClient(request_timeout)
|
||||
|
||||
async def search_norm_denetimi_decisions(
|
||||
self,
|
||||
params: AnayasaNormDenetimiSearchRequest
|
||||
params: AnayasaNormDenetimiSearchRequest,
|
||||
) -> AnayasaSearchResult:
|
||||
path_segments = []
|
||||
if params.results_per_page and params.results_per_page != 10: # Default is 10
|
||||
path_segments.append(f"SatirSayisi/{params.results_per_page}")
|
||||
query = _build_query(params)
|
||||
payload = await self.api.search(
|
||||
karar_tipi=KARAR_TIPI_NORM,
|
||||
query=query,
|
||||
page=params.page_to_fetch,
|
||||
size=params.results_per_page,
|
||||
)
|
||||
|
||||
if params.sort_by_criteria and params.sort_by_criteria != "KararTarihi": # Default is KararTarihi
|
||||
# Ensure correct quoting for criteria that might have Turkish chars or spaces
|
||||
path_segments.append(f"Siralama/{quote(params.sort_by_criteria)}")
|
||||
|
||||
path_segments.append(self.SEARCH_PATH_SEGMENT)
|
||||
request_path = "/" + "/".join(path_segments)
|
||||
|
||||
final_query_params = self._build_search_query_params_for_aym(params)
|
||||
logger.info(f"AnayasaMahkemesiApiClient: Performing Norm Denetimi search. Path: {request_path}, Params: {final_query_params}")
|
||||
|
||||
try:
|
||||
response = await self.http_client.get(request_path, params=final_query_params)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"AnayasaMahkemesiApiClient: HTTP request error during Norm Denetimi search: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaMahkemesiApiClient: Error processing Norm Denetimi search request: {e}")
|
||||
raise
|
||||
|
||||
soup = BeautifulSoup(html_content, 'html.parser')
|
||||
|
||||
total_records = None
|
||||
bulunan_karar_div = soup.find("div", class_="bulunankararsayisi")
|
||||
if not bulunan_karar_div: # Fallback for mobile view
|
||||
bulunan_karar_div = soup.find("div", class_="bulunankararsayisiMobil")
|
||||
|
||||
if bulunan_karar_div:
|
||||
match_records = re.search(r'(\d+)\s*Karar Bulundu', bulunan_karar_div.get_text(strip=True))
|
||||
if match_records:
|
||||
total_records = int(match_records.group(1))
|
||||
|
||||
processed_decisions: List[AnayasaDecisionSummary] = []
|
||||
decision_divs = soup.find_all("div", class_="birkarar")
|
||||
|
||||
for decision_div in decision_divs:
|
||||
link_tag = decision_div.find("a", href=True)
|
||||
doc_url_path = link_tag['href'] if link_tag else None
|
||||
decision_page_url_str = urljoin(self.BASE_URL, doc_url_path) if doc_url_path else None
|
||||
|
||||
title_div = decision_div.find("div", class_="bkararbaslik")
|
||||
ek_no_text_raw = title_div.get_text(strip=True, separator=" ").replace('\xa0', ' ') if title_div else ""
|
||||
ek_no_match = re.search(r"(E\.\s*\d+/\d+\s*,\s*K\.\s*\d+/\d+)", ek_no_text_raw)
|
||||
ek_no_text = ek_no_match.group(1) if ek_no_match else ek_no_text_raw.split("Sayılı Karar")[0].strip()
|
||||
|
||||
keyword_count_div = title_div.find("div", class_="BulunanKelimeSayisi") if title_div else None
|
||||
keyword_count_text = keyword_count_div.get_text(strip=True).replace("Bulunan Kelime Sayısı", "").strip() if keyword_count_div else None
|
||||
keyword_count = int(keyword_count_text) if keyword_count_text and keyword_count_text.isdigit() else None
|
||||
|
||||
info_div = decision_div.find("div", class_="kararbilgileri")
|
||||
info_parts = [part.strip() for part in info_div.get_text(separator="|").split("|")] if info_div else []
|
||||
|
||||
app_type_summary = info_parts[0] if len(info_parts) > 0 else None
|
||||
applicant_summary = info_parts[1] if len(info_parts) > 1 else None
|
||||
outcome_summary = info_parts[2] if len(info_parts) > 2 else None
|
||||
dec_date_raw = info_parts[3] if len(info_parts) > 3 else None
|
||||
decision_date_summary = dec_date_raw.replace("Karar Tarihi:", "").strip() if dec_date_raw else None
|
||||
|
||||
reviewed_norms_list: List[AnayasaReviewedNormInfo] = []
|
||||
details_table_container = decision_div.find_next_sibling("div", class_=re.compile(r"col-sm-12")) # The details table is in a sibling div
|
||||
if details_table_container:
|
||||
details_table = details_table_container.find("table", class_="table")
|
||||
if details_table and details_table.find("tbody"):
|
||||
for row in details_table.find("tbody").find_all("tr"):
|
||||
cells = row.find_all("td")
|
||||
if len(cells) == 6:
|
||||
reviewed_norms_list.append(AnayasaReviewedNormInfo(
|
||||
norm_name_or_number=cells[0].get_text(strip=True) or None,
|
||||
article_number=cells[1].get_text(strip=True) or None,
|
||||
review_type_and_outcome=cells[2].get_text(strip=True) or None,
|
||||
outcome_reason=cells[3].get_text(strip=True) or None,
|
||||
basis_constitution_articles_cited=[a.strip() for a in cells[4].get_text(strip=True).split(',') if a.strip()] if cells[4].get_text(strip=True) else [],
|
||||
postponement_period=cells[5].get_text(strip=True) or None
|
||||
))
|
||||
|
||||
processed_decisions.append(AnayasaDecisionSummary(
|
||||
decision_reference_no=ek_no_text,
|
||||
decision_page_url=decision_page_url_str,
|
||||
keywords_found_count=keyword_count,
|
||||
application_type_summary=app_type_summary,
|
||||
applicant_summary=applicant_summary,
|
||||
decision_outcome_summary=outcome_summary,
|
||||
decision_date_summary=decision_date_summary,
|
||||
reviewed_norms=reviewed_norms_list
|
||||
total_records = int(payload.get("total") or 0)
|
||||
decisions: List[AnayasaDecisionSummary] = []
|
||||
for item in payload.get("data") or []:
|
||||
esas_no = item.get("esasNo") or ""
|
||||
karar_no = item.get("kararNo") or ""
|
||||
if esas_no and karar_no:
|
||||
reference = f"E.{esas_no}, K.{karar_no}"
|
||||
else:
|
||||
reference = esas_no or karar_no or ""
|
||||
decisions.append(AnayasaDecisionSummary(
|
||||
decision_reference_no=reference,
|
||||
decision_page_url=build_document_url(KARAR_TIPI_NORM, item.get("id", "")),
|
||||
keywords_found_count=item.get("highlightCount") or 0,
|
||||
application_type_summary=item.get("basvuruTuruLabel") or "",
|
||||
applicant_summary=item.get("basvuranGenelLabel") or "",
|
||||
decision_outcome_summary=strip_html_text(item.get("kararKonusu")),
|
||||
decision_date_summary=item.get("kararTarihi") or "",
|
||||
reviewed_norms=[],
|
||||
))
|
||||
|
||||
return AnayasaSearchResult(
|
||||
decisions=processed_decisions,
|
||||
decisions=decisions,
|
||||
total_records_found=total_records,
|
||||
retrieved_page_number=params.page_to_fetch
|
||||
retrieved_page_number=params.page_to_fetch,
|
||||
)
|
||||
|
||||
def _convert_html_to_markdown_norm_denetimi(self, full_decision_html_content: str) -> Optional[str]:
|
||||
"""Converts direct HTML content from an Anayasa Mahkemesi Norm Denetimi decision page to Markdown."""
|
||||
if not full_decision_html_content:
|
||||
return None
|
||||
|
||||
processed_html = html.unescape(full_decision_html_content)
|
||||
soup = BeautifulSoup(processed_html, "html.parser")
|
||||
html_input_for_markdown = ""
|
||||
|
||||
karar_tab_content = soup.find("div", id="Karar") # "KARAR" tab content
|
||||
if karar_tab_content:
|
||||
karar_metni_div = karar_tab_content.find("div", class_="KararMetni")
|
||||
if karar_metni_div:
|
||||
# Remove scripts and styles
|
||||
for script_tag in karar_metni_div.find_all("script"): script_tag.decompose()
|
||||
for style_tag in karar_metni_div.find_all("style"): style_tag.decompose()
|
||||
# Remove "Künye Kopyala" button and other non-content divs
|
||||
for item_div in karar_metni_div.find_all("div", class_="item col-sm-12"): item_div.decompose()
|
||||
for modal_div in karar_metni_div.find_all("div", class_="modal fade"): modal_div.decompose() # If any modals
|
||||
|
||||
word_section = karar_metni_div.find("div", class_="WordSection1")
|
||||
html_input_for_markdown = str(word_section) if word_section else str(karar_metni_div)
|
||||
else:
|
||||
html_input_for_markdown = str(karar_tab_content)
|
||||
else:
|
||||
# Fallback if specific structure is not found
|
||||
word_section_fallback = soup.find("div", class_="WordSection1")
|
||||
if word_section_fallback:
|
||||
html_input_for_markdown = str(word_section_fallback)
|
||||
else:
|
||||
# Last resort: use the whole body or the raw HTML
|
||||
body_tag = soup.find("body")
|
||||
html_input_for_markdown = str(body_tag) if body_tag else processed_html
|
||||
|
||||
markdown_text = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown()
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_file:
|
||||
# Ensure the content is wrapped in basic HTML structure if it's not already
|
||||
if not html_input_for_markdown.strip().lower().startswith(("<html", "<!doctype")):
|
||||
tmp_file.write(f"<html><head><meta charset=\"UTF-8\"></head><body>{html_input_for_markdown}</body></html>")
|
||||
else:
|
||||
tmp_file.write(html_input_for_markdown)
|
||||
temp_file_path = tmp_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
markdown_text = conversion_result.text_content
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaMahkemesiApiClient: MarkItDown conversion error: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
return markdown_text
|
||||
|
||||
async def get_decision_document_as_markdown(
|
||||
self,
|
||||
document_url: str,
|
||||
page_number: int = 1
|
||||
page_number: int = 1,
|
||||
) -> AnayasaDocumentMarkdown:
|
||||
"""
|
||||
Retrieves a specific Anayasa Mahkemesi (Norm Denetimi) decision,
|
||||
converts its content to Markdown, and returns the requested page/chunk.
|
||||
"""
|
||||
full_url = urljoin(self.BASE_URL, document_url) if not document_url.startswith("http") else document_url
|
||||
logger.info(f"AnayasaMahkemesiApiClient: Fetching Norm Denetimi document for Markdown (page {page_number}) from URL: {full_url}")
|
||||
karar_tipi, uuid = parse_document_url(document_url)
|
||||
if karar_tipi is None:
|
||||
karar_tipi = KARAR_TIPI_NORM
|
||||
|
||||
decision_ek_no_from_page = None
|
||||
decision_date_from_page = None
|
||||
official_gazette_from_page = None
|
||||
record = await self.api.get_decision(karar_tipi, uuid) if uuid else None
|
||||
|
||||
try:
|
||||
# Use a new client instance for document fetching if headers/timeout needs to be different,
|
||||
# or reuse self.http_client if settings are compatible. For now, self.http_client.
|
||||
get_response = await self.http_client.get(full_url, headers={"Accept": "text/html"})
|
||||
get_response.raise_for_status()
|
||||
html_content_from_api = get_response.text
|
||||
|
||||
if not isinstance(html_content_from_api, str) or not html_content_from_api.strip():
|
||||
logger.warning(f"AnayasaMahkemesiApiClient: Received empty or non-string HTML from URL {full_url}.")
|
||||
if not record:
|
||||
logger.warning("AnayasaMahkemesiApiClient: No record for document_url %s", document_url)
|
||||
return AnayasaDocumentMarkdown(
|
||||
source_url=full_url, markdown_chunk=None, current_page=page_number, total_pages=0, is_paginated=False
|
||||
source_url=document_url, markdown_chunk=None,
|
||||
current_page=page_number, total_pages=0, is_paginated=False,
|
||||
)
|
||||
|
||||
# Extract metadata from the page content (E.K. No, Date, RG)
|
||||
soup = BeautifulSoup(html_content_from_api, "html.parser")
|
||||
karar_metni_div = soup.find("div", class_="KararMetni") # Usually within div#Karar
|
||||
if not karar_metni_div: # Fallback if not in KararMetni
|
||||
karar_metni_div = soup.find("div", class_="WordSection1")
|
||||
esas_no = record.get("esasNo") or ""
|
||||
karar_no = record.get("kararNo") or ""
|
||||
reference = f"E.{esas_no}, K.{karar_no}" if (esas_no and karar_no) else (esas_no or karar_no or "")
|
||||
rg_tarihi = record.get("resmiGazeteTarihi") or ""
|
||||
rg_sayisi = record.get("resmiGazeteSayisi")
|
||||
official_gazette = f"{rg_tarihi} / {rg_sayisi}".strip(" /") if (rg_tarihi or rg_sayisi) else ""
|
||||
|
||||
if karar_metni_div:
|
||||
# Attempt to find E.K. No (Esas No, Karar No)
|
||||
# Norm Denetimi pages often have this in bold <p> tags directly or in the WordSection1
|
||||
# Look for patterns like "Esas No.: YYYY/NN" and "Karar No.: YYYY/NN"
|
||||
|
||||
esas_no_tag = karar_metni_div.find(lambda tag: tag.name == "p" and tag.find("b") and "Esas No.:" in tag.find("b").get_text())
|
||||
karar_no_tag = karar_metni_div.find(lambda tag: tag.name == "p" and tag.find("b") and "Karar No.:" in tag.find("b").get_text())
|
||||
karar_tarihi_tag = karar_metni_div.find(lambda tag: tag.name == "p" and tag.find("b") and "Karar tarihi:" in tag.find("b").get_text()) # Less common on Norm pages
|
||||
resmi_gazete_tag = karar_metni_div.find(lambda tag: tag.name == "p" and ("Resmî Gazete tarih ve sayısı:" in tag.get_text() or "Resmi Gazete tarih/sayı:" in tag.get_text()))
|
||||
|
||||
|
||||
if esas_no_tag and esas_no_tag.find("b") and karar_no_tag and karar_no_tag.find("b"):
|
||||
esas_str = esas_no_tag.find("b").get_text(strip=True).replace('Esas No.:', '').strip()
|
||||
karar_str = karar_no_tag.find("b").get_text(strip=True).replace('Karar No.:', '').strip()
|
||||
decision_ek_no_from_page = f"E.{esas_str}, K.{karar_str}"
|
||||
|
||||
if karar_tarihi_tag and karar_tarihi_tag.find("b"):
|
||||
decision_date_from_page = karar_tarihi_tag.find("b").get_text(strip=True).replace("Karar tarihi:", "").strip()
|
||||
elif karar_metni_div: # Fallback for Karar Tarihi if not in specific tag
|
||||
date_match = re.search(r"Karar Tarihi\s*:\s*([\d\.]+)", karar_metni_div.get_text()) # Norm pages often use DD.MM.YYYY
|
||||
if date_match: decision_date_from_page = date_match.group(1).strip()
|
||||
|
||||
|
||||
if resmi_gazete_tag:
|
||||
# Try to get the bold part first if it exists
|
||||
bold_rg_tag = resmi_gazete_tag.find("b")
|
||||
rg_text_content = bold_rg_tag.get_text(strip=True) if bold_rg_tag else resmi_gazete_tag.get_text(strip=True)
|
||||
official_gazette_from_page = rg_text_content.replace("Resmî Gazete tarih ve sayısı:", "").replace("Resmi Gazete tarih/sayı:", "").strip()
|
||||
|
||||
|
||||
full_markdown_content = self._convert_html_to_markdown_norm_denetimi(html_content_from_api)
|
||||
|
||||
if not full_markdown_content:
|
||||
full_markdown = convert_icerik_to_markdown(record.get("icerik"))
|
||||
if not full_markdown:
|
||||
return AnayasaDocumentMarkdown(
|
||||
source_url=full_url,
|
||||
decision_reference_no_from_page=decision_ek_no_from_page,
|
||||
decision_date_from_page=decision_date_from_page,
|
||||
official_gazette_info_from_page=official_gazette_from_page,
|
||||
markdown_chunk=None,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False
|
||||
source_url=document_url,
|
||||
decision_reference_no_from_page=reference,
|
||||
decision_date_from_page=record.get("kararTarihi") or "",
|
||||
official_gazette_info_from_page=official_gazette,
|
||||
markdown_chunk=None, current_page=page_number, total_pages=0, is_paginated=False,
|
||||
)
|
||||
|
||||
content_length = len(full_markdown_content)
|
||||
total_pages = math.ceil(content_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
if total_pages == 0: total_pages = 1
|
||||
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
start_index = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_index = start_index + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
markdown_chunk = full_markdown_content[start_index:end_index]
|
||||
total_pages = max(1, math.ceil(len(full_markdown) / DOCUMENT_MARKDOWN_CHUNK_SIZE))
|
||||
current_page = max(1, min(page_number, total_pages))
|
||||
start = (current_page - 1) * DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
chunk = full_markdown[start:start + DOCUMENT_MARKDOWN_CHUNK_SIZE]
|
||||
|
||||
return AnayasaDocumentMarkdown(
|
||||
source_url=full_url,
|
||||
decision_reference_no_from_page=decision_ek_no_from_page,
|
||||
decision_date_from_page=decision_date_from_page,
|
||||
official_gazette_info_from_page=official_gazette_from_page,
|
||||
markdown_chunk=markdown_chunk,
|
||||
current_page=current_page_clamped,
|
||||
source_url=document_url,
|
||||
decision_reference_no_from_page=reference,
|
||||
decision_date_from_page=record.get("kararTarihi") or "",
|
||||
official_gazette_info_from_page=official_gazette,
|
||||
markdown_chunk=chunk,
|
||||
current_page=current_page,
|
||||
total_pages=total_pages,
|
||||
is_paginated=(total_pages > 1)
|
||||
is_paginated=(total_pages > 1),
|
||||
)
|
||||
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"AnayasaMahkemesiApiClient: HTTP error fetching Norm Denetimi document from {full_url}: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"AnayasaMahkemesiApiClient: General error processing Norm Denetimi document from {full_url}: {e}")
|
||||
raise
|
||||
|
||||
async def close_client_session(self):
|
||||
if hasattr(self, 'http_client') and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
await self.api.close()
|
||||
logger.info("AnayasaMahkemesiApiClient (Norm Denetimi): HTTP client session closed.")
|
||||
@@ -1,43 +1,21 @@
|
||||
# anayasa_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Dict, Any
|
||||
from typing import List, Optional, Dict, Any, Literal
|
||||
from enum import Enum
|
||||
|
||||
# --- Enums (AnayasaDonemEnum, AnayasaBasvuruTuruEnum, etc. - same as before) ---
|
||||
# --- Enums (AnayasaDonemEnum, etc. - same as before) ---
|
||||
class AnayasaDonemEnum(str, Enum):
|
||||
TUMU = "ALL"
|
||||
DONEM_1961 = "1"
|
||||
DONEM_1982 = "2"
|
||||
|
||||
class AnayasaBasvuruTuruEnum(str, Enum):
|
||||
TUMU = "ALL"
|
||||
IPTAL = "1"
|
||||
ITIRAZ = "2"
|
||||
DIGER = "3"
|
||||
|
||||
class AnayasaVarYokEnum(str, Enum):
|
||||
TUMU = "ALL"
|
||||
YOK = "0"
|
||||
VAR = "1"
|
||||
|
||||
class AnayasaNormTuruEnum(str, Enum):
|
||||
TUMU = "ALL"
|
||||
ANAYASA = "1"
|
||||
ANAYASA_DEGISTIREN_KANUN = "2"
|
||||
CUMHURBASKANLIGI_KARARNAMESI = "14"
|
||||
ICTUZUK = "3"
|
||||
KANUN = "4"
|
||||
KANUN_HUKMUNDE_KARARNAME = "5"
|
||||
KARAR = "6"
|
||||
NIZAMNAME = "7"
|
||||
TALIMATNAME = "8"
|
||||
TARIFE = "9"
|
||||
TBMM_KARARI = "10"
|
||||
TEZKERE = "11"
|
||||
TUZUK = "12"
|
||||
YOK_SECENEGI = "0"
|
||||
YONETMELIK = "13"
|
||||
|
||||
class AnayasaIncelemeSonucuEnum(str, Enum):
|
||||
TUMU = "ALL"
|
||||
@@ -89,60 +67,60 @@ class AnayasaNormDenetimiSearchRequest(BaseModel):
|
||||
keywords_all: Optional[List[str]] = Field(default_factory=list, description="Keywords for AND logic (KelimeAra[]).")
|
||||
keywords_any: Optional[List[str]] = Field(default_factory=list, description="Keywords for OR logic (HerhangiBirKelimeAra[]).")
|
||||
keywords_exclude: Optional[List[str]] = Field(default_factory=list, description="Keywords to exclude (BulunmayanKelimeAra[]).")
|
||||
period: Optional[AnayasaDonemEnum] = Field(default=AnayasaDonemEnum.TUMU, description="Constitutional period (Donemler_id).")
|
||||
case_number_esas: Optional[str] = Field(None, description="Case registry number (EsasNo), e.g., '2023/123'.")
|
||||
decision_number_karar: Optional[str] = Field(None, description="Decision number (KararNo), e.g., '2023/456'.")
|
||||
first_review_date_start: Optional[str] = Field(None, description="First review start date (IlkIncelemeTarihiIlk), format DD/MM/YYYY.")
|
||||
first_review_date_end: Optional[str] = Field(None, description="First review end date (IlkIncelemeTarihiSon), format DD/MM/YYYY.")
|
||||
decision_date_start: Optional[str] = Field(None, description="Decision start date (KararTarihiIlk), format DD/MM/YYYY.")
|
||||
decision_date_end: Optional[str] = Field(None, description="Decision end date (KararTarihiSon), format DD/MM/YYYY.")
|
||||
application_type: Optional[AnayasaBasvuruTuruEnum] = Field(default=AnayasaBasvuruTuruEnum.TUMU, description="Type of application (BasvuruTurler_id).")
|
||||
applicant_general_name: Optional[str] = Field(None, description="General applicant name (BasvuranGeneller_id).")
|
||||
applicant_specific_name: Optional[str] = Field(None, description="Specific applicant name (BasvuranOzeller_id).")
|
||||
official_gazette_date_start: Optional[str] = Field(None, description="Official Gazette start date (ResmiGazeteTarihiIlk), format DD/MM/YYYY.")
|
||||
official_gazette_date_end: Optional[str] = Field(None, description="Official Gazette end date (ResmiGazeteTarihiSon), format DD/MM/YYYY.")
|
||||
official_gazette_number_start: Optional[str] = Field(None, description="Official Gazette starting number (ResmiGazeteSayisiIlk).")
|
||||
official_gazette_number_end: Optional[str] = Field(None, description="Official Gazette ending number (ResmiGazeteSayisiSon).")
|
||||
has_press_release: Optional[AnayasaVarYokEnum] = Field(default=AnayasaVarYokEnum.TUMU, description="Press release available (BasinDuyurusu).")
|
||||
has_dissenting_opinion: Optional[AnayasaVarYokEnum] = Field(default=AnayasaVarYokEnum.TUMU, description="Dissenting opinion exists (KarsiOy).")
|
||||
has_different_reasoning: Optional[AnayasaVarYokEnum] = Field(default=AnayasaVarYokEnum.TUMU, description="Different reasoning exists (FarkliGerekce).")
|
||||
period: Optional[Literal["ALL", "1", "2"]] = Field(default="ALL", description="Constitutional period (Donemler_id).")
|
||||
case_number_esas: str = Field("", description="Case registry number (EsasNo), e.g., '2023/123'.")
|
||||
decision_number_karar: str = Field("", description="Decision number (KararNo), e.g., '2023/456'.")
|
||||
first_review_date_start: str = Field("", description="First review start date (IlkIncelemeTarihiIlk), format DD/MM/YYYY.")
|
||||
first_review_date_end: str = Field("", description="First review end date (IlkIncelemeTarihiSon), format DD/MM/YYYY.")
|
||||
decision_date_start: str = Field("", description="Decision start date (KararTarihiIlk), format DD/MM/YYYY.")
|
||||
decision_date_end: str = Field("", description="Decision end date (KararTarihiSon), format DD/MM/YYYY.")
|
||||
application_type: Optional[Literal["ALL", "1", "2", "3"]] = Field(default="ALL", description="Type of application (BasvuruTurler_id).")
|
||||
applicant_general_name: str = Field("", description="General applicant name (BasvuranGeneller_id).")
|
||||
applicant_specific_name: str = Field("", description="Specific applicant name (BasvuranOzeller_id).")
|
||||
official_gazette_date_start: str = Field("", description="Official Gazette start date (ResmiGazeteTarihiIlk), format DD/MM/YYYY.")
|
||||
official_gazette_date_end: str = Field("", description="Official Gazette end date (ResmiGazeteTarihiSon), format DD/MM/YYYY.")
|
||||
official_gazette_number_start: str = Field("", description="Official Gazette starting number (ResmiGazeteSayisiIlk).")
|
||||
official_gazette_number_end: str = Field("", description="Official Gazette ending number (ResmiGazeteSayisiSon).")
|
||||
has_press_release: Optional[Literal["ALL", "0", "1"]] = Field(default="ALL", description="Press release available (BasinDuyurusu).")
|
||||
has_dissenting_opinion: Optional[Literal["ALL", "0", "1"]] = Field(default="ALL", description="Dissenting opinion exists (KarsiOy).")
|
||||
has_different_reasoning: Optional[Literal["ALL", "0", "1"]] = Field(default="ALL", description="Different reasoning exists (FarkliGerekce).")
|
||||
attending_members_names: Optional[List[str]] = Field(default_factory=list, description="List of attending members' exact names (Uyeler_id[]).")
|
||||
rapporteur_name: Optional[str] = Field(None, description="Rapporteur's exact name (Raportorler_id).")
|
||||
norm_type: Optional[AnayasaNormTuruEnum] = Field(default=AnayasaNormTuruEnum.TUMU, description="Type of the reviewed norm (NormunTurler_id).")
|
||||
norm_id_or_name: Optional[str] = Field(None, description="Number or name of the norm (NormunNumarasiAdlar_id).")
|
||||
norm_article: Optional[str] = Field(None, description="Article number of the norm (NormunMaddeNumarasi).")
|
||||
review_outcomes: Optional[List[AnayasaIncelemeSonucuEnum]] = Field(default_factory=list, description="List of review types and outcomes (IncelemeTuruKararSonuclar_id[]).")
|
||||
reason_for_final_outcome: Optional[AnayasaSonucGerekcesiEnum] = Field(default=AnayasaSonucGerekcesiEnum.TUMU, description="Main reason for the decision outcome (KararSonucununGerekcesi).")
|
||||
rapporteur_name: str = Field("", description="Rapporteur's exact name (Raportorler_id).")
|
||||
norm_type: Optional[Literal["ALL", "1", "2", "3", "4", "5", "6", "7", "8", "9", "10", "11", "12", "13", "14", "0"]] = Field(default="ALL", description="Type of the reviewed norm (NormunTurler_id).")
|
||||
norm_id_or_name: str = Field("", description="Number or name of the norm (NormunNumarasiAdlar_id).")
|
||||
norm_article: str = Field("", description="Article number of the norm (NormunMaddeNumarasi).")
|
||||
review_outcomes: Optional[List[Literal["1", "2", "3", "4", "5", "6", "7", "8", "12"]]] = Field(default_factory=list, description="List of review types and outcomes (IncelemeTuruKararSonuclar_id[]).")
|
||||
reason_for_final_outcome: Optional[Literal["ALL", "1", "2", "3", "4", "5", "6", "7", "8", "9", "10", "11", "12", "13", "14", "15", "16", "17", "18", "19", "20", "21", "22", "23", "24", "25", "26", "27", "29", "30"]] = Field(default="ALL", description="Main reason for the decision outcome (KararSonucununGerekcesi).")
|
||||
basis_constitution_article_numbers: Optional[List[str]] = Field(default_factory=list, description="List of supporting Constitution article numbers (DayanakHukmu[]).")
|
||||
results_per_page: Optional[int] = Field(10, description="Number of results per page. Options: 10, 20, 30, 40, 50.")
|
||||
page_to_fetch: Optional[int] = Field(1, ge=1, description="Page number to fetch for results list.")
|
||||
sort_by_criteria: Optional[str] = Field("KararTarihi", description="Sort criteria. Options: 'KararTarihi', 'YayinTarihi', 'Toplam' (keyword count).")
|
||||
results_per_page: int = Field(10, ge=1, le=10, description="Results per page.")
|
||||
page_to_fetch: int = Field(1, ge=1, description="Page number to fetch for results list.")
|
||||
sort_by_criteria: str = Field("KararTarihi", description="Sort criteria. Options: 'KararTarihi', 'YayinTarihi', 'Toplam' (keyword count).")
|
||||
|
||||
class AnayasaReviewedNormInfo(BaseModel):
|
||||
"""Details of a norm reviewed within an AYM decision summary."""
|
||||
norm_name_or_number: Optional[str] = None
|
||||
article_number: Optional[str] = None
|
||||
review_type_and_outcome: Optional[str] = None
|
||||
outcome_reason: Optional[str] = None
|
||||
norm_name_or_number: str = Field("", description="Norm name or number")
|
||||
article_number: str = Field("", description="Article number")
|
||||
review_type_and_outcome: str = Field("", description="Review type and outcome")
|
||||
outcome_reason: str = Field("", description="Outcome reason")
|
||||
basis_constitution_articles_cited: List[str] = Field(default_factory=list)
|
||||
postponement_period: Optional[str] = None
|
||||
postponement_period: str = Field("", description="Postponement period")
|
||||
|
||||
class AnayasaDecisionSummary(BaseModel):
|
||||
"""Model for a single Anayasa Mahkemesi (Norm Denetimi) decision summary from search results."""
|
||||
decision_reference_no: Optional[str] = None
|
||||
decision_page_url: Optional[HttpUrl] = None
|
||||
keywords_found_count: Optional[int] = None
|
||||
application_type_summary: Optional[str] = None
|
||||
applicant_summary: Optional[str] = None
|
||||
decision_outcome_summary: Optional[str] = None
|
||||
decision_date_summary: Optional[str] = None
|
||||
decision_reference_no: str = Field("", description="Decision reference number")
|
||||
decision_page_url: str = Field("", description="Decision page URL")
|
||||
keywords_found_count: Optional[int] = Field(0, description="Keywords found count")
|
||||
application_type_summary: str = Field("", description="Application type summary")
|
||||
applicant_summary: str = Field("", description="Applicant summary")
|
||||
decision_outcome_summary: str = Field("", description="Decision outcome summary")
|
||||
decision_date_summary: str = Field("", description="Decision date summary")
|
||||
reviewed_norms: List[AnayasaReviewedNormInfo] = Field(default_factory=list)
|
||||
|
||||
class AnayasaSearchResult(BaseModel):
|
||||
"""Model for the overall search result for Anayasa Mahkemesi Norm Denetimi decisions."""
|
||||
decisions: List[AnayasaDecisionSummary]
|
||||
total_records_found: Optional[int] = None
|
||||
retrieved_page_number: Optional[int] = None
|
||||
total_records_found: int = Field(0, description="Total records found")
|
||||
retrieved_page_number: int = Field(1, description="Retrieved page number")
|
||||
|
||||
class AnayasaDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
@@ -150,10 +128,10 @@ class AnayasaDocumentMarkdown(BaseModel):
|
||||
and pagination information.
|
||||
"""
|
||||
source_url: HttpUrl
|
||||
decision_reference_no_from_page: Optional[str] = Field(None, description="E.K. No parsed from the document page.")
|
||||
decision_date_from_page: Optional[str] = Field(None, description="Decision date parsed from the document page.")
|
||||
official_gazette_info_from_page: Optional[str] = Field(None, description="Official Gazette info parsed from the document page.")
|
||||
markdown_chunk: Optional[str] = Field(None, description="A 5,000 character chunk of the Markdown content.") # Corrected chunk size
|
||||
decision_reference_no_from_page: str = Field("", description="E.K. No parsed from the document page.")
|
||||
decision_date_from_page: str = Field("", description="Decision date parsed from the document page.")
|
||||
official_gazette_info_from_page: str = Field("", description="Official Gazette info parsed from the document page.")
|
||||
markdown_chunk: str = Field("", description="A 5,000 character chunk of the Markdown content.") # Corrected chunk size
|
||||
current_page: int = Field(description="The current page number of the markdown chunk (1-indexed).")
|
||||
total_pages: int = Field(description="Total number of pages for the full markdown content.")
|
||||
is_paginated: bool = Field(description="True if the full markdown content is split into multiple pages.")
|
||||
@@ -162,33 +140,34 @@ class AnayasaDocumentMarkdown(BaseModel):
|
||||
# --- Models for Anayasa Mahkemesi - Bireysel Başvuru Karar Raporu ---
|
||||
|
||||
class AnayasaBireyselReportSearchRequest(BaseModel):
|
||||
"""Model for Anayasa Mahkemesi (Bireysel Başvuru) 'Karar Arama Raporu' search request."""
|
||||
keywords: Optional[List[str]] = Field(default_factory=list, description="Keywords for AND logic (KelimeAra[]).")
|
||||
"""Model for Anayasa Mahkemesi (Bireysel Başvuru) search request."""
|
||||
keywords: Optional[List[str]] = Field(default_factory=list, description="Keywords joined into the full-text query.")
|
||||
page_to_fetch: int = Field(1, ge=1, description="Page number to fetch for the report (page). Default is 1.")
|
||||
results_per_page: int = Field(10, ge=1, le=100, description="Results per page.")
|
||||
|
||||
class AnayasaBireyselReportDecisionDetail(BaseModel):
|
||||
"""Details of a specific right/claim within a Bireysel Başvuru decision summary in a report."""
|
||||
hak: Optional[str] = Field(None, description="İhlal edildiği iddia edilen hak (örneğin, Mülkiyet hakkı).")
|
||||
mudahale_iddiasi: Optional[str] = Field(None, description="İhlale neden olan müdahale iddiası.")
|
||||
sonuc: Optional[str] = Field(None, description="İnceleme sonucu (örneğin, İhlal, Düşme).")
|
||||
giderim: Optional[str] = Field(None, description="Kararlaştırılan giderim (örneğin, Yeniden yargılama).")
|
||||
hak: str = Field("", description="İhlal edildiği iddia edilen hak (örneğin, Mülkiyet hakkı).")
|
||||
mudahale_iddiasi: str = Field("", description="İhlale neden olan müdahale iddiası.")
|
||||
sonuc: str = Field("", description="İnceleme sonucu (örneğin, İhlal, Düşme).")
|
||||
giderim: str = Field("", description="Kararlaştırılan giderim (örneğin, Yeniden yargılama).")
|
||||
|
||||
class AnayasaBireyselReportDecisionSummary(BaseModel):
|
||||
"""Model for a single Anayasa Mahkemesi (Bireysel Başvuru) decision summary from a 'Karar Arama Raporu'."""
|
||||
title: Optional[str] = Field(None, description="Başvurunun başlığı (e.g., 'HASAN DURMUŞ Başvurusuna İlişkin Karar').")
|
||||
decision_reference_no: Optional[str] = Field(None, description="Başvuru Numarası (e.g., '2019/19126').")
|
||||
decision_page_url: Optional[HttpUrl] = Field(None, description="URL to the full decision page.")
|
||||
decision_type_summary: Optional[str] = Field(None, description="Karar Türü (Başvuru Sonucu) (e.g., 'Esas (İhlal)').")
|
||||
decision_making_body: Optional[str] = Field(None, description="Kararı Veren Birim (e.g., 'Genel Kurul', 'Birinci Bölüm').")
|
||||
application_date_summary: Optional[str] = Field(None, description="Başvuru Tarihi (DD/MM/YYYY).")
|
||||
decision_date_summary: Optional[str] = Field(None, description="Karar Tarihi (DD/MM/YYYY).")
|
||||
application_subject_summary: Optional[str] = Field(None, description="Başvuru konusunun özeti.")
|
||||
title: str = Field("", description="Başvurunun başlığı (e.g., 'HASAN DURMUŞ Başvurusuna İlişkin Karar').")
|
||||
decision_reference_no: str = Field("", description="Başvuru Numarası (e.g., '2019/19126').")
|
||||
decision_page_url: str = Field("", description="URL to the full decision page.")
|
||||
decision_type_summary: str = Field("", description="Karar Türü (Başvuru Sonucu) (e.g., 'Esas (İhlal)').")
|
||||
decision_making_body: str = Field("", description="Kararı Veren Birim (e.g., 'Genel Kurul', 'Birinci Bölüm').")
|
||||
application_date_summary: str = Field("", description="Başvuru Tarihi (DD/MM/YYYY).")
|
||||
decision_date_summary: str = Field("", description="Karar Tarihi (DD/MM/YYYY).")
|
||||
application_subject_summary: str = Field("", description="Başvuru konusunun özeti.")
|
||||
details: List[AnayasaBireyselReportDecisionDetail] = Field(default_factory=list, description="İncelenen haklar ve sonuçlarına ilişkin detaylar.")
|
||||
|
||||
class AnayasaBireyselReportSearchResult(BaseModel):
|
||||
"""Model for the overall search result for Anayasa Mahkemesi 'Karar Arama Raporu'."""
|
||||
decisions: List[AnayasaBireyselReportDecisionSummary]
|
||||
total_records_found: Optional[int] = Field(None, description="Raporda bulunan toplam karar sayısı.")
|
||||
total_records_found: int = Field(0, description="Raporda bulunan toplam karar sayısı.")
|
||||
retrieved_page_number: int = Field(description="Alınan rapor sayfa numarası.")
|
||||
|
||||
|
||||
@@ -210,3 +189,36 @@ class AnayasaBireyselBasvuruDocumentMarkdown(BaseModel):
|
||||
is_paginated: bool = Field(description="True if the full markdown content is split into multiple pages.")
|
||||
|
||||
# --- End Models for Bireysel Başvuru ---
|
||||
|
||||
# --- Unified Models ---
|
||||
class AnayasaUnifiedSearchRequest(BaseModel):
|
||||
"""Unified search request for both Norm Denetimi and Bireysel Başvuru.
|
||||
|
||||
The KBB API only exposes a single free-text "query" field plus pagination,
|
||||
so the keyword lists below are flattened into that query.
|
||||
"""
|
||||
decision_type: Literal["norm_denetimi", "bireysel_basvuru"] = Field(..., description="Decision type: norm_denetimi or bireysel_basvuru")
|
||||
|
||||
# Common parameters
|
||||
keywords: List[str] = Field(default_factory=list, description="Keywords to search for (joined into a single full-text query)")
|
||||
keywords_all: List[str] = Field(default_factory=list, description="Additional keywords to include in the query")
|
||||
keywords_any: List[str] = Field(default_factory=list, description="Additional alternative keywords to include in the query")
|
||||
page_to_fetch: int = Field(1, ge=1, le=100, description="Page number to fetch (1-100)")
|
||||
results_per_page: int = Field(10, ge=1, le=100, description="Results per page (1-100)")
|
||||
|
||||
class AnayasaUnifiedSearchResult(BaseModel):
|
||||
"""Unified search result containing decisions from either system."""
|
||||
decision_type: Literal["norm_denetimi", "bireysel_basvuru"] = Field(..., description="Type of decisions returned")
|
||||
decisions: List[Dict[str, Any]] = Field(default_factory=list, description="Decision list (structure varies by type)")
|
||||
total_records_found: int = Field(0, description="Total number of records found")
|
||||
retrieved_page_number: int = Field(1, description="Page number that was retrieved")
|
||||
|
||||
class AnayasaUnifiedDocumentMarkdown(BaseModel):
|
||||
"""Unified document model for both Norm Denetimi and Bireysel Başvuru."""
|
||||
decision_type: Literal["norm_denetimi", "bireysel_basvuru"] = Field(..., description="Type of document")
|
||||
source_url: HttpUrl = Field(..., description="Source URL of the document")
|
||||
document_data: Dict[str, Any] = Field(default_factory=dict, description="Document content and metadata")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Markdown content chunk")
|
||||
current_page: int = Field(1, description="Current page number")
|
||||
total_pages: int = Field(1, description="Total number of pages")
|
||||
is_paginated: bool = Field(False, description="Whether document is paginated")
|
||||
@@ -0,0 +1,118 @@
|
||||
# anayasa_mcp_module/unified_client.py
|
||||
# Unified client for both Norm Denetimi and Bireysel Başvuru, backed by the new
|
||||
# KBB JSON API. Routing between the two is by the "decision_type" discriminator
|
||||
# on search, and by the document URL (?type=...) on document retrieval.
|
||||
|
||||
import logging
|
||||
from typing import Optional, Tuple
|
||||
|
||||
from .models import (
|
||||
AnayasaUnifiedSearchRequest,
|
||||
AnayasaUnifiedSearchResult,
|
||||
AnayasaUnifiedDocumentMarkdown,
|
||||
AnayasaNormDenetimiSearchRequest,
|
||||
AnayasaBireyselReportSearchRequest,
|
||||
)
|
||||
from .client import AnayasaMahkemesiApiClient
|
||||
from .bireysel_client import AnayasaBireyselBasvuruApiClient
|
||||
from .api_client import (
|
||||
KARAR_TIPI_NORM,
|
||||
KARAR_TIPI_BIREYSEL,
|
||||
parse_document_url,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def normalize_anayasa_document_url(document_url: str) -> Tuple[Optional[str], str]:
|
||||
"""Detect the AYM decision type from a document URL.
|
||||
|
||||
Returns ``(decision_type, document_url)`` where ``decision_type`` is
|
||||
``"norm_denetimi"``, ``"bireysel_basvuru"``, or ``None`` if it cannot be
|
||||
determined. The URL is returned unchanged (kept for backwards compatibility
|
||||
with callers that expect a possibly-normalized URL).
|
||||
"""
|
||||
karar_tipi, _ = parse_document_url(document_url)
|
||||
if karar_tipi == KARAR_TIPI_NORM:
|
||||
return "norm_denetimi", document_url
|
||||
if karar_tipi == KARAR_TIPI_BIREYSEL:
|
||||
return "bireysel_basvuru", document_url
|
||||
return None, document_url
|
||||
|
||||
|
||||
class AnayasaUnifiedClient:
|
||||
"""Unified client that handles both Norm Denetimi and Bireysel Başvuru searches."""
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.norm_client = AnayasaMahkemesiApiClient(request_timeout)
|
||||
self.bireysel_client = AnayasaBireyselBasvuruApiClient(request_timeout)
|
||||
|
||||
async def search_unified(self, params: AnayasaUnifiedSearchRequest) -> AnayasaUnifiedSearchResult:
|
||||
"""Unified search that routes to the appropriate client based on decision_type."""
|
||||
|
||||
if params.decision_type == "norm_denetimi":
|
||||
norm_params = AnayasaNormDenetimiSearchRequest(
|
||||
keywords_all=params.keywords_all or params.keywords,
|
||||
keywords_any=params.keywords_any,
|
||||
page_to_fetch=params.page_to_fetch,
|
||||
results_per_page=params.results_per_page,
|
||||
)
|
||||
result = await self.norm_client.search_norm_denetimi_decisions(norm_params)
|
||||
|
||||
return AnayasaUnifiedSearchResult(
|
||||
decision_type="norm_denetimi",
|
||||
decisions=[d.model_dump() for d in result.decisions],
|
||||
total_records_found=result.total_records_found,
|
||||
retrieved_page_number=result.retrieved_page_number,
|
||||
)
|
||||
|
||||
elif params.decision_type == "bireysel_basvuru":
|
||||
bireysel_params = AnayasaBireyselReportSearchRequest(
|
||||
keywords=params.keywords or params.keywords_all,
|
||||
page_to_fetch=params.page_to_fetch,
|
||||
results_per_page=params.results_per_page,
|
||||
)
|
||||
result = await self.bireysel_client.search_bireysel_basvuru_report(bireysel_params)
|
||||
|
||||
return AnayasaUnifiedSearchResult(
|
||||
decision_type="bireysel_basvuru",
|
||||
decisions=[d.model_dump() for d in result.decisions],
|
||||
total_records_found=result.total_records_found,
|
||||
retrieved_page_number=result.retrieved_page_number,
|
||||
)
|
||||
|
||||
raise ValueError(f"Unsupported decision type: {params.decision_type}")
|
||||
|
||||
async def get_document_unified(self, document_url: str, page_number: int = 1) -> AnayasaUnifiedDocumentMarkdown:
|
||||
"""Unified document retrieval that auto-detects the decision type from the URL."""
|
||||
|
||||
decision_type, _ = normalize_anayasa_document_url(document_url)
|
||||
|
||||
if decision_type == "bireysel_basvuru":
|
||||
result = await self.bireysel_client.get_decision_document_as_markdown(document_url, page_number)
|
||||
return AnayasaUnifiedDocumentMarkdown(
|
||||
decision_type="bireysel_basvuru",
|
||||
source_url=result.source_url,
|
||||
document_data=result.model_dump(mode="json"),
|
||||
markdown_chunk=result.markdown_chunk,
|
||||
current_page=result.current_page,
|
||||
total_pages=result.total_pages,
|
||||
is_paginated=result.is_paginated,
|
||||
)
|
||||
|
||||
# Default to norm_denetimi (also covers explicit norm_denetimi detection).
|
||||
result = await self.norm_client.get_decision_document_as_markdown(document_url, page_number)
|
||||
return AnayasaUnifiedDocumentMarkdown(
|
||||
decision_type="norm_denetimi",
|
||||
source_url=result.source_url,
|
||||
document_data=result.model_dump(mode="json"),
|
||||
markdown_chunk=result.markdown_chunk,
|
||||
current_page=result.current_page,
|
||||
total_pages=result.total_pages,
|
||||
is_paginated=result.is_paginated,
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close both client sessions."""
|
||||
await self.norm_client.close_client_session()
|
||||
await self.bireysel_client.close_client_session()
|
||||
@@ -0,0 +1,35 @@
|
||||
"""
|
||||
ASGI application for Yargı MCP Server (simple deployment variant).
|
||||
|
||||
This is a minimal ASGI application that can be run with:
|
||||
uvicorn app:app --host 0.0.0.0 --port 8000
|
||||
|
||||
The MCP server will be available at:
|
||||
http://localhost:8000/mcp/
|
||||
|
||||
For the FastAPI-wrapped variant with CORS and extra metadata routes,
|
||||
see asgi_app.py instead.
|
||||
"""
|
||||
|
||||
from starlette.responses import JSONResponse
|
||||
from mcp_server_main import create_app
|
||||
|
||||
mcp = create_app()
|
||||
|
||||
|
||||
@mcp.custom_route("/health", methods=["GET"])
|
||||
async def health_check(request):
|
||||
"""Health check endpoint for monitoring services (Fly.io, Render, etc.)."""
|
||||
return JSONResponse({
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.2.1",
|
||||
})
|
||||
|
||||
|
||||
# Create ASGI app directly from FastMCP server
|
||||
app = mcp.http_app()
|
||||
|
||||
# Endpoints:
|
||||
# - /mcp/ - MCP server (Streamable HTTP transport, default FastMCP path)
|
||||
# - /health - Health check for monitoring
|
||||
Regular → Executable
+67
-253
@@ -2,156 +2,104 @@
|
||||
ASGI application for Yargı MCP Server
|
||||
|
||||
This module provides ASGI/HTTP access to the Yargı MCP server,
|
||||
allowing it to be deployed as a web service with FastAPI wrapper
|
||||
for Stripe webhook integration.
|
||||
allowing it to be deployed as a web service with FastAPI wrapper.
|
||||
|
||||
Usage:
|
||||
uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
"""
|
||||
|
||||
import os
|
||||
from fastapi import FastAPI, Request, HTTPException
|
||||
import json
|
||||
import logging
|
||||
from fastapi import FastAPI, Request
|
||||
from fastapi.responses import JSONResponse
|
||||
from fastapi.exception_handlers import http_exception_handler
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.responses import Response
|
||||
|
||||
# Import the fully configured MCP app with all tools
|
||||
from mcp_server_main import app as mcp_server
|
||||
from mcp_server_main import create_app
|
||||
|
||||
# Import Stripe webhook router
|
||||
from stripe_webhook import router as stripe_router
|
||||
# Setup logging
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Import MCP Auth HTTP adapter
|
||||
from mcp_auth_http_adapter import router as mcp_auth_router
|
||||
|
||||
# OAuth configuration from environment variables
|
||||
CLERK_ISSUER = os.getenv("CLERK_ISSUER", "https://accounts.yargimcp.com")
|
||||
BASE_URL = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
# Configure CORS middleware
|
||||
# Configure CORS
|
||||
cors_origins = os.getenv("ALLOWED_ORIGINS", "*").split(",")
|
||||
|
||||
# Create MCP app
|
||||
mcp_server = create_app()
|
||||
|
||||
# Create MCP Starlette sub-application
|
||||
mcp_app = mcp_server.http_app(path="/")
|
||||
|
||||
|
||||
# Configure JSON encoder for proper Turkish character support
|
||||
class UTF8JSONResponse(JSONResponse):
|
||||
def __init__(self, content=None, status_code=200, headers=None, **kwargs):
|
||||
if headers is None:
|
||||
headers = {}
|
||||
headers["Content-Type"] = "application/json; charset=utf-8"
|
||||
super().__init__(content, status_code, headers, **kwargs)
|
||||
|
||||
def render(self, content) -> bytes:
|
||||
return json.dumps(
|
||||
content,
|
||||
ensure_ascii=False,
|
||||
allow_nan=False,
|
||||
indent=None,
|
||||
separators=(",", ":"),
|
||||
).encode("utf-8")
|
||||
|
||||
custom_middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=cors_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST", "OPTIONS"],
|
||||
allow_headers=["Content-Type", "Authorization", "X-Request-ID"],
|
||||
allow_headers=["Content-Type", "X-Request-ID", "X-Session-ID"],
|
||||
),
|
||||
]
|
||||
|
||||
# Create MCP Starlette sub-application first
|
||||
mcp_app = mcp_server.http_app(
|
||||
path="/",
|
||||
middleware=custom_middleware
|
||||
)
|
||||
|
||||
# Create FastAPI wrapper application with MCP app's lifespan
|
||||
# Create FastAPI wrapper application
|
||||
app = FastAPI(
|
||||
title="Yargı MCP Server",
|
||||
description="MCP server for Turkish legal databases with OAuth authentication",
|
||||
description="MCP server for Turkish legal databases",
|
||||
version="0.1.0",
|
||||
middleware=custom_middleware,
|
||||
lifespan=mcp_app.lifespan # Critical: Get lifespan from mcp_app, not mcp_server
|
||||
)
|
||||
|
||||
# Add Stripe webhook router to FastAPI
|
||||
app.include_router(stripe_router, prefix="/api")
|
||||
|
||||
# Add MCP Auth HTTP adapter to FastAPI (replaces old OAuth router)
|
||||
app.include_router(mcp_auth_router)
|
||||
|
||||
# Custom 401 exception handler for MCP spec compliance
|
||||
@app.exception_handler(401)
|
||||
async def custom_401_handler(request: Request, exc: HTTPException):
|
||||
"""Custom 401 handler that adds WWW-Authenticate header as required by MCP spec"""
|
||||
response = await http_exception_handler(request, exc)
|
||||
|
||||
# Add WWW-Authenticate header pointing to protected resource metadata
|
||||
# as required by RFC 9728 Section 5.1 and MCP Authorization spec
|
||||
response.headers["WWW-Authenticate"] = (
|
||||
'Bearer '
|
||||
'error="invalid_token", '
|
||||
'error_description="The access token is missing or invalid", '
|
||||
f'resource="{BASE_URL}/.well-known/oauth-protected-resource"'
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
# Mount MCP app as sub-application
|
||||
app.mount("/mcp", mcp_app)
|
||||
|
||||
# Add POST handler for /mcp to forward to mounted app
|
||||
@app.post("/mcp")
|
||||
async def mcp_post_handler(request: Request):
|
||||
"""Forward POST /mcp requests to mounted MCP app"""
|
||||
# Forward to the mounted app by calling it directly
|
||||
async def receive():
|
||||
return await request.receive()
|
||||
|
||||
# Create a new scope for the mounted app
|
||||
scope = request.scope.copy()
|
||||
scope["path"] = "/" # Root path for the mounted app
|
||||
scope["path_info"] = "/"
|
||||
|
||||
# Capture response
|
||||
response_parts = {"status": 200, "headers": [], "body": b""}
|
||||
|
||||
async def send(message):
|
||||
if message["type"] == "http.response.start":
|
||||
response_parts["status"] = message["status"]
|
||||
response_parts["headers"] = message["headers"]
|
||||
elif message["type"] == "http.response.body":
|
||||
response_parts["body"] += message.get("body", b"")
|
||||
|
||||
# Call the mounted MCP app
|
||||
await mcp_app(scope, receive, send)
|
||||
|
||||
# Return the response
|
||||
from starlette.responses import Response
|
||||
|
||||
# Convert ASGI headers to dict
|
||||
headers = {}
|
||||
for name, value in response_parts["headers"]:
|
||||
headers[name.decode()] = value.decode()
|
||||
|
||||
return Response(
|
||||
content=response_parts["body"],
|
||||
status_code=response_parts["status"],
|
||||
headers=headers
|
||||
default_response_class=UTF8JSONResponse,
|
||||
redirect_slashes=False,
|
||||
)
|
||||
|
||||
|
||||
# FastAPI health check endpoint
|
||||
@app.get("/health")
|
||||
async def health_check():
|
||||
"""Health check endpoint for monitoring"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"auth_enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true"
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@app.api_route("/mcp", methods=["GET", "POST", "HEAD", "OPTIONS"])
|
||||
async def redirect_to_slash(request: Request):
|
||||
"""Redirect /mcp to /mcp/ preserving HTTP method with 308"""
|
||||
from fastapi.responses import RedirectResponse
|
||||
return RedirectResponse(url="/mcp/", status_code=308)
|
||||
|
||||
|
||||
# FastAPI root endpoint
|
||||
@app.get("/")
|
||||
async def root():
|
||||
"""Root endpoint with service information"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"service": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases with OAuth authentication",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"endpoints": {
|
||||
"mcp": "/mcp",
|
||||
"health": "/health",
|
||||
"status": "/status",
|
||||
"stripe_webhook": "/api/stripe/webhook",
|
||||
"oauth_login": "/auth/login",
|
||||
"oauth_callback": "/auth/callback",
|
||||
"oauth_google": "/auth/google/login",
|
||||
"user_info": "/auth/user"
|
||||
},
|
||||
"transports": {
|
||||
"http": "/mcp"
|
||||
},
|
||||
"supported_databases": [
|
||||
"Yargıtay (Court of Cassation)",
|
||||
@@ -162,147 +110,15 @@ async def root():
|
||||
"Kamu İhale Kurulu (Public Procurement Authority)",
|
||||
"Rekabet Kurumu (Competition Authority)",
|
||||
"Sayıştay (Court of Accounts)",
|
||||
"Bedesten API (Multiple courts)"
|
||||
"KVKK (Personal Data Protection Authority)",
|
||||
"BDDK (Banking Regulation and Supervision Agency)",
|
||||
"BTK (Information and Communication Technologies Authority)",
|
||||
"Bedesten API (Multiple courts)",
|
||||
"Sigorta Tahkim Komisyonu (Insurance Arbitration Commission)",
|
||||
],
|
||||
"authentication": {
|
||||
"enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true",
|
||||
"type": "OAuth 2.0 via Clerk",
|
||||
"issuer": os.getenv("CLERK_ISSUER", "https://clerk.accounts.dev"),
|
||||
"providers": ["google"],
|
||||
"flow": "authorization_code"
|
||||
}
|
||||
})
|
||||
|
||||
# OAuth 2.0 Authorization Server Metadata proxy (for MCP clients that can't reach Clerk directly)
|
||||
@app.get("/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server():
|
||||
"""OAuth 2.0 Authorization Server Metadata proxy to Clerk"""
|
||||
return JSONResponse({
|
||||
"issuer": CLERK_ISSUER,
|
||||
"authorization_endpoint": f"{BASE_URL}/auth/login",
|
||||
"token_endpoint": f"{BASE_URL}/auth/callback",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"token_endpoint_auth_methods_supported": ["client_secret_basic", "none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"subject_types_supported": ["public"],
|
||||
"id_token_signing_alg_values_supported": ["RS256"],
|
||||
"claims_supported": ["sub", "iss", "aud", "exp", "iat", "email", "name"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/auth/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
|
||||
# MCP endpoint info for GET requests (ChatGPT compatibility)
|
||||
@app.get("/mcp")
|
||||
async def mcp_info():
|
||||
"""MCP endpoint information for discovery"""
|
||||
return JSONResponse({
|
||||
"mcp_server": True,
|
||||
"name": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"protocol": "mcp/1.0",
|
||||
"transport": "http",
|
||||
"authentication_required": True,
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": f"{BASE_URL}/auth/login",
|
||||
"token_url": f"{BASE_URL}/auth/callback",
|
||||
"scopes": ["read", "search"],
|
||||
"provider": "clerk"
|
||||
},
|
||||
"endpoints": {
|
||||
"mcp_protocol": "/mcp",
|
||||
"discovery": "/mcp/discovery",
|
||||
"well_known": "/.well-known/mcp",
|
||||
"health": "/health",
|
||||
"oauth_login": "/auth/login"
|
||||
},
|
||||
"capabilities": {
|
||||
"tools": True,
|
||||
"resources": True,
|
||||
"prompts": False
|
||||
},
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"usage": {
|
||||
"note": "This is an MCP server. Use POST to /mcp/ with proper MCP protocol headers.",
|
||||
"headers_required": [
|
||||
"Content-Type: application/json",
|
||||
"Accept: application/json, text/event-stream",
|
||||
"Authorization: Bearer <token>",
|
||||
"X-Session-ID: <session-id>"
|
||||
]
|
||||
}
|
||||
})
|
||||
|
||||
# OAuth 2.0 Protected Resource Metadata (RFC 9728) - MCP Spec Required
|
||||
@app.get("/.well-known/oauth-protected-resource")
|
||||
async def oauth_protected_resource():
|
||||
"""OAuth 2.0 Protected Resource Metadata as required by MCP spec"""
|
||||
return JSONResponse({
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [
|
||||
BASE_URL
|
||||
],
|
||||
"scopes_supported": ["read", "search"],
|
||||
"bearer_methods_supported": ["header"],
|
||||
"resource_documentation": f"{BASE_URL}/mcp",
|
||||
"resource_policy_uri": f"{BASE_URL}/privacy"
|
||||
})
|
||||
|
||||
# Standard well-known discovery endpoint
|
||||
@app.get("/.well-known/mcp")
|
||||
async def well_known_mcp():
|
||||
"""Standard MCP discovery endpoint"""
|
||||
return JSONResponse({
|
||||
"mcp_server": {
|
||||
"name": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"endpoint": f"{BASE_URL}/mcp",
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": f"{BASE_URL}/auth/login",
|
||||
"scopes": ["read", "search"]
|
||||
},
|
||||
"capabilities": ["tools", "resources"],
|
||||
"tools_count": len(mcp_server._tool_manager._tools)
|
||||
}
|
||||
})
|
||||
|
||||
# MCP Discovery endpoint for ChatGPT integration
|
||||
@app.get("/mcp/discovery")
|
||||
async def mcp_discovery():
|
||||
"""MCP Discovery endpoint for ChatGPT and other MCP clients"""
|
||||
return JSONResponse({
|
||||
"name": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"version": "0.1.0",
|
||||
"protocol": "mcp",
|
||||
"transport": "http",
|
||||
"endpoint": "/mcp",
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": "/auth/login",
|
||||
"token_url": "/auth/callback",
|
||||
"scopes": ["read", "search"],
|
||||
"provider": "clerk"
|
||||
},
|
||||
"capabilities": {
|
||||
"tools": True,
|
||||
"resources": True,
|
||||
"prompts": False
|
||||
},
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"contact": {
|
||||
"url": BASE_URL,
|
||||
"email": "support@yargi-mcp.dev"
|
||||
}
|
||||
})
|
||||
|
||||
# FastAPI status endpoint
|
||||
@app.get("/status")
|
||||
async def status():
|
||||
"""Status endpoint with detailed information"""
|
||||
@@ -313,21 +129,19 @@ async def status():
|
||||
"description": tool.description[:100] + "..." if len(tool.description) > 100 else tool.description
|
||||
})
|
||||
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "operational",
|
||||
"tools": tools,
|
||||
"total_tools": len(tools),
|
||||
"transport": "streamable_http",
|
||||
"architecture": "FastAPI wrapper + MCP Starlette sub-app",
|
||||
"auth_status": "enabled" if os.getenv("ENABLE_AUTH", "false").lower() == "true" else "disabled"
|
||||
})
|
||||
}
|
||||
|
||||
# Alternative: SSE transport (for compatibility)
|
||||
sse_app = mcp_server.http_app(
|
||||
path="/sse",
|
||||
transport="sse",
|
||||
middleware=custom_middleware
|
||||
)
|
||||
|
||||
# Mount MCP app at /mcp/
|
||||
app.mount("/mcp/", mcp_app)
|
||||
|
||||
# Set the lifespan context after mounting
|
||||
app.router.lifespan_context = mcp_app.lifespan
|
||||
|
||||
# Export for uvicorn
|
||||
__all__ = ["app", "sse_app"]
|
||||
__all__ = ["app"]
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
# bddk_mcp_module/__init__.py
|
||||
|
||||
from .client import BddkApiClient
|
||||
from .models import (
|
||||
BddkSearchRequest,
|
||||
BddkDecisionSummary,
|
||||
BddkSearchResult,
|
||||
BddkDocumentMarkdown
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"BddkApiClient",
|
||||
"BddkSearchRequest",
|
||||
"BddkDecisionSummary",
|
||||
"BddkSearchResult",
|
||||
"BddkDocumentMarkdown"
|
||||
]
|
||||
@@ -0,0 +1,253 @@
|
||||
# bddk_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from typing import List, Optional, Dict, Any
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urlparse
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
BddkSearchRequest,
|
||||
BddkDecisionSummary,
|
||||
BddkSearchResult,
|
||||
BddkDocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
class BddkApiClient:
|
||||
"""
|
||||
API client for searching and retrieving BDDK (Banking Regulation Authority) decisions
|
||||
using Tavily Search API for discovery and direct HTTP requests for content retrieval.
|
||||
"""
|
||||
|
||||
TAVILY_API_URL = "https://api.tavily.com/search"
|
||||
BDDK_BASE_URL = "https://www.bddk.org.tr"
|
||||
DOCUMENT_URL_TEMPLATE = "https://www.bddk.org.tr/Mevzuat/DokumanGetir/{document_id}"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000 # Character limit per page
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
"""Initialize the BDDK API client."""
|
||||
self.tavily_api_key = os.getenv("TAVILY_API_KEY")
|
||||
if not self.tavily_api_key:
|
||||
# Fallback to development token
|
||||
self.tavily_api_key = "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"
|
||||
logger.info("Using fallback Tavily API token (development token)")
|
||||
else:
|
||||
logger.info("Using Tavily API key from environment variable")
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
headers={
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
|
||||
},
|
||||
timeout=httpx.Timeout(request_timeout)
|
||||
)
|
||||
self.markitdown = MarkItDown()
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close the HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("BddkApiClient: HTTP client session closed.")
|
||||
|
||||
def _extract_document_id(self, url: str) -> Optional[str]:
|
||||
"""Extract document ID from BDDK URL."""
|
||||
# Primary pattern: https://www.bddk.org.tr/Mevzuat/DokumanGetir/310
|
||||
match = re.search(r'/DokumanGetir/(\d+)', url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Alternative patterns for different BDDK URL formats
|
||||
# Pattern: /Liste/55 -> use as document ID
|
||||
match = re.search(r'/Liste/(\d+)', url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: /EkGetir/13?ekId=381 -> use ekId as document ID
|
||||
match = re.search(r'ekId=(\d+)', url)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
return None
|
||||
|
||||
async def search_decisions(
|
||||
self,
|
||||
request: BddkSearchRequest
|
||||
) -> BddkSearchResult:
|
||||
"""
|
||||
Search for BDDK decisions using Tavily API.
|
||||
|
||||
Args:
|
||||
request: Search request parameters
|
||||
|
||||
Returns:
|
||||
BddkSearchResult with matching decisions
|
||||
"""
|
||||
try:
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.tavily_api_key}"
|
||||
}
|
||||
|
||||
# Tavily API request - enhanced for BDDK decision documents
|
||||
query = f"{request.keywords} \"Karar Sayısı\""
|
||||
payload = {
|
||||
"query": query,
|
||||
"country": "turkey",
|
||||
"include_domains": ["https://www.bddk.org.tr/Mevzuat/DokumanGetir"],
|
||||
"max_results": request.pageSize,
|
||||
"search_depth": "advanced"
|
||||
}
|
||||
|
||||
# Calculate offset for pagination
|
||||
if request.page > 1:
|
||||
# Tavily doesn't have direct pagination, so we'll need to handle this
|
||||
# For now, we'll just return empty for pages > 1
|
||||
logger.warning(f"Tavily API doesn't support pagination. Page {request.page} requested.")
|
||||
|
||||
response = await self.http_client.post(
|
||||
self.TAVILY_API_URL,
|
||||
json=payload,
|
||||
headers=headers
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
|
||||
# Log raw Tavily response for debugging
|
||||
logger.info(f"Tavily returned {len(data.get('results', []))} results")
|
||||
|
||||
# Convert Tavily results to our format
|
||||
decisions = []
|
||||
for result in data.get("results", []):
|
||||
# Extract document ID from URL
|
||||
url = result.get("url", "")
|
||||
logger.debug(f"Processing URL: {url}")
|
||||
doc_id = self._extract_document_id(url)
|
||||
if doc_id:
|
||||
decision = BddkDecisionSummary(
|
||||
title=result.get("title", "").replace("[PDF] ", "").strip(),
|
||||
document_id=doc_id,
|
||||
content=result.get("content", "")[:500] # Limit content length
|
||||
)
|
||||
decisions.append(decision)
|
||||
logger.debug(f"Added decision: {decision.title} (ID: {doc_id})")
|
||||
else:
|
||||
logger.warning(f"Could not extract document ID from URL: {url}")
|
||||
|
||||
return BddkSearchResult(
|
||||
decisions=decisions,
|
||||
total_results=len(data.get("results", [])),
|
||||
page=request.page,
|
||||
pageSize=request.pageSize
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error searching BDDK decisions: {e}")
|
||||
if e.response.status_code == 401:
|
||||
raise Exception("Tavily API authentication failed. Check API key.")
|
||||
raise Exception(f"Failed to search BDDK decisions: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error searching BDDK decisions: {e}")
|
||||
raise Exception(f"Failed to search BDDK decisions: {str(e)}")
|
||||
|
||||
async def get_document_markdown(
|
||||
self,
|
||||
document_id: str,
|
||||
page_number: int = 1
|
||||
) -> BddkDocumentMarkdown:
|
||||
"""
|
||||
Retrieve a BDDK document and convert it to Markdown format.
|
||||
|
||||
Args:
|
||||
document_id: BDDK document ID (e.g., '310')
|
||||
page_number: Page number for paginated content (1-indexed)
|
||||
|
||||
Returns:
|
||||
BddkDocumentMarkdown with paginated content
|
||||
"""
|
||||
try:
|
||||
# Try different URL patterns for BDDK documents
|
||||
potential_urls = [
|
||||
f"https://www.bddk.org.tr/Mevzuat/DokumanGetir/{document_id}",
|
||||
f"https://www.bddk.org.tr/Mevzuat/Liste/{document_id}",
|
||||
f"https://www.bddk.org.tr/KurumHakkinda/EkGetir/13?ekId={document_id}",
|
||||
f"https://www.bddk.org.tr/KurumHakkinda/EkGetir/5?ekId={document_id}"
|
||||
]
|
||||
|
||||
document_url = None
|
||||
response = None
|
||||
|
||||
# Try each URL pattern until one works
|
||||
for url in potential_urls:
|
||||
try:
|
||||
logger.info(f"Trying BDDK document URL: {url}")
|
||||
response = await self.http_client.get(
|
||||
url,
|
||||
follow_redirects=True
|
||||
)
|
||||
response.raise_for_status()
|
||||
document_url = url
|
||||
break
|
||||
except httpx.HTTPStatusError:
|
||||
continue
|
||||
|
||||
if not response or not document_url:
|
||||
raise Exception(f"Could not find document with ID {document_id}")
|
||||
|
||||
logger.info(f"Successfully fetched BDDK document from: {document_url}")
|
||||
|
||||
# Determine content type
|
||||
content_type = response.headers.get("content-type", "").lower()
|
||||
|
||||
# Convert to Markdown based on content type
|
||||
if "pdf" in content_type:
|
||||
# Handle PDF documents. markitdown is sync; offload to thread
|
||||
# so PDF parsing doesn't block the event-loop / other requests.
|
||||
pdf_stream = io.BytesIO(response.content)
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, pdf_stream, file_extension=".pdf"
|
||||
)
|
||||
markdown_content = result.text_content
|
||||
else:
|
||||
# Handle HTML documents (sync conversion offloaded to thread)
|
||||
html_stream = io.BytesIO(response.content)
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, html_stream, file_extension=".html"
|
||||
)
|
||||
markdown_content = result.text_content
|
||||
|
||||
# Clean up the markdown content
|
||||
markdown_content = markdown_content.strip()
|
||||
|
||||
# Calculate pagination
|
||||
total_length = len(markdown_content)
|
||||
total_pages = math.ceil(total_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
|
||||
# Extract the requested page
|
||||
start_idx = (page_number - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_idx = start_idx + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
page_content = markdown_content[start_idx:end_idx]
|
||||
|
||||
return BddkDocumentMarkdown(
|
||||
document_id=document_id,
|
||||
markdown_content=page_content,
|
||||
page_number=page_number,
|
||||
total_pages=total_pages
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error fetching BDDK document {document_id}: {e}")
|
||||
raise Exception(f"Failed to fetch BDDK document: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error processing BDDK document {document_id}: {e}")
|
||||
raise Exception(f"Failed to process BDDK document: {str(e)}")
|
||||
@@ -0,0 +1,43 @@
|
||||
# bddk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional
|
||||
|
||||
class BddkSearchRequest(BaseModel):
|
||||
"""
|
||||
Request model for searching BDDK decisions via Tavily API.
|
||||
|
||||
BDDK (Bankacılık Düzenleme ve Denetleme Kurumu) is Turkey's Banking
|
||||
Regulation and Supervision Agency responsible for banking licenses,
|
||||
electronic money institutions, and financial regulations.
|
||||
"""
|
||||
keywords: str = Field(..., description="Search keywords in Turkish")
|
||||
page: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)")
|
||||
|
||||
class BddkDecisionSummary(BaseModel):
|
||||
"""Summary of a BDDK decision from search results."""
|
||||
title: str = Field(..., description="Decision title")
|
||||
document_id: str = Field(..., description="BDDK document ID (e.g., '310')")
|
||||
content: str = Field(..., description="Decision summary/excerpt")
|
||||
|
||||
class BddkSearchResult(BaseModel):
|
||||
"""Response model for BDDK decision search results."""
|
||||
decisions: List[BddkDecisionSummary] = Field(
|
||||
default_factory=list,
|
||||
description="List of matching BDDK decisions"
|
||||
)
|
||||
total_results: int = Field(0, description="Total number of results")
|
||||
page: int = Field(1, description="Current page number")
|
||||
pageSize: int = Field(10, description="Results per page")
|
||||
|
||||
class BddkDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
BDDK decision document converted to Markdown format.
|
||||
|
||||
Supports paginated content for long documents (5000 chars per page).
|
||||
"""
|
||||
document_id: str = Field(..., description="BDDK document ID")
|
||||
markdown_content: str = Field("", description="Document content in Markdown")
|
||||
page_number: int = Field(1, description="Current page number")
|
||||
total_pages: int = Field(1, description="Total number of pages")
|
||||
+160
-35
@@ -1,21 +1,91 @@
|
||||
# bedesten_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
import asyncio
|
||||
import base64
|
||||
from typing import Optional
|
||||
import io
|
||||
import logging
|
||||
from markitdown import MarkItDown
|
||||
import tempfile
|
||||
import os
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
import httpx
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
BedestenSearchRequest, BedestenSearchResponse,
|
||||
BedestenDocumentRequest, BedestenDocumentResponse,
|
||||
BedestenDocumentMarkdown, BedestenDocumentRequestData
|
||||
)
|
||||
from .enums import get_full_birim_adi
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BedestenRateLimited(Exception):
|
||||
"""Raised when the local rate-limit bucket would block longer than allowed.
|
||||
|
||||
Carries the suggested retry-after (seconds) so callers can surface a
|
||||
structured 429-style response to the MCP client instead of silently
|
||||
blocking the event-loop slot for the full bucket-pause window.
|
||||
"""
|
||||
|
||||
def __init__(self, retry_after: float) -> None:
|
||||
self.retry_after = retry_after
|
||||
super().__init__(f"local bucket would block {retry_after:.1f}s")
|
||||
|
||||
|
||||
class _TokenBucket:
|
||||
"""Asyncio token bucket with explicit back-pressure.
|
||||
|
||||
Measured Bedesten limit (per source IP, 2026-05-08): 10 requests per
|
||||
rolling 30s window with full refill — equivalent to capacity=10,
|
||||
refill_rate=1 token / 3s. Even with margin, 429s still leak through
|
||||
when other clients share the egress IP, so we also expose
|
||||
``penalize_until`` so callers can freeze the bucket when the server
|
||||
actually returns 429 (Retry-After).
|
||||
"""
|
||||
|
||||
def __init__(self, capacity: int, refill_per_s: float) -> None:
|
||||
self.capacity = float(capacity)
|
||||
self.refill_per_s = float(refill_per_s)
|
||||
self._tokens = float(capacity)
|
||||
self._last = time.monotonic()
|
||||
self._not_before = 0.0
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def acquire(self, max_wait: Optional[float] = None) -> None:
|
||||
"""Acquire one token. If ``max_wait`` is set and the next wait would
|
||||
exceed it, raise :class:`BedestenRateLimited` immediately instead of
|
||||
sleeping — keeps a single rate-limited request from holding the
|
||||
worker-slot for the full bucket-pause window (up to ~30s on 429)."""
|
||||
deadline = (time.monotonic() + max_wait) if max_wait is not None else None
|
||||
while True:
|
||||
async with self._lock:
|
||||
now = time.monotonic()
|
||||
if now < self._not_before:
|
||||
wait_s = self._not_before - now
|
||||
else:
|
||||
self._tokens = min(
|
||||
self.capacity,
|
||||
self._tokens + (now - self._last) * self.refill_per_s,
|
||||
)
|
||||
self._last = now
|
||||
if self._tokens >= 1.0:
|
||||
self._tokens -= 1.0
|
||||
return
|
||||
wait_s = (1.0 - self._tokens) / self.refill_per_s
|
||||
if deadline is not None:
|
||||
remaining = deadline - time.monotonic()
|
||||
if wait_s > remaining:
|
||||
raise BedestenRateLimited(retry_after=wait_s)
|
||||
await asyncio.sleep(wait_s)
|
||||
|
||||
def penalize_until(self, monotonic_deadline: float) -> None:
|
||||
"""Pause the bucket until ``monotonic_deadline`` (drains tokens)."""
|
||||
self._not_before = max(self._not_before, monotonic_deadline)
|
||||
self._tokens = 0.0
|
||||
self._last = time.monotonic()
|
||||
|
||||
class BedestenApiClient:
|
||||
"""
|
||||
API Client for Bedesten (bedesten.adalet.gov.tr) - Alternative legal decision search system.
|
||||
@@ -25,6 +95,17 @@ class BedestenApiClient:
|
||||
SEARCH_ENDPOINT = "/emsal-karar/searchDocuments"
|
||||
DOCUMENT_ENDPOINT = "/emsal-karar/getDocumentContent"
|
||||
|
||||
# Measured limit (per source IP): 10 requests per 30s window with full
|
||||
# refill (≈ 1 token / 3s steady). We default to 1-token capacity and
|
||||
# 3.5s spacing (no burst, ~14% safety margin). Override via env:
|
||||
# BEDESTEN_RATE_CAPACITY (default 1)
|
||||
# BEDESTEN_RATE_REFILL_S (default 3.5; seconds per token)
|
||||
# BEDESTEN_RATE_MAX_WAIT_S (default 8.0; max seconds to wait in the
|
||||
# local bucket before returning a structured 429 to the caller)
|
||||
_DEFAULT_CAPACITY = int(os.getenv("BEDESTEN_RATE_CAPACITY", "1"))
|
||||
_DEFAULT_REFILL_S = float(os.getenv("BEDESTEN_RATE_REFILL_S", "3.5"))
|
||||
_DEFAULT_MAX_WAIT_S = float(os.getenv("BEDESTEN_RATE_MAX_WAIT_S", "8.0"))
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
@@ -42,6 +123,24 @@ class BedestenApiClient:
|
||||
},
|
||||
timeout=request_timeout
|
||||
)
|
||||
self._bucket = _TokenBucket(
|
||||
capacity=self._DEFAULT_CAPACITY,
|
||||
refill_per_s=1.0 / self._DEFAULT_REFILL_S,
|
||||
)
|
||||
|
||||
def _handle_429(self, response: httpx.Response, op: str) -> None:
|
||||
"""Apply back-pressure to the shared bucket based on Retry-After."""
|
||||
retry_after_raw = response.headers.get("Retry-After", "")
|
||||
try:
|
||||
retry_after = float(retry_after_raw)
|
||||
except (TypeError, ValueError):
|
||||
retry_after = 30.0
|
||||
# Cap penalty so a hostile/buggy server can't freeze us indefinitely.
|
||||
retry_after = max(1.0, min(retry_after, 60.0))
|
||||
self._bucket.penalize_until(time.monotonic() + retry_after + 0.5)
|
||||
logger.warning(
|
||||
f"BedestenApiClient: 429 on {op}; bucket paused {retry_after + 0.5:.1f}s"
|
||||
)
|
||||
|
||||
async def search_documents(self, search_request: BedestenSearchRequest) -> BedestenSearchResponse:
|
||||
"""
|
||||
@@ -50,11 +149,26 @@ class BedestenApiClient:
|
||||
"""
|
||||
logger.info(f"BedestenApiClient: Searching documents with phrase: {search_request.data.phrase}")
|
||||
|
||||
# Map abbreviated birimAdi to full Turkish name before sending to API
|
||||
original_birim_adi = search_request.data.birimAdi
|
||||
mapped_birim_adi = get_full_birim_adi(original_birim_adi)
|
||||
search_request.data.birimAdi = mapped_birim_adi
|
||||
if original_birim_adi != "ALL":
|
||||
logger.info(f"BedestenApiClient: Mapped birimAdi '{original_birim_adi}' to '{mapped_birim_adi}'")
|
||||
|
||||
try:
|
||||
# Create request dict and remove birimAdi if empty
|
||||
request_dict = search_request.model_dump()
|
||||
if not request_dict["data"]["birimAdi"]: # Remove if empty string
|
||||
del request_dict["data"]["birimAdi"]
|
||||
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.post(
|
||||
self.SEARCH_ENDPOINT,
|
||||
json=search_request.model_dump()
|
||||
json=request_dict
|
||||
)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, "search")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -82,26 +196,51 @@ class BedestenApiClient:
|
||||
)
|
||||
|
||||
# Get document
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.post(
|
||||
self.DOCUMENT_ENDPOINT,
|
||||
json=doc_request.model_dump()
|
||||
)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, f"document {document_id}")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
doc_response = BedestenDocumentResponse(**response_json)
|
||||
|
||||
# Decode base64 content
|
||||
# Add null safety checks for document data
|
||||
if not hasattr(doc_response, 'data') or doc_response.data is None:
|
||||
raise ValueError("Document response does not contain data")
|
||||
|
||||
if not hasattr(doc_response.data, 'content') or doc_response.data.content is None:
|
||||
raise ValueError("Document data does not contain content")
|
||||
|
||||
if not hasattr(doc_response.data, 'mimeType') or doc_response.data.mimeType is None:
|
||||
raise ValueError("Document data does not contain mimeType")
|
||||
|
||||
# Decode base64 content with error handling
|
||||
try:
|
||||
content_bytes = base64.b64decode(doc_response.data.content)
|
||||
except Exception as e:
|
||||
raise ValueError(f"Failed to decode base64 content: {str(e)}")
|
||||
|
||||
mime_type = doc_response.data.mimeType
|
||||
|
||||
logger.info(f"BedestenApiClient: Document mime type: {mime_type}")
|
||||
|
||||
# Convert to markdown based on mime type
|
||||
# Convert to markdown based on mime type. markitdown is sync and
|
||||
# PDF parsing in particular can block the event-loop for seconds,
|
||||
# which on a single-worker uvicorn deployment stalls every other
|
||||
# in-flight MCP request and new TLS handshakes. Offload to a
|
||||
# thread so the event-loop stays responsive.
|
||||
if mime_type == "text/html":
|
||||
html_content = content_bytes.decode('utf-8')
|
||||
markdown_content = self._convert_html_to_markdown(html_content)
|
||||
markdown_content = await asyncio.to_thread(
|
||||
self._convert_html_to_markdown, html_content
|
||||
)
|
||||
elif mime_type == "application/pdf":
|
||||
markdown_content = self._convert_pdf_to_markdown(content_bytes)
|
||||
markdown_content = await asyncio.to_thread(
|
||||
self._convert_pdf_to_markdown, content_bytes
|
||||
)
|
||||
else:
|
||||
logger.warning(f"Unsupported mime type: {mime_type}")
|
||||
markdown_content = f"Unsupported content type: {mime_type}. Unable to convert to markdown."
|
||||
@@ -109,7 +248,7 @@ class BedestenApiClient:
|
||||
return BedestenDocumentMarkdown(
|
||||
documentId=document_id,
|
||||
markdown_content=markdown_content,
|
||||
source_url=f"{self.BASE_URL}/document/{document_id}",
|
||||
source_url=f"https://mevzuat.adalet.gov.tr/ictihat/{document_id}",
|
||||
mime_type=mime_type
|
||||
)
|
||||
|
||||
@@ -125,17 +264,14 @@ class BedestenApiClient:
|
||||
if not html_content:
|
||||
return None
|
||||
|
||||
temp_file_path = None
|
||||
try:
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
|
||||
# Write HTML to temp file
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp:
|
||||
tmp.write(html_content)
|
||||
temp_file_path = tmp.name
|
||||
|
||||
# Convert
|
||||
result = md_converter.convert(temp_file_path)
|
||||
result = md_converter.convert(html_stream)
|
||||
markdown_content = result.text_content
|
||||
|
||||
logger.info("Successfully converted HTML to Markdown")
|
||||
@@ -144,27 +280,19 @@ class BedestenApiClient:
|
||||
except Exception as e:
|
||||
logger.error(f"Error converting HTML to Markdown: {e}")
|
||||
return f"Error converting HTML content: {str(e)}"
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
|
||||
def _convert_pdf_to_markdown(self, pdf_bytes: bytes) -> Optional[str]:
|
||||
"""Convert PDF to Markdown using MarkItDown"""
|
||||
if not pdf_bytes:
|
||||
return None
|
||||
|
||||
temp_file_path = None
|
||||
try:
|
||||
# MarkItDown supports PDF with markitdown[pdf]
|
||||
# Create BytesIO stream from PDF bytes
|
||||
pdf_stream = io.BytesIO(pdf_bytes)
|
||||
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
|
||||
# Write PDF to temp file
|
||||
with tempfile.NamedTemporaryFile(mode="wb", delete=False, suffix=".pdf") as tmp:
|
||||
tmp.write(pdf_bytes)
|
||||
temp_file_path = tmp.name
|
||||
|
||||
# Convert
|
||||
result = md_converter.convert(temp_file_path)
|
||||
result = md_converter.convert(pdf_stream)
|
||||
markdown_content = result.text_content
|
||||
|
||||
logger.info("Successfully converted PDF to Markdown")
|
||||
@@ -173,9 +301,6 @@ class BedestenApiClient:
|
||||
except Exception as e:
|
||||
logger.error(f"Error converting PDF to Markdown: {e}")
|
||||
return f"Error converting PDF content: {str(e)}. The document may be corrupted or in an unsupported format."
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close HTTP client session"""
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
# bedesten_mcp_module/enums.py
|
||||
|
||||
from typing import Literal
|
||||
|
||||
# Unified compressed enum for both Yargıtay and Danıştay chambers
|
||||
BirimAdiEnum = Literal[
|
||||
"ALL", # All chambers
|
||||
|
||||
# Yargıtay (Court of Cassation) - Civil Chambers
|
||||
"H1", "H2", "H3", "H4", "H5", "H6", "H7", "H8", "H9", "H10",
|
||||
"H11", "H12", "H13", "H14", "H15", "H16", "H17", "H18", "H19", "H20",
|
||||
"H21", "H22", "H23",
|
||||
|
||||
# Yargıtay - Criminal Chambers
|
||||
"C1", "C2", "C3", "C4", "C5", "C6", "C7", "C8", "C9", "C10",
|
||||
"C11", "C12", "C13", "C14", "C15", "C16", "C17", "C18", "C19", "C20",
|
||||
"C21", "C22", "C23",
|
||||
|
||||
# Yargıtay - Councils and Assemblies
|
||||
"HGK", # Hukuk Genel Kurulu
|
||||
"CGK", # Ceza Genel Kurulu
|
||||
"BGK", # Büyük Genel Kurulu
|
||||
"HBK", # Hukuk Daireleri Başkanlar Kurulu
|
||||
"CBK", # Ceza Daireleri Başkanlar Kurulu
|
||||
|
||||
# Danıştay (Council of State) - Chambers
|
||||
"D1", "D2", "D3", "D4", "D5", "D6", "D7", "D8", "D9", "D10",
|
||||
"D11", "D12", "D13", "D14", "D15", "D16", "D17",
|
||||
|
||||
# Danıştay - Councils and Boards
|
||||
"DBGK", # Büyük Gen.Kur. (Grand General Assembly)
|
||||
"IDDK", # İdare Dava Daireleri Kurulu
|
||||
"VDDK", # Vergi Dava Daireleri Kurulu
|
||||
"IBK", # İçtihatları Birleştirme Kurulu
|
||||
"IIK", # İdari İşler Kurulu
|
||||
"DBK", # Başkanlar Kurulu
|
||||
|
||||
# Military High Administrative Court
|
||||
"AYIM", # Askeri Yüksek İdare Mahkemesi
|
||||
"AYIMDK", # Askeri Yüksek İdare Mahkemesi Daireler Kurulu
|
||||
"AYIMB", # Askeri Yüksek İdare Mahkemesi Başsavcılığı
|
||||
"AYIM1", # Askeri Yüksek İdare Mahkemesi 1. Daire
|
||||
"AYIM2", # Askeri Yüksek İdare Mahkemesi 2. Daire
|
||||
"AYIM3" # Askeri Yüksek İdare Mahkemesi 3. Daire
|
||||
]
|
||||
|
||||
# Mapping from abbreviated values to full Turkish API values
|
||||
BIRIM_ADI_MAPPING = {
|
||||
"ALL": None, # Will be handled specially in client
|
||||
|
||||
# Yargıtay Civil Chambers (1-23)
|
||||
"H1": "1. Hukuk Dairesi", "H2": "2. Hukuk Dairesi", "H3": "3. Hukuk Dairesi",
|
||||
"H4": "4. Hukuk Dairesi", "H5": "5. Hukuk Dairesi", "H6": "6. Hukuk Dairesi",
|
||||
"H7": "7. Hukuk Dairesi", "H8": "8. Hukuk Dairesi", "H9": "9. Hukuk Dairesi",
|
||||
"H10": "10. Hukuk Dairesi", "H11": "11. Hukuk Dairesi", "H12": "12. Hukuk Dairesi",
|
||||
"H13": "13. Hukuk Dairesi", "H14": "14. Hukuk Dairesi", "H15": "15. Hukuk Dairesi",
|
||||
"H16": "16. Hukuk Dairesi", "H17": "17. Hukuk Dairesi", "H18": "18. Hukuk Dairesi",
|
||||
"H19": "19. Hukuk Dairesi", "H20": "20. Hukuk Dairesi", "H21": "21. Hukuk Dairesi",
|
||||
"H22": "22. Hukuk Dairesi", "H23": "23. Hukuk Dairesi",
|
||||
|
||||
# Yargıtay Criminal Chambers (1-23)
|
||||
"C1": "1. Ceza Dairesi", "C2": "2. Ceza Dairesi", "C3": "3. Ceza Dairesi",
|
||||
"C4": "4. Ceza Dairesi", "C5": "5. Ceza Dairesi", "C6": "6. Ceza Dairesi",
|
||||
"C7": "7. Ceza Dairesi", "C8": "8. Ceza Dairesi", "C9": "9. Ceza Dairesi",
|
||||
"C10": "10. Ceza Dairesi", "C11": "11. Ceza Dairesi", "C12": "12. Ceza Dairesi",
|
||||
"C13": "13. Ceza Dairesi", "C14": "14. Ceza Dairesi", "C15": "15. Ceza Dairesi",
|
||||
"C16": "16. Ceza Dairesi", "C17": "17. Ceza Dairesi", "C18": "18. Ceza Dairesi",
|
||||
"C19": "19. Ceza Dairesi", "C20": "20. Ceza Dairesi", "C21": "21. Ceza Dairesi",
|
||||
"C22": "22. Ceza Dairesi", "C23": "23. Ceza Dairesi",
|
||||
|
||||
# Yargıtay Councils and Assemblies
|
||||
"HGK": "Hukuk Genel Kurulu",
|
||||
"CGK": "Ceza Genel Kurulu",
|
||||
"BGK": "Büyük Genel Kurulu",
|
||||
"HBK": "Hukuk Daireleri Başkanlar Kurulu",
|
||||
"CBK": "Ceza Daireleri Başkanlar Kurulu",
|
||||
|
||||
# Danıştay Chambers (1-17)
|
||||
"D1": "1. Daire", "D2": "2. Daire", "D3": "3. Daire", "D4": "4. Daire",
|
||||
"D5": "5. Daire", "D6": "6. Daire", "D7": "7. Daire", "D8": "8. Daire",
|
||||
"D9": "9. Daire", "D10": "10. Daire", "D11": "11. Daire", "D12": "12. Daire",
|
||||
"D13": "13. Daire", "D14": "14. Daire", "D15": "15. Daire", "D16": "16. Daire",
|
||||
"D17": "17. Daire",
|
||||
|
||||
# Danıştay Councils and Boards
|
||||
"DBGK": "Büyük Gen.Kur.",
|
||||
"IDDK": "İdare Dava Daireleri Kurulu",
|
||||
"VDDK": "Vergi Dava Daireleri Kurulu",
|
||||
"IBK": "İçtihatları Birleştirme Kurulu",
|
||||
"IIK": "İdari İşler Kurulu",
|
||||
"DBK": "Başkanlar Kurulu",
|
||||
|
||||
# Military High Administrative Court
|
||||
"AYIM": "Askeri Yüksek İdare Mahkemesi",
|
||||
"AYIMDK": "Askeri Yüksek İdare Mahkemesi Daireler Kurulu",
|
||||
"AYIMB": "Askeri Yüksek İdare Mahkemesi Başsavcılığı",
|
||||
"AYIM1": "Askeri Yüksek İdare Mahkemesi 1. Daire",
|
||||
"AYIM2": "Askeri Yüksek İdare Mahkemesi 2. Daire",
|
||||
"AYIM3": "Askeri Yüksek İdare Mahkemesi 3. Daire"
|
||||
}
|
||||
|
||||
# Helper function to get full Turkish name from abbreviated value
|
||||
def get_full_birim_adi(abbreviated_value: str) -> str:
|
||||
"""Convert abbreviated birimAdi value to full Turkish name for API calls."""
|
||||
if abbreviated_value == "ALL" or not abbreviated_value:
|
||||
return "" # Empty string for ALL or None
|
||||
|
||||
return BIRIM_ADI_MAPPING.get(abbreviated_value, abbreviated_value)
|
||||
|
||||
# Helper function to validate abbreviated value
|
||||
def is_valid_birim_adi(abbreviated_value: str) -> bool:
|
||||
"""Check if abbreviated birimAdi value is valid."""
|
||||
return abbreviated_value in BIRIM_ADI_MAPPING
|
||||
@@ -4,94 +4,33 @@ from pydantic import BaseModel, Field
|
||||
from typing import List, Optional, Dict, Any, Literal, Union
|
||||
from datetime import datetime
|
||||
|
||||
# Import YargitayBirimEnum for chamber filtering
|
||||
from yargitay_mcp_module.models import YargitayBirimEnum
|
||||
# Import compressed BirimAdiEnum for chamber filtering
|
||||
from .enums import BirimAdiEnum
|
||||
|
||||
# Danıştay Chamber/Board Options
|
||||
DanistayBirimEnum = Literal[
|
||||
"ALL", # "ALL" for all chambers
|
||||
# Main Councils
|
||||
"Büyük Gen.Kur.", # Grand General Assembly
|
||||
"İdare Dava Daireleri Kurulu", # Administrative Cases Chambers Council
|
||||
"Vergi Dava Daireleri Kurulu", # Tax Cases Chambers Council
|
||||
"İçtihatları Birleştirme Kurulu", # Precedents Unification Council
|
||||
"İdari İşler Kurulu", # Administrative Affairs Council
|
||||
"Başkanlar Kurulu", # Presidents Council
|
||||
# Chambers
|
||||
"1. Daire", "2. Daire", "3. Daire", "4. Daire", "5. Daire",
|
||||
"6. Daire", "7. Daire", "8. Daire", "9. Daire", "10. Daire",
|
||||
"11. Daire", "12. Daire", "13. Daire", "14. Daire", "15. Daire",
|
||||
"16. Daire", "17. Daire",
|
||||
# Military High Administrative Court
|
||||
"Askeri Yüksek İdare Mahkemesi",
|
||||
"Askeri Yüksek İdare Mahkemesi Daireler Kurulu",
|
||||
"Askeri Yüksek İdare Mahkemesi Başsavcılığı",
|
||||
"Askeri Yüksek İdare Mahkemesi 1. Daire",
|
||||
"Askeri Yüksek İdare Mahkemesi 2. Daire",
|
||||
"Askeri Yüksek İdare Mahkemesi 3. Daire"
|
||||
# Court Type Options for Unified Search
|
||||
BedestenCourtTypeEnum = Literal[
|
||||
"YARGITAYKARARI", # Yargıtay (Court of Cassation)
|
||||
"DANISTAYKARAR", # Danıştay (Council of State)
|
||||
"YERELHUKUK", # Local Civil Courts
|
||||
"ISTINAFHUKUK", # Civil Courts of Appeals
|
||||
"KYB" # Extraordinary Appeals (Kanun Yararına Bozma)
|
||||
]
|
||||
|
||||
# Search Request Models
|
||||
class BedestenSearchData(BaseModel):
|
||||
pageSize: int = Field(..., description="""Number of results per page.
|
||||
Range: 1-100 results per page
|
||||
Recommended: 10-50 for balanced performance
|
||||
Higher values for comprehensive analysis""")
|
||||
pageNumber: int = Field(..., description="""Page number to retrieve (1-indexed).
|
||||
Start with 1 for first page
|
||||
Calculate total pages from response.data.total / pageSize
|
||||
Navigate: pageNumber=2 gets next set of results""")
|
||||
itemTypeList: List[str] = Field(..., description="""Court type filter - determines which court decisions to search:
|
||||
• ["YARGITAYKARARI"]: Court of Cassation (Yargıtay) - supreme court civil/criminal decisions
|
||||
• ["DANISTAYKARAR"]: Council of State (Danıştay) - administrative court decisions
|
||||
• ["YERELHUKUK"]: Local Civil Courts (Yerel Hukuk Mahkemeleri) - first instance civil decisions
|
||||
• ["ISTINAFHUKUK"]: Civil Courts of Appeals (İstinaf Hukuk Mahkemeleri) - appellate court decisions
|
||||
• ["KYB"]: Extraordinary Appeal (Kanun Yararına Bozma) - extraordinary appeal decisions
|
||||
Note: Use single-item list for specific court type targeting""")
|
||||
phrase: str = Field(..., description="""Search phrase/keyword with advanced search support:
|
||||
• Regular search: "mülkiyet kararı" - searches words separately
|
||||
• Exact phrase: "\"mülkiyet kararı\"" - searches exact phrase (more precise)
|
||||
• Legal concepts: "\"idari işlem\"", "\"sözleşme ihlali\"", "\"tazminat davası\""
|
||||
• Empty string: searches all documents (use with filters)
|
||||
Exact phrases significantly reduce false positives for precise legal research""")
|
||||
birimAdi: Optional[Union[YargitayBirimEnum, DanistayBirimEnum]] = Field(None, description="""
|
||||
Chamber/Department (Daire) filter (optional). Available options depend on itemTypeList:
|
||||
|
||||
For YARGITAYKARARI - Court of Cassation (52 options):
|
||||
- None/null for ALL chambers
|
||||
- 'Civil General Assembly (Hukuk Genel Kurulu)', '1st Civil Chamber (1. Hukuk Dairesi)' through '23rd Civil Chamber (23. Hukuk Dairesi)'
|
||||
- 'Criminal General Assembly (Ceza Genel Kurulu)', '1st Criminal Chamber (1. Ceza Dairesi)' through '23rd Criminal Chamber (23. Ceza Dairesi)'
|
||||
- 'Civil Chambers Presidents Board (Hukuk Daireleri Başkanlar Kurulu)', 'Criminal Chambers Presidents Board (Ceza Daireleri Başkanlar Kurulu)'
|
||||
- 'Grand General Assembly (Büyük Genel Kurulu)'
|
||||
|
||||
For DANISTAYKARAR - Council of State (27 options):
|
||||
- None/null for ALL chambers
|
||||
- 'Grand General Assembly (Büyük Gen.Kur.)', 'Administrative Cases Chambers Council (İdare Dava Daireleri Kurulu)', 'Tax Cases Chambers Council (Vergi Dava Daireleri Kurulu)'
|
||||
- '1st Chamber (1. Daire)' through '17th Chamber (17. Daire)'
|
||||
- 'Precedents Unification Council (İçtihatları Birleştirme Kurulu)', 'Administrative Affairs Council (İdari İşler Kurulu)', 'Presidents Council (Başkanlar Kurulu)'
|
||||
- Military courts: 'Military High Administrative Court (Askeri Yüksek İdare Mahkemesi)' variants
|
||||
pageSize: int = Field(..., description="Results per page (1-10)")
|
||||
pageNumber: int = Field(..., description="Page number (1-indexed)")
|
||||
itemTypeList: List[str] = Field(..., description="Court type filter (YARGITAYKARARI/DANISTAYKARAR/YERELHUKUK/ISTINAFHUKUK/KYB)")
|
||||
phrase: str = Field(..., description="Search phrase. Supports: 'word', \"exact phrase\", +required, -exclude, AND/OR/NOT operators. No wildcards or regex.")
|
||||
birimAdi: BirimAdiEnum = Field("ALL", description="""
|
||||
Chamber filter (optional). Abbreviated values with Turkish names:
|
||||
• Yargıtay: H1-H23 (1-23. Hukuk Dairesi), C1-C23 (1-23. Ceza Dairesi), HGK (Hukuk Genel Kurulu), CGK (Ceza Genel Kurulu), BGK (Büyük Genel Kurulu), HBK (Hukuk Daireleri Başkanlar Kurulu), CBK (Ceza Daireleri Başkanlar Kurulu)
|
||||
• Danıştay: D1-D17 (1-17. Daire), DBGK (Büyük Gen.Kur.), IDDK (İdare Dava Daireleri Kurulu), VDDK (Vergi Dava Daireleri Kurulu), IBK (İçtihatları Birleştirme Kurulu), IIK (İdari İşler Kurulu), DBK (Başkanlar Kurulu), AYIM (Askeri Yüksek İdare Mahkemesi), AYIM1-3 (Askeri Yüksek İdare Mahkemesi 1-3. Daire)
|
||||
""")
|
||||
kararTarihiStart: Optional[str] = Field(None, description="""Decision start date (Karar Tarihi Başlangıç) filter (optional).
|
||||
Format: YYYY-MM-DDTHH:MM:SS.000Z (ISO 8601 with Z timezone)
|
||||
Examples:
|
||||
• "2024-01-01T00:00:00.000Z" - from beginning of 2024
|
||||
• "2023-06-15T00:00:00.000Z" - from June 15, 2023
|
||||
• "2024-03-01T00:00:00.000Z" - from March 1, 2024
|
||||
Use with kararTarihiEnd for date range, or alone for "from date" filtering""")
|
||||
kararTarihiEnd: Optional[str] = Field(None, description="""Decision end date (Karar Tarihi Bitiş) filter (optional).
|
||||
Format: YYYY-MM-DDTHH:MM:SS.000Z (ISO 8601 with Z timezone)
|
||||
Examples:
|
||||
• "2024-12-31T23:59:59.999Z" - until end of 2024
|
||||
• "2023-12-31T23:59:59.999Z" - until end of 2023
|
||||
• "2024-06-30T23:59:59.999Z" - until end of June 2024
|
||||
Use with kararTarihiStart for date range, or alone for "until date" filtering""")
|
||||
sortFields: List[str] = Field(default=["KARAR_TARIHI"], description="""Sorting field (Sıralama Alanı) specification.
|
||||
["KARAR_TARIHI"]: Sort by decision date (Karar Tarihi) [DEFAULT]
|
||||
Most common use case for chronological ordering""")
|
||||
sortDirection: str = Field(default="desc", description="""Sort direction (Sıralama Yönü) for results.
|
||||
"desc": Descending order - newest decisions first [DEFAULT]
|
||||
"asc": Ascending order - oldest decisions first
|
||||
Recommended: "desc" for latest legal developments""")
|
||||
kararTarihiStart: Optional[str] = Field(None, description="Start date (ISO 8601 format)")
|
||||
kararTarihiEnd: Optional[str] = Field(None, description="End date (ISO 8601 format)")
|
||||
sortFields: List[str] = Field(default=["KARAR_TARIHI"], description="Sort fields")
|
||||
sortDirection: str = Field(default="desc", description="Sort direction (asc/desc)")
|
||||
|
||||
class BedestenSearchRequest(BaseModel):
|
||||
data: BedestenSearchData
|
||||
@@ -125,7 +64,7 @@ class BedestenSearchDataResponse(BaseModel):
|
||||
start: int
|
||||
|
||||
class BedestenSearchResponse(BaseModel):
|
||||
data: BedestenSearchDataResponse
|
||||
data: Optional[BedestenSearchDataResponse]
|
||||
metadata: Dict[str, Any]
|
||||
|
||||
# Document Request/Response Models
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
# btk_mcp_module/__init__.py
|
||||
|
||||
from .client import BtkApiClient
|
||||
from .models import (
|
||||
BtkDocumentMarkdown,
|
||||
BtkDecisionSummary,
|
||||
BtkSearchRequest,
|
||||
BtkSearchResult,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"BtkApiClient",
|
||||
"BtkDocumentMarkdown",
|
||||
"BtkDecisionSummary",
|
||||
"BtkSearchRequest",
|
||||
"BtkSearchResult",
|
||||
]
|
||||
@@ -0,0 +1,206 @@
|
||||
# btk_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import io
|
||||
import logging
|
||||
import math
|
||||
from datetime import datetime
|
||||
from typing import Any, Dict, Optional
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import httpx
|
||||
from markitdown import MarkItDown
|
||||
from pydantic import HttpUrl
|
||||
|
||||
from .models import (
|
||||
BtkDecisionSummary,
|
||||
BtkDocumentMarkdown,
|
||||
BtkSearchRequest,
|
||||
BtkSearchResult,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s - %(name)s - %(levelname)s - %(message)s",
|
||||
)
|
||||
|
||||
|
||||
class BtkApiClient:
|
||||
"""Client for BTK (Information and Communication Technologies Authority) decisions."""
|
||||
|
||||
BASE_URL = "https://www.btk.tr"
|
||||
API_PATH = "/api/content/board-decisions"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "application/json,text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"User-Agent": (
|
||||
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
),
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True,
|
||||
)
|
||||
self.markitdown = MarkItDown(enable_plugins=False)
|
||||
|
||||
def _build_search_params(self, request: BtkSearchRequest) -> Dict[str, str]:
|
||||
params: Dict[str, str] = {
|
||||
"page": str(request.page),
|
||||
"limit": str(request.pageSize),
|
||||
"locale": "tr",
|
||||
}
|
||||
|
||||
if request.keywords.strip():
|
||||
params["search"] = request.keywords.strip()
|
||||
if request.decision_no.strip():
|
||||
params["filter[decision_no]"] = request.decision_no.strip()
|
||||
if request.decision_date.strip():
|
||||
params["filter[decision_date]"] = request.decision_date.strip()
|
||||
if request.publication_date.strip():
|
||||
params["date_from"] = request.publication_date.strip()
|
||||
params["date_to"] = request.publication_date.strip()
|
||||
if request.relevant_unit.strip():
|
||||
params["filter[relevant_unit]"] = request.relevant_unit.strip()
|
||||
|
||||
return params
|
||||
|
||||
@staticmethod
|
||||
def _format_date(value: Optional[str]) -> Optional[str]:
|
||||
if not value:
|
||||
return None
|
||||
normalized = value.replace("Z", "+00:00")
|
||||
try:
|
||||
return datetime.fromisoformat(normalized).date().isoformat()
|
||||
except ValueError:
|
||||
return value[:10] if len(value) >= 10 else value
|
||||
|
||||
@staticmethod
|
||||
def _extract_pdf_url(file_data: Any) -> Optional[str]:
|
||||
if not isinstance(file_data, dict):
|
||||
return None
|
||||
for key in ("url", "storageUrl"):
|
||||
value = file_data.get(key)
|
||||
if isinstance(value, str) and value.strip():
|
||||
return value.strip()
|
||||
return None
|
||||
|
||||
def _parse_decision(self, item: Dict[str, Any]) -> BtkDecisionSummary:
|
||||
data = item.get("data") if isinstance(item.get("data"), dict) else {}
|
||||
file_data = data.get("file_url") if isinstance(data.get("file_url"), dict) else {}
|
||||
pdf_url = self._extract_pdf_url(file_data)
|
||||
|
||||
return BtkDecisionSummary(
|
||||
id=str(item.get("id") or ""),
|
||||
title=str(item.get("title") or ""),
|
||||
slug=str(item.get("slug") or ""),
|
||||
decision_no=data.get("decision_no"),
|
||||
decision_date=self._format_date(data.get("decision_date")),
|
||||
publication_date=self._format_date(item.get("publishedAt")),
|
||||
relevant_unit=data.get("relevant_unit"),
|
||||
pdf_url=HttpUrl(pdf_url) if pdf_url else None,
|
||||
original_filename=file_data.get("originalFilename") or file_data.get("filename"),
|
||||
)
|
||||
|
||||
async def search_decisions(self, request: BtkSearchRequest) -> BtkSearchResult:
|
||||
params = self._build_search_params(request)
|
||||
query_string = urlencode(params, doseq=True)
|
||||
query_url = f"{self.BASE_URL}{self.API_PATH}?{query_string}"
|
||||
logger.info("BtkApiClient: searching BTK decisions with URL: %s", query_url)
|
||||
|
||||
try:
|
||||
response = await self.http_client.get(self.API_PATH, params=params)
|
||||
response.raise_for_status()
|
||||
payload = response.json()
|
||||
except Exception as e:
|
||||
logger.error("BtkApiClient: error searching decisions: %s", e, exc_info=True)
|
||||
raise Exception(f"Failed to search BTK decisions: {str(e)}")
|
||||
|
||||
raw_items = payload.get("data") if isinstance(payload, dict) else []
|
||||
decisions = [
|
||||
self._parse_decision(item)
|
||||
for item in raw_items
|
||||
if isinstance(item, dict)
|
||||
]
|
||||
meta = payload.get("meta") if isinstance(payload.get("meta"), dict) else {}
|
||||
|
||||
return BtkSearchResult(
|
||||
decisions=decisions,
|
||||
total_results=int(meta.get("total") or len(decisions)),
|
||||
page=int(meta.get("page") or request.page),
|
||||
pageSize=int(meta.get("limit") or request.pageSize),
|
||||
total_pages=int(meta.get("totalPages") or 0),
|
||||
query_url=query_url,
|
||||
)
|
||||
|
||||
def _convert_pdf_to_markdown(self, pdf_bytes: bytes) -> str:
|
||||
pdf_stream = io.BytesIO(pdf_bytes)
|
||||
result = self.markitdown.convert_stream(pdf_stream, file_extension=".pdf")
|
||||
return (result.text_content or "").strip()
|
||||
|
||||
async def get_document_markdown(self, pdf_url: str, page_number: int = 1) -> BtkDocumentMarkdown:
|
||||
if not pdf_url or not pdf_url.strip():
|
||||
return BtkDocumentMarkdown(
|
||||
source_url=HttpUrl(f"{self.BASE_URL}/kurul-kararlari"),
|
||||
markdown_chunk=None,
|
||||
current_page=max(1, page_number),
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="pdf_url is required.",
|
||||
)
|
||||
|
||||
pdf_url = pdf_url.strip()
|
||||
if not pdf_url.startswith(("https://www.btk.gov.tr/", "https://www.btk.tr/")):
|
||||
return BtkDocumentMarkdown(
|
||||
source_url=HttpUrl(pdf_url),
|
||||
markdown_chunk=None,
|
||||
current_page=max(1, page_number),
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="Invalid BTK document URL. URL must start with https://www.btk.gov.tr/ or https://www.btk.tr/.",
|
||||
)
|
||||
|
||||
try:
|
||||
response = await self.http_client.get(pdf_url)
|
||||
response.raise_for_status()
|
||||
|
||||
content_type = response.headers.get("content-type", "").lower()
|
||||
if "pdf" not in content_type and not pdf_url.lower().endswith(".pdf"):
|
||||
raise Exception(f"Expected a PDF document, got content type: {content_type}")
|
||||
|
||||
markdown_content = await asyncio.to_thread(self._convert_pdf_to_markdown, response.content)
|
||||
total_pages = max(1, math.ceil(len(markdown_content) / self.DOCUMENT_MARKDOWN_CHUNK_SIZE))
|
||||
current_page = max(1, min(page_number, total_pages))
|
||||
start_index = (current_page - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_index = start_index + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
|
||||
return BtkDocumentMarkdown(
|
||||
source_url=HttpUrl(pdf_url),
|
||||
markdown_chunk=markdown_content[start_index:end_index],
|
||||
current_page=current_page,
|
||||
total_pages=total_pages,
|
||||
is_paginated=total_pages > 1,
|
||||
error_message=None,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error("BtkApiClient: error retrieving BTK PDF %s: %s", pdf_url, e, exc_info=True)
|
||||
return BtkDocumentMarkdown(
|
||||
source_url=HttpUrl(pdf_url),
|
||||
markdown_chunk=None,
|
||||
current_page=max(1, page_number),
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=f"Failed to retrieve BTK document: {str(e)}",
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
if hasattr(self, "http_client") and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("BtkApiClient: HTTP client session closed.")
|
||||
@@ -0,0 +1,58 @@
|
||||
# btk_mcp_module/models.py
|
||||
|
||||
from typing import List, Optional
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
|
||||
|
||||
class BtkSearchRequest(BaseModel):
|
||||
"""Request model for searching BTK Board decisions."""
|
||||
|
||||
keywords: str = Field("", description="Keywords searched in decision title/content metadata.")
|
||||
decision_no: str = Field("", description="BTK decision number, e.g. 2026/DK-THD/91.")
|
||||
decision_date: str = Field("", description="Decision date as YYYY-MM-DD.")
|
||||
publication_date: str = Field("", description="Publication date as YYYY-MM-DD.")
|
||||
relevant_unit: str = Field("", description="Related BTK department name.")
|
||||
page: int = Field(1, ge=1, description="Page number for results.")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page.")
|
||||
|
||||
|
||||
class BtkDecisionSummary(BaseModel):
|
||||
"""Summary of a BTK Board decision from search results."""
|
||||
|
||||
id: str = Field("", description="BTK content ID.")
|
||||
title: str = Field("", description="Decision title.")
|
||||
slug: str = Field("", description="BTK content slug.")
|
||||
decision_no: Optional[str] = Field(None, description="Decision number.")
|
||||
decision_date: Optional[str] = Field(None, description="Decision date.")
|
||||
publication_date: Optional[str] = Field(None, description="Publication date.")
|
||||
relevant_unit: Optional[str] = Field(None, description="Related BTK department.")
|
||||
pdf_url: Optional[HttpUrl] = Field(None, description="Direct URL of the decision PDF.")
|
||||
original_filename: Optional[str] = Field(None, description="Original PDF filename when available.")
|
||||
|
||||
|
||||
class BtkSearchResult(BaseModel):
|
||||
"""Response model for BTK Board decision search results."""
|
||||
|
||||
decisions: List[BtkDecisionSummary] = Field(default_factory=list)
|
||||
total_results: int = Field(0, description="Total number of matching results.")
|
||||
page: int = Field(1, description="Current page.")
|
||||
pageSize: int = Field(10, description="Results per page.")
|
||||
total_pages: int = Field(0, description="Total result pages.")
|
||||
query_url: str = Field("", description="BTK API URL used for the search.")
|
||||
|
||||
|
||||
class BtkDocumentMarkdown(BaseModel):
|
||||
"""BTK decision PDF converted to paginated Markdown."""
|
||||
|
||||
source_url: HttpUrl = Field(description="Source PDF URL.")
|
||||
markdown_chunk: Optional[str] = Field(None, description="A chunk of the Markdown content.")
|
||||
current_page: int = Field(1, description="Current Markdown chunk page.")
|
||||
total_pages: int = Field(1, description="Total Markdown chunk pages.")
|
||||
is_paginated: bool = Field(False, description="True when content spans multiple chunks.")
|
||||
error_message: Optional[str] = Field(None, description="Error message, if retrieval failed.")
|
||||
|
||||
class Config:
|
||||
json_encoders = {
|
||||
HttpUrl: str
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
#!/usr/bin/env python3
|
||||
from fastmcp import Client
|
||||
from mcp_server_main import app
|
||||
import json
|
||||
import asyncio
|
||||
|
||||
async def check_response_format():
|
||||
client = Client(app)
|
||||
async with client:
|
||||
result = await client.call_tool('search_bedesten_unified', {
|
||||
'phrase': 'mülkiyet',
|
||||
'court_types': ['YARGITAYKARARI'],
|
||||
'birimAdi': 'H1',
|
||||
'pageSize': 3
|
||||
})
|
||||
if result and result.content:
|
||||
data = json.loads(result.content[0].text)
|
||||
print('Response keys:', list(data.keys()))
|
||||
print('Sample response:', json.dumps(data, indent=2, ensure_ascii=False)[:500])
|
||||
|
||||
asyncio.run(check_response_format())
|
||||
@@ -1,13 +1,13 @@
|
||||
# danistay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
@@ -77,12 +77,16 @@ class DanistayApiClient:
|
||||
mevzuatNumarasi=params.mevzuatNumarasi or "",
|
||||
mevzuatAdi=params.mevzuatAdi or "",
|
||||
madde=params.madde or "",
|
||||
siralama=params.siralama,
|
||||
siralamaDirection=params.siralamaDirection,
|
||||
siralama="1",
|
||||
siralamaDirection="desc",
|
||||
pageSize=params.pageSize,
|
||||
pageNumber=params.pageNumber
|
||||
)
|
||||
final_payload = {"data": data_for_payload.model_dump(exclude_defaults=False, exclude_none=False)}
|
||||
# Create request dict and remove empty string fields to avoid API issues
|
||||
payload_dict = data_for_payload.model_dump(exclude_defaults=False, exclude_none=False)
|
||||
# Remove empty string fields that might cause API issues
|
||||
cleaned_payload = {k: v for k, v in payload_dict.items() if v != ""}
|
||||
final_payload = {"data": cleaned_payload}
|
||||
logger.info(f"DanistayApiClient: Performing DETAILED search via {self.DETAILED_SEARCH_ENDPOINT} with payload: {final_payload}")
|
||||
return await self._execute_api_search(self.DETAILED_SEARCH_ENDPOINT, final_payload)
|
||||
|
||||
@@ -124,31 +128,28 @@ class DanistayApiClient:
|
||||
html_input_for_markdown = processed_html
|
||||
|
||||
markdown_text = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown() # Basic conversion
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_input_for_markdown.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_file:
|
||||
tmp_file.write(html_input_for_markdown) # Write the full HTML string
|
||||
temp_file_path = tmp_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
conversion_result = md_converter.convert(html_stream)
|
||||
markdown_text = conversion_result.text_content
|
||||
logger.info("DanistayApiClient: HTML to Markdown conversion successful.")
|
||||
except Exception as e:
|
||||
logger.error(f"DanistayApiClient: Error during MarkItDown HTML to Markdown conversion: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
|
||||
return markdown_text
|
||||
|
||||
async def get_decision_document_as_markdown(self, id: str) -> DanistayDocumentMarkdown:
|
||||
"""
|
||||
Retrieves a specific Danıştay decision by ID and returns its content as Markdown.
|
||||
The /getDokuman endpoint for Danıştay returns direct HTML.
|
||||
The /getDokuman endpoint for Danıştay requires arananKelime parameter.
|
||||
"""
|
||||
document_api_url = f"{self.DOCUMENT_ENDPOINT}?id={id}"
|
||||
# Add required arananKelime parameter - using empty string as minimum requirement
|
||||
document_api_url = f"{self.DOCUMENT_ENDPOINT}?id={id}&arananKelime="
|
||||
source_url = f"{self.BASE_URL}{document_api_url}"
|
||||
logger.info(f"DanistayApiClient: Fetching Danistay document for Markdown (ID: {id}) from {source_url}")
|
||||
|
||||
@@ -170,7 +171,7 @@ class DanistayApiClient:
|
||||
source_url=source_url
|
||||
)
|
||||
|
||||
markdown_content = self._convert_html_to_markdown_danistay(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown_danistay, html_content_from_api)
|
||||
|
||||
return DanistayDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
@@ -5,7 +5,7 @@ from typing import List, Optional, Dict, Any
|
||||
|
||||
class DanistayBaseSearchRequest(BaseModel):
|
||||
"""Base model for common search parameters for Danistay."""
|
||||
pageSize: int = Field(default=10, ge=1, le=100)
|
||||
pageSize: int = Field(default=10, ge=1, le=10)
|
||||
pageNumber: int = Field(default=1, ge=1)
|
||||
# siralama and siralamaDirection are part of detailed search, not necessarily keyword search
|
||||
# as per user's provided payloads.
|
||||
@@ -21,11 +21,11 @@ class DanistayKeywordSearchRequestData(BaseModel):
|
||||
|
||||
class DanistayKeywordSearchRequest(BaseModel): # This is the model the MCP tool will accept
|
||||
"""Model for keyword-based search request for Danistay."""
|
||||
andKelimeler: List[str] = Field(default_factory=list, description="Keywords for AND logic (VE Mantığı), e.g., ['word1', 'word2']")
|
||||
orKelimeler: List[str] = Field(default_factory=list, description="Keywords for OR logic (VEYA Mantığı).")
|
||||
notAndKelimeler: List[str] = Field(default_factory=list, description="Keywords for NOT AND logic (VE DEĞİL Mantığı).")
|
||||
notOrKelimeler: List[str] = Field(default_factory=list, description="Keywords for NOT OR logic (VEYA DEĞİL Mantığı).")
|
||||
pageSize: int = Field(default=10, ge=1, le=100)
|
||||
andKelimeler: List[str] = Field(default_factory=list, description="AND keywords")
|
||||
orKelimeler: List[str] = Field(default_factory=list, description="OR keywords")
|
||||
notAndKelimeler: List[str] = Field(default_factory=list, description="NOT AND keywords")
|
||||
notOrKelimeler: List[str] = Field(default_factory=list, description="NOT OR keywords")
|
||||
pageSize: int = Field(default=10, ge=1, le=10)
|
||||
pageNumber: int = Field(default=1, ge=1)
|
||||
|
||||
class DanistayDetailedSearchRequestData(BaseModel): # Internal data model for detailed search payload
|
||||
@@ -51,20 +51,18 @@ class DanistayDetailedSearchRequestData(BaseModel): # Internal data model for de
|
||||
|
||||
class DanistayDetailedSearchRequest(DanistayBaseSearchRequest): # MCP tool will accept this
|
||||
"""Model for detailed search request for Danistay."""
|
||||
daire: Optional[str] = Field(None, description="Chamber/Department name (e.g., '1. Daire').")
|
||||
esasYil: Optional[str] = Field(None, description="Case year for 'Esas No'.")
|
||||
esasIlkSiraNo: Optional[str] = Field(None, description="Starting sequence for 'Esas No'.")
|
||||
esasSonSiraNo: Optional[str] = Field(None, description="Ending sequence for 'Esas No'.")
|
||||
kararYil: Optional[str] = Field(None, description="Decision year for 'Karar No'.")
|
||||
kararIlkSiraNo: Optional[str] = Field(None, description="Starting sequence for 'Karar No'.")
|
||||
kararSonSiraNo: Optional[str] = Field(None, description="Ending sequence for 'Karar No'.")
|
||||
baslangicTarihi: Optional[str] = Field(None, description="Start date for decision (DD.MM.YYYY).")
|
||||
bitisTarihi: Optional[str] = Field(None, description="End date for decision (DD.MM.YYYY).")
|
||||
mevzuatNumarasi: Optional[str] = Field(None, description="Legislation number.")
|
||||
mevzuatAdi: Optional[str] = Field(None, description="Legislation name.")
|
||||
madde: Optional[str] = Field(None, description="Article number.")
|
||||
siralama: str = Field("1", description="Sorting criteria (e.g., 1: Esas No, 3: Karar Tarihi).")
|
||||
siralamaDirection: str = Field("desc", description="Sorting direction ('asc' or 'desc').")
|
||||
daire: str = Field("", description="Chamber")
|
||||
esasYil: str = Field("", description="Case year")
|
||||
esasIlkSiraNo: str = Field("", description="Start case no")
|
||||
esasSonSiraNo: str = Field("", description="End case no")
|
||||
kararYil: str = Field("", description="Decision year")
|
||||
kararIlkSiraNo: str = Field("", description="Start decision no")
|
||||
kararSonSiraNo: str = Field("", description="End decision no")
|
||||
baslangicTarihi: str = Field("", description="Start date")
|
||||
bitisTarihi: str = Field("", description="End date")
|
||||
mevzuatNumarasi: str = Field("", description="Law number")
|
||||
mevzuatAdi: str = Field("", description="Law name")
|
||||
madde: str = Field("", description="Article")
|
||||
# Add a general keyword field if detailed search also supports it
|
||||
# arananKelime: Optional[str] = Field(None, description="General keyword for detailed search.")
|
||||
|
||||
@@ -76,15 +74,15 @@ class DanistayApiDecisionEntry(BaseModel):
|
||||
id: str
|
||||
# The API response for keyword search uses "daireKurul", detailed search example uses "daire".
|
||||
# We use an alias to handle both and map to a consistent field name "chamber".
|
||||
chamber: Optional[str] = Field(None, alias="daire", description="The chamber or board.")
|
||||
esasNo: Optional[str] = Field(None)
|
||||
kararNo: Optional[str] = Field(None)
|
||||
kararTarihi: Optional[str] = Field(None)
|
||||
arananKelime: Optional[str] = Field(None, description="Matched keyword (Aranan Kelime) if provided in response.")
|
||||
chamber: str = Field("", alias="daire", description="Chamber")
|
||||
esasNo: str = Field("", description="Case number")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
kararTarihi: str = Field("", description="Decision date")
|
||||
arananKelime: str = Field("", description="Keyword")
|
||||
# index: Optional[int] = None # Present in response, can be added if needed by MCP tool
|
||||
# siraNo: Optional[int] = None # Present in detailed response, can be added
|
||||
|
||||
document_url: Optional[HttpUrl] = Field(None, description="URL (Belge URL) to the full document, constructed by the client.")
|
||||
document_url: Optional[HttpUrl] = Field(None, description="Document URL")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True, extra='ignore') # Important for alias to work and ignore extra fields
|
||||
|
||||
@@ -93,17 +91,17 @@ class DanistayApiResponseInnerData(BaseModel):
|
||||
data: List[DanistayApiDecisionEntry]
|
||||
recordsTotal: int
|
||||
recordsFiltered: int
|
||||
draw: Optional[int] = Field(None, description="Draw counter (Çizim Sayıcısı) from API, usually for DataTables.")
|
||||
draw: int = Field(0, description="Draw counter")
|
||||
|
||||
class DanistayApiResponse(BaseModel):
|
||||
"""Model for the complete search response from the Danistay API."""
|
||||
data: DanistayApiResponseInnerData
|
||||
data: Optional[DanistayApiResponseInnerData] = Field(None, description="Response data, can be null when no results found")
|
||||
metadata: Optional[Dict[str, Any]] = Field(None, description="Optional metadata (Meta Veri) from API.")
|
||||
|
||||
class DanistayDocumentMarkdown(BaseModel):
|
||||
"""Model for a Danistay decision document, containing only Markdown content."""
|
||||
id: str
|
||||
markdown_content: Optional[str] = Field(None, description="The decision content (Karar İçeriği) converted to Markdown.")
|
||||
markdown_content: str = Field("", description="The decision content (Karar İçeriği) converted to Markdown.")
|
||||
source_url: HttpUrl
|
||||
|
||||
class CompactDanistaySearchResult(BaseModel):
|
||||
|
||||
@@ -1,66 +0,0 @@
|
||||
version: '3.8'
|
||||
|
||||
services:
|
||||
yargi-mcp:
|
||||
build: .
|
||||
image: yargi-mcp:latest
|
||||
container_name: yargi-mcp-server
|
||||
ports:
|
||||
- "${PORT:-8000}:8000"
|
||||
environment:
|
||||
- HOST=0.0.0.0
|
||||
- PORT=8000
|
||||
- LOG_LEVEL=${LOG_LEVEL:-info}
|
||||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-*}
|
||||
- API_TOKEN=${API_TOKEN:-}
|
||||
- PYTHONUNBUFFERED=1
|
||||
volumes:
|
||||
# Mount logs directory
|
||||
- ./logs:/app/logs
|
||||
# Mount .env file if it exists
|
||||
- ./.env:/app/.env:ro
|
||||
restart: unless-stopped
|
||||
healthcheck:
|
||||
test: ["CMD", "python", "-c", "import httpx; httpx.get('http://localhost:8000/health').raise_for_status()"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 10s
|
||||
networks:
|
||||
- yargi-network
|
||||
|
||||
# Optional: Nginx reverse proxy
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
container_name: yargi-nginx
|
||||
ports:
|
||||
- "80:80"
|
||||
- "443:443"
|
||||
volumes:
|
||||
- ./nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ./ssl:/etc/nginx/ssl:ro
|
||||
depends_on:
|
||||
- yargi-mcp
|
||||
networks:
|
||||
- yargi-network
|
||||
profiles:
|
||||
- production
|
||||
|
||||
# Optional: Redis for caching (future enhancement)
|
||||
redis:
|
||||
image: redis:alpine
|
||||
container_name: yargi-redis
|
||||
command: redis-server --appendonly yes
|
||||
volumes:
|
||||
- redis-data:/data
|
||||
networks:
|
||||
- yargi-network
|
||||
profiles:
|
||||
- with-cache
|
||||
|
||||
networks:
|
||||
yargi-network:
|
||||
driver: bridge
|
||||
|
||||
volumes:
|
||||
redis-data:
|
||||
@@ -1,428 +0,0 @@
|
||||
# Yargı MCP Server Dağıtım Rehberi
|
||||
|
||||
Bu rehber, Yargı MCP Server'ın ASGI web servisi olarak çeşitli dağıtım seçeneklerini kapsar.
|
||||
|
||||
## İçindekiler
|
||||
|
||||
- [Hızlı Başlangıç](#hızlı-başlangıç)
|
||||
- [Yerel Geliştirme](#yerel-geliştirme)
|
||||
- [Production Dağıtımı](#production-dağıtımı)
|
||||
- [Cloud Dağıtımı](#cloud-dağıtımı)
|
||||
- [Docker Dağıtımı](#docker-dağıtımı)
|
||||
- [Güvenlik Hususları](#güvenlik-hususları)
|
||||
- [İzleme](#izleme)
|
||||
|
||||
## Hızlı Başlangıç
|
||||
|
||||
### 1. Bağımlılıkları Yükleyin
|
||||
|
||||
```bash
|
||||
# ASGI sunucusu için uvicorn yükleyin
|
||||
pip install uvicorn
|
||||
|
||||
# Veya tüm bağımlılıklarla birlikte yükleyin
|
||||
pip install -e .
|
||||
pip install uvicorn
|
||||
```
|
||||
|
||||
### 2. Sunucuyu Çalıştırın
|
||||
|
||||
```bash
|
||||
# Temel başlatma
|
||||
python run_asgi.py
|
||||
|
||||
# Veya doğrudan uvicorn ile
|
||||
uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
```
|
||||
|
||||
Sunucu şu adreslerde kullanılabilir olacak:
|
||||
- MCP Endpoint: `http://localhost:8000/mcp/`
|
||||
- Sağlık Kontrolü: `http://localhost:8000/health`
|
||||
- API Durumu: `http://localhost:8000/status`
|
||||
|
||||
## Yerel Geliştirme
|
||||
|
||||
### Otomatik Yeniden Yükleme ile Geliştirme Sunucusu
|
||||
|
||||
```bash
|
||||
python run_asgi.py --reload --log-level debug
|
||||
```
|
||||
|
||||
### FastAPI Entegrasyonunu Kullanma
|
||||
|
||||
Ek REST API endpoint'leri için:
|
||||
|
||||
```bash
|
||||
uvicorn fastapi_app:app --reload
|
||||
```
|
||||
|
||||
Bu şunları sağlar:
|
||||
- `/docs` adresinde interaktif API dokümantasyonu
|
||||
- `/api/tools` adresinde araç listesi
|
||||
- `/api/databases` adresinde veritabanı bilgileri
|
||||
|
||||
### Ortam Değişkenleri
|
||||
|
||||
`.env.example` dosyasını temel alarak bir `.env` dosyası oluşturun:
|
||||
|
||||
```bash
|
||||
cp .env.example .env
|
||||
```
|
||||
|
||||
Temel değişkenler:
|
||||
- `HOST`: Sunucu host adresi (varsayılan: 127.0.0.1)
|
||||
- `PORT`: Sunucu portu (varsayılan: 8000)
|
||||
- `ALLOWED_ORIGINS`: CORS kökenleri (virgülle ayrılmış)
|
||||
- `LOG_LEVEL`: Log seviyesi (debug, info, warning, error)
|
||||
|
||||
## Production Dağıtımı
|
||||
|
||||
### 1. Uvicorn ile Çoklu Worker Kullanımı
|
||||
|
||||
```bash
|
||||
python run_asgi.py --host 0.0.0.0 --port 8000 --workers 4
|
||||
```
|
||||
|
||||
### 2. Gunicorn Kullanımı
|
||||
|
||||
```bash
|
||||
pip install gunicorn
|
||||
gunicorn asgi_app:app -w 4 -k uvicorn.workers.UvicornWorker --bind 0.0.0.0:8000
|
||||
```
|
||||
|
||||
### 3. Nginx Reverse Proxy ile
|
||||
|
||||
1. Nginx'i yükleyin
|
||||
2. Sağlanan `nginx.conf` dosyasını kullanın:
|
||||
|
||||
```bash
|
||||
sudo cp nginx.conf /etc/nginx/sites-available/yargi-mcp
|
||||
sudo ln -s /etc/nginx/sites-available/yargi-mcp /etc/nginx/sites-enabled/
|
||||
sudo nginx -t
|
||||
sudo systemctl reload nginx
|
||||
```
|
||||
|
||||
### 4. Systemd Servisi
|
||||
|
||||
`/etc/systemd/system/yargi-mcp.service` dosyasını oluşturun:
|
||||
|
||||
```ini
|
||||
[Unit]
|
||||
Description=Yargı MCP Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=exec
|
||||
User=www-data
|
||||
WorkingDirectory=/opt/yargi-mcp
|
||||
Environment="PATH=/opt/yargi-mcp/venv/bin"
|
||||
ExecStart=/opt/yargi-mcp/venv/bin/uvicorn asgi_app:app --host 0.0.0.0 --port 8000 --workers 4
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
```
|
||||
|
||||
Etkinleştirin ve başlatın:
|
||||
|
||||
```bash
|
||||
sudo systemctl enable yargi-mcp
|
||||
sudo systemctl start yargi-mcp
|
||||
```
|
||||
|
||||
## Cloud Dağıtımı
|
||||
|
||||
### Heroku
|
||||
|
||||
1. `Procfile` oluşturun:
|
||||
```
|
||||
web: uvicorn asgi_app:app --host 0.0.0.0 --port $PORT
|
||||
```
|
||||
|
||||
2. Dağıtın:
|
||||
```bash
|
||||
heroku create uygulama-isminiz
|
||||
git push heroku main
|
||||
```
|
||||
|
||||
### Railway
|
||||
|
||||
1. `railway.json` ekleyin:
|
||||
```json
|
||||
{
|
||||
"build": {
|
||||
"builder": "NIXPACKS"
|
||||
},
|
||||
"deploy": {
|
||||
"startCommand": "uvicorn asgi_app:app --host 0.0.0.0 --port $PORT"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
2. Railway CLI veya GitHub entegrasyonu ile dağıtın
|
||||
|
||||
### Google Cloud Run
|
||||
|
||||
1. Container oluşturun:
|
||||
```bash
|
||||
docker build -t yargi-mcp .
|
||||
docker tag yargi-mcp gcr.io/PROJE_ADINIZ/yargi-mcp
|
||||
docker push gcr.io/PROJE_ADINIZ/yargi-mcp
|
||||
```
|
||||
|
||||
2. Dağıtın:
|
||||
```bash
|
||||
gcloud run deploy yargi-mcp \
|
||||
--image gcr.io/PROJE_ADINIZ/yargi-mcp \
|
||||
--platform managed \
|
||||
--region us-central1 \
|
||||
--allow-unauthenticated
|
||||
```
|
||||
|
||||
### AWS Lambda (Mangum kullanarak)
|
||||
|
||||
1. Mangum'u yükleyin:
|
||||
```bash
|
||||
pip install mangum
|
||||
```
|
||||
|
||||
2. `lambda_handler.py` oluşturun:
|
||||
```python
|
||||
from mangum import Mangum
|
||||
from asgi_app import app
|
||||
|
||||
handler = Mangum(app, lifespan="off")
|
||||
```
|
||||
|
||||
3. AWS SAM veya Serverless Framework kullanarak dağıtın
|
||||
|
||||
## Docker Dağıtımı
|
||||
|
||||
### Tek Container
|
||||
|
||||
```bash
|
||||
# Oluşturun
|
||||
docker build -t yargi-mcp .
|
||||
|
||||
# Çalıştırın
|
||||
docker run -p 8000:8000 --env-file .env yargi-mcp
|
||||
```
|
||||
|
||||
### Docker Compose
|
||||
|
||||
```bash
|
||||
# Geliştirme
|
||||
docker-compose up
|
||||
|
||||
# Nginx ile Production
|
||||
docker-compose --profile production up
|
||||
|
||||
# Redis önbellekleme ile
|
||||
docker-compose --profile with-cache up
|
||||
```
|
||||
|
||||
### Kubernetes
|
||||
|
||||
Deployment YAML oluşturun:
|
||||
|
||||
```yaml
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: yargi-mcp
|
||||
spec:
|
||||
replicas: 3
|
||||
selector:
|
||||
matchLabels:
|
||||
app: yargi-mcp
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: yargi-mcp
|
||||
spec:
|
||||
containers:
|
||||
- name: yargi-mcp
|
||||
image: yargi-mcp:latest
|
||||
ports:
|
||||
- containerPort: 8000
|
||||
env:
|
||||
- name: HOST
|
||||
value: "0.0.0.0"
|
||||
- name: PORT
|
||||
value: "8000"
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: 8000
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 30
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: yargi-mcp-service
|
||||
spec:
|
||||
selector:
|
||||
app: yargi-mcp
|
||||
ports:
|
||||
- port: 80
|
||||
targetPort: 8000
|
||||
type: LoadBalancer
|
||||
```
|
||||
|
||||
## Güvenlik Hususları
|
||||
|
||||
### 1. Kimlik Doğrulama
|
||||
|
||||
`API_TOKEN` ortam değişkenini ayarlayarak token kimlik doğrulamasını etkinleştirin:
|
||||
|
||||
```bash
|
||||
export API_TOKEN=gizli-token-degeri
|
||||
```
|
||||
|
||||
Ardından isteklere ekleyin:
|
||||
```bash
|
||||
curl -H "Authorization: Bearer gizli-token-degeri" http://localhost:8000/api/tools
|
||||
```
|
||||
|
||||
### 2. HTTPS/SSL
|
||||
|
||||
Production için her zaman HTTPS kullanın:
|
||||
|
||||
1. SSL sertifikası edinin (Let's Encrypt vb.)
|
||||
2. Nginx veya cloud sağlayıcıda yapılandırın
|
||||
3. `ALLOWED_ORIGINS` değerini https:// kullanacak şekilde güncelleyin
|
||||
|
||||
### 3. Rate Limiting (Hız Sınırlama)
|
||||
|
||||
Sağlanan Nginx yapılandırması rate limiting içerir:
|
||||
- API endpoint'leri: 10 istek/saniye
|
||||
- MCP endpoint: 100 istek/saniye
|
||||
|
||||
### 4. CORS Yapılandırması
|
||||
|
||||
Production için belirli kaynaklara izin verin:
|
||||
|
||||
```bash
|
||||
ALLOWED_ORIGINS=https://app.sizindomain.com,https://www.sizindomain.com
|
||||
```
|
||||
|
||||
## İzleme
|
||||
|
||||
### Sağlık Kontrolleri
|
||||
|
||||
`/health` endpoint'ini izleyin:
|
||||
|
||||
```bash
|
||||
curl http://localhost:8000/health
|
||||
```
|
||||
|
||||
Yanıt:
|
||||
```json
|
||||
{
|
||||
"status": "healthy",
|
||||
"timestamp": "2024-12-26T10:00:00",
|
||||
"uptime_seconds": 3600,
|
||||
"tools_operational": true
|
||||
}
|
||||
```
|
||||
|
||||
### Loglama
|
||||
|
||||
Ortam değişkeni ile log seviyesini yapılandırın:
|
||||
|
||||
```bash
|
||||
LOG_LEVEL=info # veya debug, warning, error
|
||||
```
|
||||
|
||||
Loglar şuraya yazılır:
|
||||
- Konsol (stdout)
|
||||
- `logs/mcp_server.log` dosyası
|
||||
|
||||
### Metrikler (Opsiyonel)
|
||||
|
||||
OpenTelemetry desteği için:
|
||||
|
||||
```bash
|
||||
pip install opentelemetry-instrumentation-fastapi
|
||||
```
|
||||
|
||||
Ortam değişkenlerini ayarlayın:
|
||||
```bash
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317
|
||||
OTEL_SERVICE_NAME=yargi-mcp-server
|
||||
```
|
||||
|
||||
## Sorun Giderme
|
||||
|
||||
### Port Zaten Kullanımda
|
||||
|
||||
```bash
|
||||
# 8000 portunu kullanan işlemi bulun
|
||||
lsof -i :8000
|
||||
|
||||
# İşlemi sonlandırın
|
||||
kill -9 <PID>
|
||||
```
|
||||
|
||||
### İzin Hataları
|
||||
|
||||
Dosya izinlerinin doğru olduğundan emin olun:
|
||||
|
||||
```bash
|
||||
chmod +x run_asgi.py
|
||||
chown -R www-data:www-data /opt/yargi-mcp
|
||||
```
|
||||
|
||||
### Bellek Sorunları
|
||||
|
||||
Büyük belge işleme için worker belleğini artırın:
|
||||
|
||||
```bash
|
||||
# systemd servisinde
|
||||
Environment="PYTHONMALLOC=malloc"
|
||||
LimitNOFILE=65536
|
||||
```
|
||||
|
||||
### Zaman Aşımı Sorunları
|
||||
|
||||
Zaman aşımlarını ayarlayın:
|
||||
1. Uvicorn: `--timeout-keep-alive 75`
|
||||
2. Nginx: `proxy_read_timeout 300s;`
|
||||
3. Cloud sağlayıcılar: Platform özel zaman aşımı ayarlarını kontrol edin
|
||||
|
||||
## Performans Ayarlama
|
||||
|
||||
### 1. Worker İşlemleri
|
||||
|
||||
- Geliştirme: 1 worker
|
||||
- Production: CPU çekirdeği başına 2-4 worker
|
||||
|
||||
### 2. Bağlantı Havuzlama
|
||||
|
||||
Sunucu varsayılan olarak httpx ile bağlantı havuzlama kullanır.
|
||||
|
||||
### 3. Önbellekleme (Gelecek Geliştirme)
|
||||
|
||||
Redis önbellekleme docker-compose ile etkinleştirilebilir:
|
||||
|
||||
```bash
|
||||
docker-compose --profile with-cache up
|
||||
```
|
||||
|
||||
### 4. Veritabanı Zaman Aşımları
|
||||
|
||||
`.env` dosyasında veritabanı başına zaman aşımlarını ayarlayın:
|
||||
|
||||
```bash
|
||||
YARGITAY_TIMEOUT=60
|
||||
DANISTAY_TIMEOUT=60
|
||||
ANAYASA_TIMEOUT=90
|
||||
```
|
||||
|
||||
## Destek
|
||||
|
||||
Sorunlar ve sorular için:
|
||||
- GitHub Issues: https://github.com/saidsurucu/yargi-mcp/issues
|
||||
- Dokümantasyon: README.md dosyasına bakın
|
||||
+118
-14
@@ -1,13 +1,15 @@
|
||||
# emsal_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
|
||||
from typing import Dict, Any, List, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import time
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
@@ -21,12 +23,87 @@ logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
||||
|
||||
|
||||
class EmsalRateLimited(Exception):
|
||||
"""Raised when the local rate-limit bucket would block longer than allowed.
|
||||
|
||||
Carries the suggested retry-after (seconds) so callers can surface a
|
||||
structured 429-style response instead of silently blocking the
|
||||
event-loop slot for the full bucket-pause window.
|
||||
"""
|
||||
|
||||
def __init__(self, retry_after: float) -> None:
|
||||
self.retry_after = retry_after
|
||||
super().__init__(f"local bucket would block {retry_after:.1f}s")
|
||||
|
||||
|
||||
class _TokenBucket:
|
||||
"""Asyncio token bucket with explicit back-pressure.
|
||||
|
||||
The UYAP Emsal endpoint (emsal.uyap.gov.tr) rate-limits per source IP and
|
||||
returns HTTP 429 (an HTML error page, no Retry-After header) after a small
|
||||
burst of rapid requests. On the shared-egress-IP production deployment this
|
||||
is hit constantly, making unrelated searches appear to "return 0 results"
|
||||
depending only on request order. This bucket spaces requests to a safe rate
|
||||
and freezes on an actual 429 via ``penalize_until``.
|
||||
"""
|
||||
|
||||
def __init__(self, capacity: int, refill_per_s: float) -> None:
|
||||
self.capacity = float(capacity)
|
||||
self.refill_per_s = float(refill_per_s)
|
||||
self._tokens = float(capacity)
|
||||
self._last = time.monotonic()
|
||||
self._not_before = 0.0
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def acquire(self, max_wait: Optional[float] = None) -> None:
|
||||
"""Acquire one token. If ``max_wait`` is set and the next wait would
|
||||
exceed it, raise :class:`EmsalRateLimited` immediately instead of
|
||||
sleeping — keeps a single rate-limited request from holding the
|
||||
worker-slot for the full bucket-pause window."""
|
||||
deadline = (time.monotonic() + max_wait) if max_wait is not None else None
|
||||
while True:
|
||||
async with self._lock:
|
||||
now = time.monotonic()
|
||||
if now < self._not_before:
|
||||
wait_s = self._not_before - now
|
||||
else:
|
||||
self._tokens = min(
|
||||
self.capacity,
|
||||
self._tokens + (now - self._last) * self.refill_per_s,
|
||||
)
|
||||
self._last = now
|
||||
if self._tokens >= 1.0:
|
||||
self._tokens -= 1.0
|
||||
return
|
||||
wait_s = (1.0 - self._tokens) / self.refill_per_s
|
||||
if deadline is not None:
|
||||
remaining = deadline - time.monotonic()
|
||||
if wait_s > remaining:
|
||||
raise EmsalRateLimited(retry_after=wait_s)
|
||||
await asyncio.sleep(wait_s)
|
||||
|
||||
def penalize_until(self, monotonic_deadline: float) -> None:
|
||||
"""Pause the bucket until ``monotonic_deadline`` (drains tokens)."""
|
||||
self._not_before = max(self._not_before, monotonic_deadline)
|
||||
self._tokens = 0.0
|
||||
self._last = time.monotonic()
|
||||
|
||||
class EmsalApiClient:
|
||||
"""API Client for Emsal (UYAP Precedent Decision) search system."""
|
||||
BASE_URL = "https://emsal.uyap.gov.tr"
|
||||
DETAILED_SEARCH_ENDPOINT = "/aramadetaylist"
|
||||
DOCUMENT_ENDPOINT = "/getDokuman"
|
||||
|
||||
# UYAP Emsal rate-limits per source IP. Defaults mirror the sibling
|
||||
# Bedesten client (conservative: no burst, ~3.5s spacing). Override via env:
|
||||
# EMSAL_RATE_CAPACITY (default 1)
|
||||
# EMSAL_RATE_REFILL_S (default 3.5; seconds per token)
|
||||
# EMSAL_RATE_MAX_WAIT_S (default 8.0; max local wait before a structured 429)
|
||||
_DEFAULT_CAPACITY = int(os.getenv("EMSAL_RATE_CAPACITY", "1"))
|
||||
_DEFAULT_REFILL_S = float(os.getenv("EMSAL_RATE_REFILL_S", "3.5"))
|
||||
_DEFAULT_MAX_WAIT_S = float(os.getenv("EMSAL_RATE_MAX_WAIT_S", "8.0"))
|
||||
|
||||
def __init__(self, request_timeout: float = 30.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
@@ -38,6 +115,27 @@ class EmsalApiClient:
|
||||
timeout=request_timeout,
|
||||
verify=False # As per user's original FastAPI code
|
||||
)
|
||||
self._bucket = _TokenBucket(
|
||||
capacity=self._DEFAULT_CAPACITY,
|
||||
refill_per_s=1.0 / self._DEFAULT_REFILL_S,
|
||||
)
|
||||
|
||||
def _handle_429(self, response: httpx.Response, op: str) -> None:
|
||||
"""Apply back-pressure to the shared bucket based on Retry-After.
|
||||
|
||||
Emsal returns 429 as an HTML error page with no Retry-After header, so
|
||||
the 30s fallback almost always applies."""
|
||||
retry_after_raw = response.headers.get("Retry-After", "")
|
||||
try:
|
||||
retry_after = float(retry_after_raw)
|
||||
except (TypeError, ValueError):
|
||||
retry_after = 30.0
|
||||
# Cap penalty so a hostile/buggy server can't freeze us indefinitely.
|
||||
retry_after = max(1.0, min(retry_after, 60.0))
|
||||
self._bucket.penalize_until(time.monotonic() + retry_after + 0.5)
|
||||
logger.warning(
|
||||
f"EmsalApiClient: 429 on {op}; bucket paused {retry_after + 0.5:.1f}s"
|
||||
)
|
||||
|
||||
async def search_detailed_decisions(
|
||||
self,
|
||||
@@ -64,7 +162,11 @@ class EmsalApiClient:
|
||||
pageNumber=params.page_number
|
||||
)
|
||||
|
||||
final_payload = {"data": data_for_api_payload.model_dump(by_alias=True, exclude_none=True)}
|
||||
# Create request dict and remove empty string fields to avoid API issues
|
||||
payload_dict = data_for_api_payload.model_dump(by_alias=True, exclude_none=True)
|
||||
# Remove empty string fields that might cause API issues
|
||||
cleaned_payload = {k: v for k, v in payload_dict.items() if v != ""}
|
||||
final_payload = {"data": cleaned_payload}
|
||||
|
||||
logger.info(f"EmsalApiClient: Performing DETAILED search with payload: {final_payload}")
|
||||
return await self._execute_api_search(self.DETAILED_SEARCH_ENDPOINT, final_payload)
|
||||
@@ -72,7 +174,10 @@ class EmsalApiClient:
|
||||
async def _execute_api_search(self, endpoint: str, payload: Dict) -> EmsalApiResponse:
|
||||
"""Helper method to execute search POST request and process response for Emsal."""
|
||||
try:
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.post(endpoint, json=payload)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, "search")
|
||||
response.raise_for_status()
|
||||
response_json_data = response.json()
|
||||
logger.debug(f"EmsalApiClient: Raw API response from {endpoint}: {response_json_data}")
|
||||
@@ -114,22 +219,18 @@ class EmsalApiClient:
|
||||
html_input_for_markdown = content
|
||||
|
||||
markdown_text = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_input_for_markdown.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_file:
|
||||
tmp_file.write(html_input_for_markdown)
|
||||
temp_file_path = tmp_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
conversion_result = md_converter.convert(html_stream)
|
||||
markdown_text = conversion_result.text_content
|
||||
logger.info("EmsalApiClient: HTML to Markdown conversion successful.")
|
||||
except Exception as e:
|
||||
logger.error(f"EmsalApiClient: Error during MarkItDown HTML to Markdown conversion for Emsal: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
|
||||
return markdown_text
|
||||
|
||||
@@ -143,7 +244,10 @@ class EmsalApiClient:
|
||||
logger.info(f"EmsalApiClient: Fetching Emsal document for Markdown (ID: {id}) from {source_url}")
|
||||
|
||||
try:
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.get(document_api_url)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, f"document {id}")
|
||||
response.raise_for_status()
|
||||
|
||||
# Emsal /getDokuman returns JSON with HTML in 'data' field (confirmed by user example)
|
||||
@@ -154,7 +258,7 @@ class EmsalApiClient:
|
||||
logger.warning(f"EmsalApiClient: Received empty or non-string HTML in 'data' field for Emsal ID {id}.")
|
||||
return EmsalDocumentMarkdown(id=id, markdown_content=None, source_url=source_url)
|
||||
|
||||
markdown_content = self._clean_html_and_convert_to_markdown_emsal(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._clean_html_and_convert_to_markdown_emsal, html_content_from_api)
|
||||
|
||||
return EmsalDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
+28
-28
@@ -12,12 +12,12 @@ class EmsalDetailedSearchRequestData(BaseModel):
|
||||
"""
|
||||
arananKelime: Optional[str] = ""
|
||||
|
||||
Bam_Hukuk_Mahkemeleri: Optional[str] = Field(None, alias="Bam Hukuk Mahkemeleri")
|
||||
Hukuk_Mahkemeleri: Optional[str] = Field(None, alias="Hukuk Mahkemeleri")
|
||||
Bam_Hukuk_Mahkemeleri: str = Field("", alias="Bam Hukuk Mahkemeleri")
|
||||
Hukuk_Mahkemeleri: str = Field("", alias="Hukuk Mahkemeleri")
|
||||
# Add other specific court type fields from the form if they are separate keys in payload
|
||||
# E.g., "Ceza Mahkemeleri", "İdari Mahkemeler" etc.
|
||||
|
||||
birimHukukMah: Optional[str] = Field("", description="List of selected Regional Civil Chambers (Bölge Hukuk Mahkemeleri), '+' separated.")
|
||||
birimHukukMah: Optional[str] = Field("", description="Regional chambers (+ separated)")
|
||||
|
||||
esasYil: Optional[str] = ""
|
||||
esasIlkSiraNo: Optional[str] = ""
|
||||
@@ -36,42 +36,42 @@ class EmsalDetailedSearchRequestData(BaseModel):
|
||||
|
||||
class EmsalSearchRequest(BaseModel): # This is the model the MCP tool will accept
|
||||
"""Model for Emsal detailed search request, with user-friendly field names."""
|
||||
keyword: Optional[str] = Field(None, description="Keyword (Anahtar Kelime) to search.")
|
||||
keyword: str = Field("", description="Keyword")
|
||||
|
||||
selected_bam_civil_court: Optional[str] = Field(None, description="Selected BAM Civil Court (Seçilen BAM Hukuk Mahkemesi) (maps to 'Bam Hukuk Mahkemeleri' payload key).")
|
||||
selected_civil_court: Optional[str] = Field(None, description="Selected Civil Court (Seçilen Hukuk Mahkemesi) (maps to 'Hukuk Mahkemeleri' payload key).")
|
||||
selected_regional_civil_chambers: Optional[List[str]] = Field(default_factory=list, description="Selected Regional Civil Chambers (Seçilen Bölge Hukuk Daireleri) (for 'birimHukukMah', joined by '+').")
|
||||
selected_bam_civil_court: str = Field("", description="BAM Civil Court")
|
||||
selected_civil_court: str = Field("", description="Civil Court")
|
||||
selected_regional_civil_chambers: List[str] = Field(default_factory=list, description="Regional chambers")
|
||||
|
||||
case_year_esas: Optional[str] = Field(None, description="Case year (Dava Yılı) for 'Esas No'.")
|
||||
case_start_seq_esas: Optional[str] = Field(None, description="Starting sequence (Başlangıç Sırası) for 'Esas No'.")
|
||||
case_end_seq_esas: Optional[str] = Field(None, description="Ending sequence (Bitiş Sırası) for 'Esas No'.")
|
||||
case_year_esas: str = Field("", description="Case year")
|
||||
case_start_seq_esas: str = Field("", description="Start case no")
|
||||
case_end_seq_esas: str = Field("", description="End case no")
|
||||
|
||||
decision_year_karar: Optional[str] = Field(None, description="Decision year (Karar Yılı) for 'Karar No'.")
|
||||
decision_start_seq_karar: Optional[str] = Field(None, description="Starting sequence (Başlangıç Sırası) for 'Karar No'.")
|
||||
decision_end_seq_karar: Optional[str] = Field(None, description="Ending sequence (Bitiş Sırası) for 'Karar No'.")
|
||||
decision_year_karar: str = Field("", description="Decision year")
|
||||
decision_start_seq_karar: str = Field("", description="Start decision no")
|
||||
decision_end_seq_karar: str = Field("", description="End decision no")
|
||||
|
||||
start_date: Optional[str] = Field(None, description="Start date (Başlangıç Tarihi) for decision (DD.MM.YYYY).")
|
||||
end_date: Optional[str] = Field(None, description="End date (Bitiş Tarihi) for decision (DD.MM.YYYY).")
|
||||
start_date: str = Field("", description="Start date (DD.MM.YYYY)")
|
||||
end_date: str = Field("", description="End date (DD.MM.YYYY)")
|
||||
|
||||
sort_criteria: str = Field("1", description="Sorting criteria (Sıralama Kriteri) (e.g., 1: Esas No).")
|
||||
sort_direction: str = Field("desc", description="Sorting direction (Sıralama Yönü) ('asc' or 'desc').")
|
||||
sort_criteria: str = Field("1", description="Sort by")
|
||||
sort_direction: str = Field("desc", description="Direction")
|
||||
|
||||
page_number: int = Field(default=1, ge=1)
|
||||
page_size: int = Field(default=10, ge=1, le=100)
|
||||
page_size: int = Field(default=10, ge=1, le=10)
|
||||
|
||||
|
||||
class EmsalApiDecisionEntry(BaseModel):
|
||||
"""Model for an individual decision entry from the Emsal API search response."""
|
||||
id: str
|
||||
daire: Optional[str] = Field(None, description="The chamber/court (Daire/Mahkeme) that made the decision.")
|
||||
esasNo: Optional[str] = Field(None)
|
||||
kararNo: Optional[str] = Field(None)
|
||||
kararTarihi: Optional[str] = Field(None)
|
||||
arananKelime: Optional[str] = Field(None, description="Matched keyword (Aranan Kelime) from the search.")
|
||||
durum: Optional[str] = Field(None, description="Status (Durum) of the decision (e.g., 'KESİNLEŞMEDİ').")
|
||||
daire: str = Field("", description="Chamber")
|
||||
esasNo: str = Field("", description="Case number")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
kararTarihi: str = Field("", description="Decision date")
|
||||
arananKelime: str = Field("", description="Keyword")
|
||||
durum: str = Field("", description="Status")
|
||||
# index: Optional[int] = None # Present in Emsal response, can be added if tool needs it
|
||||
|
||||
document_url: Optional[HttpUrl] = Field(None, description="URL (Belge URL) to the full document, constructed by the client.")
|
||||
document_url: Optional[HttpUrl] = Field(None, description="Document URL")
|
||||
|
||||
model_config = ConfigDict(extra='ignore')
|
||||
|
||||
@@ -80,17 +80,17 @@ class EmsalApiResponseInnerData(BaseModel):
|
||||
data: List[EmsalApiDecisionEntry]
|
||||
recordsTotal: int
|
||||
recordsFiltered: int
|
||||
draw: Optional[int] = Field(None, description="Draw counter (Çizim Sayıcısı) from API, usually for DataTables.")
|
||||
draw: int = Field(0, description="Draw counter (Çizim Sayıcısı) from API, usually for DataTables.")
|
||||
|
||||
class EmsalApiResponse(BaseModel):
|
||||
"""Model for the complete search response from the Emsal API."""
|
||||
data: EmsalApiResponseInnerData
|
||||
data: Optional[EmsalApiResponseInnerData] = None
|
||||
metadata: Optional[Dict[str, Any]] = Field(None, description="Optional metadata (Meta Veri) from API, if any.")
|
||||
|
||||
class EmsalDocumentMarkdown(BaseModel):
|
||||
"""Model for an Emsal decision document, containing only Markdown content."""
|
||||
id: str
|
||||
markdown_content: Optional[str] = Field(None, description="The decision content (Karar İçeriği) converted to Markdown.")
|
||||
markdown_content: str = Field("", description="The decision content (Karar İçeriği) converted to Markdown.")
|
||||
source_url: HttpUrl
|
||||
|
||||
class CompactEmsalSearchResult(BaseModel):
|
||||
|
||||
@@ -1,34 +0,0 @@
|
||||
# fly.toml app configuration file generated for yargi-mcp on 2025-06-29T00:23:47+03:00
|
||||
#
|
||||
# See https://fly.io/docs/reference/configuration/ for information about how to use this file.
|
||||
#
|
||||
|
||||
app = 'yargi-mcp'
|
||||
primary_region = 'fra'
|
||||
|
||||
[env]
|
||||
ENABLE_AUTH = "true"
|
||||
HOST = "0.0.0.0"
|
||||
PORT = "8000"
|
||||
LOG_LEVEL = "info"
|
||||
|
||||
[build]
|
||||
|
||||
[http_service]
|
||||
internal_port = 8000
|
||||
force_https = true
|
||||
auto_stop_machines = 'stop'
|
||||
auto_start_machines = true
|
||||
min_machines_running = 0
|
||||
processes = ['app']
|
||||
|
||||
[[vm]]
|
||||
memory = '1gb'
|
||||
cpu_kind = 'shared'
|
||||
cpus = 1
|
||||
|
||||
[checks.http_health] # keep MCP /health live
|
||||
type = "http"
|
||||
interval = "30s"
|
||||
timeout = "10s"
|
||||
path = "/health"
|
||||
@@ -0,0 +1 @@
|
||||
# gib_mcp_module/__init__.py
|
||||
@@ -0,0 +1,355 @@
|
||||
# gib_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
import io
|
||||
import logging
|
||||
import math
|
||||
from typing import Optional, Any, Dict
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
GibSearchRequest,
|
||||
GibOzelgeSummary,
|
||||
GibSearchResult,
|
||||
GibDocumentMarkdown,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
|
||||
class GibApiClient:
|
||||
"""
|
||||
API client for searching and retrieving GİB özelgeler (Turkish Revenue
|
||||
Administration tax rulings) via the public gib.gov.tr JSON API.
|
||||
|
||||
The endpoint is a single POST list endpoint; document retrieval is done
|
||||
by filtering the same endpoint with an exact `id`.
|
||||
"""
|
||||
|
||||
BASE_URL = "https://gib.gov.tr/api"
|
||||
LIST_PATH = "/gibportal/mevzuat/ozelge/list"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
# Fixed filter values required by the backend
|
||||
_REQUIRED_STATUS = 2
|
||||
_REQUIRED_DELETED = False
|
||||
_REQUIRED_KTYPE = 99 # ktype=99 selects özelge
|
||||
_SORT_FIELD = "ozelgeTarih"
|
||||
_SORT_TYPE = "DESC"
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en;q=0.7",
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": "Mozilla/5.0 (compatible; yargi-mcp/1.0; +https://github.com/saidsurucu/yargi-mcp)",
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_date(value: str, end_of_day: bool = False) -> Optional[str]:
|
||||
"""
|
||||
Accept 'YYYY-MM-DD' or full ISO 8601; always return full ISO 8601.
|
||||
|
||||
GİB backend rejects date-only strings.
|
||||
"""
|
||||
if not value:
|
||||
return None
|
||||
v = value.strip()
|
||||
if not v:
|
||||
return None
|
||||
# Already ISO with time component
|
||||
if "T" in v:
|
||||
return v
|
||||
# Simple YYYY-MM-DD - expand to start/end of day
|
||||
suffix = "T23:59:59.999Z" if end_of_day else "T00:00:00.000Z"
|
||||
return f"{v}{suffix}"
|
||||
|
||||
def _build_search_body(self, params: GibSearchRequest) -> Dict[str, Any]:
|
||||
body: Dict[str, Any] = {
|
||||
"status": self._REQUIRED_STATUS,
|
||||
"deleted": self._REQUIRED_DELETED,
|
||||
"ktype": self._REQUIRED_KTYPE,
|
||||
}
|
||||
|
||||
keywords = params.keywords.strip()
|
||||
kanun_no = params.kanunNo.strip()
|
||||
# Frontend sets title/kanunNo/description to the SAME value; the backend
|
||||
# ORs across them. If the caller supplies both, combine them so kanun_no
|
||||
# still biases toward ruling text, while keywords remain primary.
|
||||
search_term = keywords or kanun_no
|
||||
if keywords and kanun_no and kanun_no not in keywords:
|
||||
search_term = f"{keywords} {kanun_no}"
|
||||
if search_term:
|
||||
body["title"] = search_term
|
||||
body["kanunNo"] = search_term
|
||||
body["description"] = search_term
|
||||
|
||||
if params.ozelgeNo.strip():
|
||||
body["ozelgeNo"] = params.ozelgeNo.strip()
|
||||
|
||||
if params.kanunId and params.kanunId > 0:
|
||||
body["kanunIds"] = [params.kanunId]
|
||||
|
||||
start_iso = self._normalize_date(params.ozelgeStartDate, end_of_day=False)
|
||||
end_iso = self._normalize_date(params.ozelgeEndDate, end_of_day=True)
|
||||
if start_iso:
|
||||
body["ozelgeStartDate"] = start_iso
|
||||
if end_iso:
|
||||
body["ozelgeEndDate"] = end_iso
|
||||
|
||||
return body
|
||||
|
||||
def _build_query_params(self, page_1_indexed: int, page_size: int) -> Dict[str, Any]:
|
||||
# API expects 0-indexed page
|
||||
zero_indexed = max(0, page_1_indexed - 1)
|
||||
return {
|
||||
"page": zero_indexed,
|
||||
"size": page_size,
|
||||
"sortFieldName": self._SORT_FIELD,
|
||||
"sortType": self._SORT_TYPE,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _to_summary(item: Dict[str, Any]) -> Optional[GibOzelgeSummary]:
|
||||
if not isinstance(item, dict):
|
||||
return None
|
||||
raw_id = item.get("id")
|
||||
if raw_id is None:
|
||||
return None
|
||||
try:
|
||||
ozelge_id = int(raw_id)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return GibOzelgeSummary(
|
||||
id=ozelge_id,
|
||||
ozelgeNo=item.get("ozelgeNo"),
|
||||
ozelgeTarih=item.get("ozelgeTarih"),
|
||||
title=item.get("title"),
|
||||
kanunNo=item.get("kanunNo"),
|
||||
kanunTitle=item.get("kanunTitle"),
|
||||
siteLink=item.get("siteLink"),
|
||||
)
|
||||
|
||||
async def search_ozelge(self, params: GibSearchRequest) -> GibSearchResult:
|
||||
"""Search GİB özelgeler."""
|
||||
body = self._build_search_body(params)
|
||||
query = self._build_query_params(params.page, params.pageSize)
|
||||
logger.info(
|
||||
"GibApiClient: search page=%s size=%s body_keys=%s",
|
||||
params.page, params.pageSize, sorted(body.keys()),
|
||||
)
|
||||
|
||||
try:
|
||||
resp = await self.http_client.post(self.LIST_PATH, params=query, json=body)
|
||||
resp.raise_for_status()
|
||||
payload = resp.json()
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error("GibApiClient: HTTP %s during search", e.response.status_code)
|
||||
return GibSearchResult(
|
||||
ozelgeler=[],
|
||||
total_results=0,
|
||||
total_pages=0,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error("GibApiClient: search request failed: %s", e)
|
||||
return GibSearchResult(
|
||||
ozelgeler=[],
|
||||
total_results=0,
|
||||
total_pages=0,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
|
||||
container = (payload or {}).get("resultContainer") or {}
|
||||
raw_items = container.get("content") or []
|
||||
|
||||
summaries = []
|
||||
for raw in raw_items:
|
||||
summary = self._to_summary(raw)
|
||||
if summary is not None:
|
||||
summaries.append(summary)
|
||||
|
||||
total_results = container.get("totalElements") or 0
|
||||
total_pages = container.get("totalPages") or 0
|
||||
try:
|
||||
total_results = int(total_results)
|
||||
except (TypeError, ValueError):
|
||||
total_results = 0
|
||||
try:
|
||||
total_pages = int(total_pages)
|
||||
except (TypeError, ValueError):
|
||||
total_pages = 0
|
||||
|
||||
return GibSearchResult(
|
||||
ozelgeler=summaries,
|
||||
total_results=total_results,
|
||||
total_pages=total_pages,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
|
||||
def _convert_html_to_markdown(self, html_content: str) -> Optional[str]:
|
||||
"""Convert HTML content to Markdown using MarkItDown with BytesIO."""
|
||||
if not html_content:
|
||||
return None
|
||||
try:
|
||||
html_bytes = html_content.encode("utf-8")
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
md_converter = MarkItDown(enable_plugins=False)
|
||||
result = md_converter.convert(html_stream)
|
||||
return result.text_content
|
||||
except Exception as e:
|
||||
logger.error("GibApiClient: HTML→Markdown conversion failed: %s", e)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _build_header_block(item: Dict[str, Any]) -> str:
|
||||
"""Build a small Markdown header block summarising the ruling metadata."""
|
||||
parts = []
|
||||
title = item.get("title")
|
||||
if title:
|
||||
parts.append(f"# {title}")
|
||||
meta_lines = []
|
||||
if item.get("ozelgeNo"):
|
||||
meta_lines.append(f"**Sayı:** {item['ozelgeNo']}")
|
||||
if item.get("ozelgeTarih"):
|
||||
meta_lines.append(f"**Tarih:** {item['ozelgeTarih']}")
|
||||
if item.get("kanunTitle"):
|
||||
kanun_no = item.get("kanunNo")
|
||||
if kanun_no:
|
||||
meta_lines.append(f"**Kanun:** {item['kanunTitle']} ({kanun_no})")
|
||||
else:
|
||||
meta_lines.append(f"**Kanun:** {item['kanunTitle']}")
|
||||
if item.get("siteLink"):
|
||||
meta_lines.append(f"**Kaynak:** {item['siteLink']}")
|
||||
if meta_lines:
|
||||
parts.append("\n".join(meta_lines))
|
||||
return "\n\n".join(parts).strip()
|
||||
|
||||
async def get_ozelge_document(
|
||||
self, ozelge_id: int, page_number: int = 1
|
||||
) -> GibDocumentMarkdown:
|
||||
"""Retrieve a single özelge and return its paginated Markdown form."""
|
||||
logger.info(
|
||||
"GibApiClient: fetching özelge id=%s page=%s", ozelge_id, page_number
|
||||
)
|
||||
|
||||
if not isinstance(ozelge_id, int) or ozelge_id <= 0:
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id if isinstance(ozelge_id, int) else 0,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="ozelge_id must be a positive integer",
|
||||
)
|
||||
|
||||
body = {
|
||||
"status": self._REQUIRED_STATUS,
|
||||
"deleted": self._REQUIRED_DELETED,
|
||||
"ktype": self._REQUIRED_KTYPE,
|
||||
"id": ozelge_id,
|
||||
}
|
||||
query = {"page": 0, "size": 1}
|
||||
|
||||
try:
|
||||
resp = await self.http_client.post(self.LIST_PATH, params=query, json=body)
|
||||
resp.raise_for_status()
|
||||
payload = resp.json()
|
||||
except httpx.HTTPStatusError as e:
|
||||
msg = f"HTTP {e.response.status_code} when fetching özelge {ozelge_id}"
|
||||
logger.error("GibApiClient: %s", msg)
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=msg,
|
||||
)
|
||||
except Exception as e:
|
||||
msg = f"Request failed: {e}"
|
||||
logger.error("GibApiClient: %s", msg)
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=msg,
|
||||
)
|
||||
|
||||
container = (payload or {}).get("resultContainer") or {}
|
||||
content = container.get("content") or []
|
||||
if not content:
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=f"Özelge {ozelge_id} not found",
|
||||
)
|
||||
|
||||
item = content[0] if isinstance(content[0], dict) else {}
|
||||
description_html = item.get("description") or ""
|
||||
markdown_body = (await asyncio.to_thread(self._convert_html_to_markdown, description_html)) or ""
|
||||
header_block = self._build_header_block(item)
|
||||
|
||||
if header_block and markdown_body:
|
||||
full_markdown = f"{header_block}\n\n---\n\n{markdown_body}"
|
||||
else:
|
||||
full_markdown = header_block or markdown_body
|
||||
|
||||
if not full_markdown.strip():
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
ozelge_no=item.get("ozelgeNo"),
|
||||
title=item.get("title"),
|
||||
ozelge_tarih=item.get("ozelgeTarih"),
|
||||
kanun_title=item.get("kanunTitle"),
|
||||
kanun_no=item.get("kanunNo"),
|
||||
site_link=item.get("siteLink"),
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="Document body is empty",
|
||||
)
|
||||
|
||||
total_pages = max(
|
||||
1, math.ceil(len(full_markdown) / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
)
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
start = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end = start + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
chunk = full_markdown[start:end]
|
||||
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
ozelge_no=item.get("ozelgeNo"),
|
||||
title=item.get("title"),
|
||||
ozelge_tarih=item.get("ozelgeTarih"),
|
||||
kanun_title=item.get("kanunTitle"),
|
||||
kanun_no=item.get("kanunNo"),
|
||||
site_link=item.get("siteLink"),
|
||||
markdown_chunk=chunk,
|
||||
current_page=current_page_clamped,
|
||||
total_pages=total_pages,
|
||||
is_paginated=total_pages > 1,
|
||||
error_message=None,
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
if hasattr(self, "http_client") and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("GibApiClient: HTTP client session closed.")
|
||||
@@ -0,0 +1,64 @@
|
||||
# gib_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
class GibSearchRequest(BaseModel):
|
||||
"""
|
||||
Request model for searching GİB özelgeler (Turkish Revenue Administration tax rulings).
|
||||
|
||||
GİB (Gelir İdaresi Başkanlığı) publishes official tax-ruling letters
|
||||
("özelge") responding to taxpayer questions on VAT, income tax,
|
||||
corporate tax, stamp duty, and other tax matters. 18,000+ rulings
|
||||
are searchable via the public gib.gov.tr API.
|
||||
"""
|
||||
keywords: str = Field("", description="Keywords searched across title, kanunNo and description (Turkish)")
|
||||
ozelgeNo: str = Field("", description="Exact özelge reference number (e.g., 'E-40247694-130-15524')")
|
||||
kanunNo: str = Field("", description="Law number filter, e.g. '3065' for KDV")
|
||||
kanunId: int = Field(0, description="Optional numeric law ID filter (0=ignore)")
|
||||
ozelgeStartDate: str = Field("", description="Start date YYYY-MM-DD or full ISO 8601")
|
||||
ozelgeEndDate: str = Field("", description="End date YYYY-MM-DD or full ISO 8601")
|
||||
page: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)")
|
||||
|
||||
|
||||
class GibOzelgeSummary(BaseModel):
|
||||
"""Summary of a single GİB özelge from search results (no full HTML)."""
|
||||
id: int = Field(..., description="Numeric özelge ID for document retrieval")
|
||||
ozelgeNo: Optional[str] = Field(None, description="Official ruling reference number")
|
||||
ozelgeTarih: Optional[str] = Field(None, description="Ruling date (ISO datetime)")
|
||||
title: Optional[str] = Field(None, description="Subject/title of the ruling")
|
||||
kanunNo: Optional[str] = Field(None, description="Law number (e.g., '3065')")
|
||||
kanunTitle: Optional[str] = Field(None, description="Law title (e.g., 'KATMA DEĞER VERGİSİ KANUNU')")
|
||||
siteLink: Optional[str] = Field(None, description="Direct URL to the ruling on gib.gov.tr")
|
||||
|
||||
|
||||
class GibSearchResult(BaseModel):
|
||||
"""Response model for GİB özelge search results."""
|
||||
ozelgeler: List[GibOzelgeSummary] = Field(default_factory=list, description="Matching özelge summaries")
|
||||
total_results: int = Field(0, description="Total number of matching özelgeler across all pages")
|
||||
total_pages: int = Field(0, description="Total number of pages for this query")
|
||||
current_page: int = Field(1, description="Current page (1-indexed)")
|
||||
page_size: int = Field(10, description="Results per page")
|
||||
|
||||
|
||||
class GibDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
GİB özelge document converted to paginated Markdown.
|
||||
|
||||
Long rulings are split into 5000-character chunks; request successive
|
||||
pages via page_number to read the full text.
|
||||
"""
|
||||
ozelge_id: int = Field(..., description="Numeric özelge ID")
|
||||
ozelge_no: Optional[str] = Field(None, description="Official ruling reference number")
|
||||
title: Optional[str] = Field(None, description="Subject/title of the ruling")
|
||||
ozelge_tarih: Optional[str] = Field(None, description="Ruling date (ISO datetime)")
|
||||
kanun_title: Optional[str] = Field(None, description="Related law title")
|
||||
kanun_no: Optional[str] = Field(None, description="Related law number")
|
||||
site_link: Optional[str] = Field(None, description="Direct URL to the ruling on gib.gov.tr")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Current 5000-character Markdown chunk")
|
||||
current_page: int = Field(1, description="Current page number (1-indexed)")
|
||||
total_pages: int = Field(0, description="Total pages for the full Markdown content")
|
||||
is_paginated: bool = Field(False, description="True if split across multiple pages")
|
||||
error_message: Optional[str] = Field(None, description="Populated when retrieval failed")
|
||||
@@ -1,441 +0,0 @@
|
||||
# kik_mcp_module/client.py
|
||||
import asyncio
|
||||
from playwright.async_api import (
|
||||
async_playwright,
|
||||
Page,
|
||||
BrowserContext,
|
||||
Browser,
|
||||
Error as PlaywrightError,
|
||||
TimeoutError as PlaywrightTimeoutError
|
||||
)
|
||||
from bs4 import BeautifulSoup
|
||||
import logging
|
||||
from typing import Dict, Any, List, Optional
|
||||
import urllib.parse
|
||||
import base64 # Base64 için
|
||||
import re
|
||||
import html as html_parser
|
||||
from markitdown import MarkItDown
|
||||
import os
|
||||
import math
|
||||
import tempfile
|
||||
|
||||
from .models import (
|
||||
KikSearchRequest,
|
||||
KikDecisionEntry,
|
||||
KikSearchResult,
|
||||
KikDocumentMarkdown,
|
||||
KikKararTipi
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class KikApiClient:
|
||||
BASE_URL = "https://ekap.kik.gov.tr"
|
||||
SEARCH_PAGE_PATH = "/EKAP/Vatandas/kurulkararsorgu.aspx"
|
||||
FIELD_LOCATORS = {
|
||||
"karar_tipi_radio_group": "input[name='ctl00$ContentPlaceHolder1$kurulKararTip']",
|
||||
"karar_no": "input[name='ctl00$ContentPlaceHolder1$txtKararNo']",
|
||||
"karar_tarihi_baslangic": "input[name='ctl00$ContentPlaceHolder1$etKararTarihBaslangic$EkapTakvimTextBox_etKararTarihBaslangic']",
|
||||
"karar_tarihi_bitis": "input[name='ctl00$ContentPlaceHolder1$etKararTarihBitis$EkapTakvimTextBox_etKararTarihBitis']",
|
||||
"resmi_gazete_sayisi": "input[name='ctl00$ContentPlaceHolder1$txtResmiGazeteSayisi']",
|
||||
"resmi_gazete_tarihi": "input[name='ctl00$ContentPlaceHolder1$etResmiGazeteTarihi$EkapTakvimTextBox_etResmiGazeteTarihi']",
|
||||
"basvuru_konusu_ihale": "input[name='ctl00$ContentPlaceHolder1$txtBasvuruKonusuIhale']",
|
||||
"basvuru_sahibi": "input[name='ctl00$ContentPlaceHolder1$txtSikayetci']",
|
||||
"ihaleyi_yapan_idare": "input[name='ctl00$ContentPlaceHolder1$txtIhaleyiYapanIdare']",
|
||||
"yil": "select[name='ctl00$ContentPlaceHolder1$ddlYil']",
|
||||
"karar_metni": "input[name='ctl00$ContentPlaceHolder1$txtKararMetni']",
|
||||
"search_button_id": "ctl00_ContentPlaceHolder1_btnAra"
|
||||
}
|
||||
RESULTS_TABLE_ID = "grdKurulKararSorguSonuc"
|
||||
NO_RESULTS_MESSAGE_SELECTOR = "div#ctl00_MessageContent1"
|
||||
VALIDATION_SUMMARY_SELECTOR = "div#ctl00_ValidationSummary1"
|
||||
MODAL_CLOSE_BUTTON_SELECTOR = "div#detayPopUp.in a#btnKapatPencere_0.close"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
def __init__(self, request_timeout: float = 60000):
|
||||
self.playwright_instance: Optional[async_playwright] = None
|
||||
self.browser: Optional[Browser] = None
|
||||
self.context: Optional[BrowserContext] = None
|
||||
self.page: Optional[Page] = None
|
||||
self.request_timeout = request_timeout
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def _ensure_playwright_ready(self, force_new_page: bool = False):
|
||||
async with self._lock:
|
||||
browser_recreated = False
|
||||
context_recreated = False
|
||||
if not self.playwright_instance:
|
||||
self.playwright_instance = await async_playwright().start()
|
||||
if not self.browser or not self.browser.is_connected():
|
||||
if self.browser: await self.browser.close()
|
||||
self.browser = await self.playwright_instance.chromium.launch(headless=True)
|
||||
browser_recreated = True
|
||||
if not self.context or browser_recreated:
|
||||
if self.context: await self.context.close()
|
||||
if not self.browser: raise PlaywrightError("Browser not initialized.")
|
||||
self.context = await self.browser.new_context(
|
||||
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/100.0.0.0 Safari/537.36",
|
||||
java_script_enabled=True,
|
||||
)
|
||||
context_recreated = True
|
||||
if not self.page or self.page.is_closed() or force_new_page or context_recreated or browser_recreated:
|
||||
if self.page and not self.page.is_closed(): await self.page.close()
|
||||
if not self.context: raise PlaywrightError("Context is None.")
|
||||
self.page = await self.context.new_page()
|
||||
if not self.page: raise PlaywrightError("Failed to create new page.")
|
||||
self.page.set_default_navigation_timeout(self.request_timeout)
|
||||
self.page.set_default_timeout(self.request_timeout)
|
||||
if not self.page or self.page.is_closed():
|
||||
raise PlaywrightError("Playwright page initialization failed.")
|
||||
logger.debug("_ensure_playwright_ready completed.")
|
||||
|
||||
async def close_client_session(self):
|
||||
async with self._lock:
|
||||
# ... (öncekiyle aynı)
|
||||
if self.page and not self.page.is_closed(): await self.page.close(); self.page = None
|
||||
if self.context: await self.context.close(); self.context = None
|
||||
if self.browser: await self.browser.close(); self.browser = None
|
||||
if self.playwright_instance: await self.playwright_instance.stop(); self.playwright_instance = None
|
||||
logger.info("KikApiClient (Playwright): Resources closed.")
|
||||
|
||||
def _parse_decision_entries_from_soup(self, soup: BeautifulSoup, search_karar_tipi: KikKararTipi) -> List[KikDecisionEntry]:
|
||||
entries: List[KikDecisionEntry] = []
|
||||
table = soup.find("table", {"id": self.RESULTS_TABLE_ID})
|
||||
if not table: return entries
|
||||
rows = table.find_all("tr")
|
||||
for row_idx, row in enumerate(rows):
|
||||
if row_idx < 2: continue
|
||||
cells = row.find_all("td")
|
||||
if len(cells) == 6:
|
||||
try:
|
||||
preview_button_tag = cells[0].find("a", id=re.compile(r"btnOnizle$"))
|
||||
event_target = ""
|
||||
if preview_button_tag and preview_button_tag.has_attr('href'):
|
||||
match = re.search(r"__doPostBack\('([^']*)','([^']*)'\)", preview_button_tag['href'])
|
||||
if match: event_target = match.group(1)
|
||||
karar_no_span = cells[1].find("span", id=re.compile(r"lblKno$"))
|
||||
karar_tarihi_span = cells[2].find("span", id=re.compile(r"lblKtar$"))
|
||||
idare_span = cells[3].find("span", id=re.compile(r"lblIdare$"))
|
||||
basvuru_sahibi_span = cells[4].find("span", id=re.compile(r"lblSikayetci$"))
|
||||
ihale_span = cells[5].find("span", id=re.compile(r"lblIhale$"))
|
||||
if not (event_target and karar_no_span and karar_tarihi_span): continue
|
||||
|
||||
# Karar tipini arama parametresinden alıyoruz, çünkü HTML'de direkt olarak bulunmuyor.
|
||||
entry = KikDecisionEntry(
|
||||
preview_event_target=event_target,
|
||||
kararNo=karar_no_span.get_text(strip=True),
|
||||
karar_tipi=search_karar_tipi, # Arama yapılan karar tipini ekle
|
||||
kararTarihi=karar_tarihi_span.get_text(strip=True),
|
||||
idare=idare_span.get_text(strip=True) if idare_span else None,
|
||||
basvuruSahibi=basvuru_sahibi_span.get_text(strip=True) if basvuru_sahibi_span else None,
|
||||
ihaleKonusu=ihale_span.get_text(strip=True) if ihale_span else None,
|
||||
)
|
||||
entries.append(entry)
|
||||
except Exception as e:
|
||||
logger.error(f"Error parsing a KIK decision entry row: {e}", exc_info=True)
|
||||
return entries
|
||||
|
||||
def _parse_total_records_from_soup(self, soup: BeautifulSoup) -> int:
|
||||
# ... (öncekiyle aynı) ...
|
||||
try:
|
||||
pager_div = soup.find("div", class_="gridToplamSayi")
|
||||
if pager_div:
|
||||
match = re.search(r"Toplam Kayıt Sayısı:(\d+)", pager_div.get_text(strip=True))
|
||||
if match: return int(match.group(1))
|
||||
except: pass
|
||||
return 0
|
||||
|
||||
def _parse_current_page_from_soup(self, soup: BeautifulSoup) -> int:
|
||||
# ... (öncekiyle aynı) ...
|
||||
try:
|
||||
pager_div = soup.find("div", class_="sayfalama")
|
||||
if pager_div:
|
||||
active_page_span = pager_div.find("span", class_="active")
|
||||
if active_page_span: return int(active_page_span.get_text(strip=True))
|
||||
except: pass
|
||||
return 1
|
||||
|
||||
async def search_decisions(self, search_params: KikSearchRequest) -> KikSearchResult:
|
||||
await self._ensure_playwright_ready()
|
||||
page = self.page
|
||||
search_url = f"{self.BASE_URL}{self.SEARCH_PAGE_PATH}"
|
||||
try:
|
||||
if page.url != search_url:
|
||||
await page.goto(search_url, wait_until="networkidle", timeout=self.request_timeout)
|
||||
search_button_selector = f"a[id='{self.FIELD_LOCATORS['search_button_id']}']"
|
||||
await page.wait_for_selector(search_button_selector, state="visible", timeout=self.request_timeout)
|
||||
|
||||
current_karar_tipi_value = search_params.karar_tipi.value
|
||||
radio_locator_selector = f"{self.FIELD_LOCATORS['karar_tipi_radio_group']}[value='{current_karar_tipi_value}']"
|
||||
if not await page.locator(radio_locator_selector).is_checked():
|
||||
js_target_radio = f"ctl00$ContentPlaceHolder1${current_karar_tipi_value}"
|
||||
async with page.expect_navigation(wait_until="networkidle", timeout=self.request_timeout):
|
||||
await page.evaluate(f"javascript:__doPostBack('{js_target_radio}','')")
|
||||
await page.wait_for_timeout(1000)
|
||||
|
||||
async def fill_if_value(selector_key: str, value: Optional[str]):
|
||||
if value is not None: await page.fill(self.FIELD_LOCATORS[selector_key], value)
|
||||
|
||||
# Karar No'yu KİK sitesine göndermeden önce '_' -> '/' dönüşümü yap
|
||||
karar_no_for_kik_form = None
|
||||
if search_params.karar_no: # search_params.karar_no Claude'dan '_' ile gelmiş olabilir
|
||||
karar_no_for_kik_form = search_params.karar_no.replace('_', '/')
|
||||
logger.info(f"Using karar_no '{karar_no_for_kik_form}' (transformed from '{search_params.karar_no}') for KIK form.")
|
||||
|
||||
await fill_if_value('karar_metni', search_params.karar_metni)
|
||||
await fill_if_value('karar_no', karar_no_for_kik_form) # Dönüştürülmüş halini kullan
|
||||
# ... (diğer fill_if_value çağrıları aynı) ...
|
||||
await fill_if_value('karar_tarihi_baslangic', search_params.karar_tarihi_baslangic)
|
||||
await fill_if_value('karar_tarihi_bitis', search_params.karar_tarihi_bitis)
|
||||
await fill_if_value('resmi_gazete_sayisi', search_params.resmi_gazete_sayisi)
|
||||
await fill_if_value('resmi_gazete_tarihi', search_params.resmi_gazete_tarihi)
|
||||
await fill_if_value('basvuru_konusu_ihale', search_params.basvuru_konusu_ihale)
|
||||
await fill_if_value('basvuru_sahibi', search_params.basvuru_sahibi)
|
||||
await fill_if_value('ihaleyi_yapan_idare', search_params.ihaleyi_yapan_idare)
|
||||
|
||||
if search_params.yil:
|
||||
await page.select_option(self.FIELD_LOCATORS['yil'], value=search_params.yil)
|
||||
|
||||
action_is_search_button_click = (search_params.page == 1)
|
||||
event_target_for_submit: str
|
||||
if action_is_search_button_click:
|
||||
event_target_for_submit = self.FIELD_LOCATORS['search_button_id']
|
||||
else: # Pagination
|
||||
page_link_ctl_number = search_params.page + 2
|
||||
event_target_for_submit = f"ctl00$ContentPlaceHolder1$grdKurulKararSorguSonuc$ctl14$ctl{page_link_ctl_number:02d}"
|
||||
|
||||
try:
|
||||
async with page.expect_navigation(wait_until="networkidle", timeout=self.request_timeout):
|
||||
if action_is_search_button_click:
|
||||
await page.locator(search_button_selector).click()
|
||||
else:
|
||||
await page.evaluate(f"javascript:__doPostBack('{event_target_for_submit}','')")
|
||||
except PlaywrightTimeoutError:
|
||||
await page.wait_for_timeout(2000)
|
||||
|
||||
results_table_dom_selector = f"table#{self.RESULTS_TABLE_ID}"
|
||||
try:
|
||||
await page.wait_for_selector(results_table_dom_selector, timeout=30000, state="attached")
|
||||
await page.wait_for_timeout(2000)
|
||||
except PlaywrightTimeoutError:
|
||||
logger.warning(f"Timeout waiting for results table '{results_table_dom_selector}'.")
|
||||
|
||||
html_content = await page.content()
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
# ... (hata ve sonuç yok mesajı kontrolü aynı) ...
|
||||
validation_summary_tag = soup.find("div", id=self.VALIDATION_SUMMARY_SELECTOR.split('[')[0].split(':')[0])
|
||||
if validation_summary_tag and validation_summary_tag.get_text(strip=True) and \
|
||||
("display: none" not in validation_summary_tag.get("style", "").lower() if validation_summary_tag.has_attr("style") else True) and \
|
||||
validation_summary_tag.get_text(strip=True) != "":
|
||||
return KikSearchResult(decisions=[], total_records=0, current_page=search_params.page)
|
||||
message_content_div = soup.find("div", id=self.NO_RESULTS_MESSAGE_SELECTOR.split(':')[0])
|
||||
if message_content_div and "kayıt bulunamamıştır" in message_content_div.get_text(strip=True).lower():
|
||||
return KikSearchResult(decisions=[], total_records=0, current_page=1)
|
||||
|
||||
# _parse_decision_entries_from_soup'a arama yapılan karar_tipi'ni gönder
|
||||
decisions = self._parse_decision_entries_from_soup(soup, search_params.karar_tipi)
|
||||
total_records = self._parse_total_records_from_soup(soup)
|
||||
current_page_from_html = self._parse_current_page_from_soup(soup)
|
||||
return KikSearchResult(decisions=decisions, total_records=total_records, current_page=current_page_from_html)
|
||||
except Exception as e:
|
||||
logger.error(f"Error during KIK decision search: {e}", exc_info=True)
|
||||
return KikSearchResult(decisions=[], current_page=search_params.page)
|
||||
|
||||
def _clean_html_for_markdown(self, html_content: str) -> str:
|
||||
# ... (öncekiyle aynı) ...
|
||||
if not html_content: return ""
|
||||
return html_parser.unescape(html_content)
|
||||
|
||||
def _convert_html_to_markdown_internal(self, html_fragment: str) -> Optional[str]:
|
||||
# ... (öncekiyle aynı) ...
|
||||
if not html_fragment: return None
|
||||
cleaned_html = self._clean_html_for_markdown(html_fragment)
|
||||
markdown_output = None; temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown(enable_plugins=True, remove_alt_whitespace=True, keep_underline=True)
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_html_file:
|
||||
tmp_html_file.write(cleaned_html); temp_file_path = tmp_html_file.name
|
||||
markdown_output = md_converter.convert(temp_file_path).text_content
|
||||
if markdown_output: markdown_output = re.sub(r'\n{3,}', '\n\n', markdown_output).strip()
|
||||
except Exception as e: logger.error(f"MarkItDown conversion error: {e}", exc_info=True)
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path): os.remove(temp_file_path)
|
||||
return markdown_output
|
||||
|
||||
|
||||
async def get_decision_document_as_markdown(
|
||||
self,
|
||||
karar_id_b64: str,
|
||||
page_number: int = 1
|
||||
) -> KikDocumentMarkdown:
|
||||
await self._ensure_playwright_ready()
|
||||
# Bu metodun kendi içinde yeni bir 'page' nesnesi ('doc_page_for_content') kullanacağını unutmayın,
|
||||
# ana 'self.page' arama sonuçları sayfasında kalır.
|
||||
current_main_page = self.page # Ana arama sonuçları sayfasını referans alalım
|
||||
|
||||
try:
|
||||
decoded_key = base64.b64decode(karar_id_b64.encode('utf-8')).decode('utf-8')
|
||||
karar_tipi_value, karar_no_for_search = decoded_key.split('|', 1)
|
||||
original_karar_tipi = KikKararTipi(karar_tipi_value)
|
||||
logger.info(f"KIK Get Detail: Decoded karar_id '{karar_id_b64}' to Karar Tipi: {original_karar_tipi.value}, Karar No: {karar_no_for_search}. Requested Markdown Page: {page_number}")
|
||||
except Exception as e_decode:
|
||||
logger.error(f"Invalid karar_id format. Could not decode Base64 or split: {karar_id_b64}. Error: {e_decode}")
|
||||
return KikDocumentMarkdown(retrieved_with_karar_id=karar_id_b64, error_message="Invalid karar_id format.", current_page=page_number)
|
||||
|
||||
default_error_response_data = {
|
||||
"retrieved_with_karar_id": karar_id_b64,
|
||||
"retrieved_karar_no": karar_no_for_search,
|
||||
"retrieved_karar_tipi": original_karar_tipi,
|
||||
"error_message": "An unspecified error occurred.",
|
||||
"current_page": page_number, "total_pages": 1, "is_paginated": False
|
||||
}
|
||||
|
||||
# Ana arama sayfasında olduğumuzdan emin olalım
|
||||
if self.SEARCH_PAGE_PATH not in current_main_page.url:
|
||||
logger.info(f"Not on search page ({current_main_page.url}). Navigating to {self.SEARCH_PAGE_PATH} before targeted search for document.")
|
||||
await current_main_page.goto(f"{self.BASE_URL}{self.SEARCH_PAGE_PATH}", wait_until="networkidle", timeout=self.request_timeout)
|
||||
await current_main_page.wait_for_selector(f"a[id='{self.FIELD_LOCATORS['search_button_id']}']", state="visible", timeout=self.request_timeout)
|
||||
|
||||
targeted_search_params = KikSearchRequest(
|
||||
karar_no=karar_no_for_search,
|
||||
karar_tipi=original_karar_tipi,
|
||||
page=1
|
||||
)
|
||||
logger.info(f"Performing targeted search for Karar No: {karar_no_for_search}")
|
||||
# search_decisions kendi içinde _ensure_playwright_ready çağırır ve self.page'i kullanır.
|
||||
# Bu, current_main_page ile aynı olmalı.
|
||||
search_results = await self.search_decisions(targeted_search_params)
|
||||
|
||||
if not search_results.decisions:
|
||||
default_error_response_data["error_message"] = f"Decision with Karar No '{karar_no_for_search}' (Tipi: {original_karar_tipi.value}) not found by internal search."
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
decision_to_fetch = None
|
||||
for dec_entry in search_results.decisions:
|
||||
if dec_entry.karar_no_str == karar_no_for_search and dec_entry.karar_tipi == original_karar_tipi:
|
||||
decision_to_fetch = dec_entry
|
||||
break
|
||||
|
||||
if not decision_to_fetch:
|
||||
default_error_response_data["error_message"] = f"Karar No '{karar_no_for_search}' (Tipi: {original_karar_tipi.value}) not present with an exact match in first page of targeted search results."
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
decision_preview_event_target = decision_to_fetch.preview_event_target
|
||||
logger.info(f"Found target decision. Using preview_event_target: {decision_preview_event_target} for Karar No: {decision_to_fetch.karar_no_str}")
|
||||
|
||||
iframe_document_url_str = None
|
||||
karar_id_param_from_url_on_doc_page = None
|
||||
document_html_content = ""
|
||||
|
||||
try:
|
||||
logger.info(f"Evaluating __doPostBack on main page to show modal for: {decision_preview_event_target}")
|
||||
# Bu evaluate, self.page (yani current_main_page) üzerinde çalışır
|
||||
await current_main_page.evaluate(f"javascript:__doPostBack('{decision_preview_event_target}','')")
|
||||
await current_main_page.wait_for_timeout(1000)
|
||||
logger.info(f"Executed __doPostBack for {decision_preview_event_target} on main page.")
|
||||
|
||||
iframe_selector = "iframe#iframe_detayPopUp"
|
||||
modal_visible_selector = "div#detayPopUp.in"
|
||||
|
||||
try:
|
||||
logger.info(f"Waiting for modal '{modal_visible_selector}' to be visible and iframe '{iframe_selector}' src to be populated on main page...")
|
||||
await current_main_page.wait_for_function(
|
||||
f"""
|
||||
() => {{
|
||||
const modal = document.querySelector('{modal_visible_selector}');
|
||||
const iframe = document.querySelector('{iframe_selector}');
|
||||
const modalIsTrulyVisible = modal && (window.getComputedStyle(modal).display !== 'none');
|
||||
return modalIsTrulyVisible &&
|
||||
iframe && iframe.getAttribute('src') &&
|
||||
iframe.getAttribute('src').includes('KurulKararGoster.aspx');
|
||||
}}
|
||||
""",
|
||||
timeout=self.request_timeout / 2
|
||||
)
|
||||
iframe_src_value = await current_main_page.locator(iframe_selector).get_attribute("src")
|
||||
logger.info(f"Iframe src populated: {iframe_src_value}")
|
||||
|
||||
except PlaywrightTimeoutError:
|
||||
logger.warning(f"Timeout waiting for KIK iframe src for {decision_preview_event_target}. Trying to parse from static content after presumed update.")
|
||||
html_after_postback = await current_main_page.content()
|
||||
# ... (fallback parsing öncekiyle aynı, default_error_response_data set edilir ve return edilir) ...
|
||||
soup_after_postback = BeautifulSoup(html_after_postback, "html.parser")
|
||||
detay_popup_div = soup_after_postback.find("div", {"id": "detayPopUp", "class": re.compile(r"\bin\b")})
|
||||
if not detay_popup_div: detay_popup_div = soup_after_postback.find("div", {"id": "detayPopUp", "style": re.compile(r"display:\s*block", re.I)})
|
||||
iframe_tag = detay_popup_div.find("iframe", {"id": "iframe_detayPopUp"}) if detay_popup_div else None
|
||||
if iframe_tag and iframe_tag.has_attr("src") and iframe_tag["src"]: iframe_src_value = iframe_tag["src"]
|
||||
else:
|
||||
default_error_response_data["error_message"]="Timeout or failure finding decision content iframe URL after postback."
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
if not iframe_src_value or not iframe_src_value.strip():
|
||||
default_error_response_data["error_message"]="Extracted iframe URL for decision content is empty."
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
# iframe_src_value göreceli bir URL ise, ana sayfanın URL'si ile birleştir
|
||||
iframe_document_url_str = urllib.parse.urljoin(current_main_page.url, iframe_src_value)
|
||||
logger.info(f"Constructed absolute iframe_document_url_str for goto: {iframe_document_url_str}") # Log this absolute URL
|
||||
default_error_response_data["source_url"] = iframe_document_url_str
|
||||
|
||||
parsed_url = urllib.parse.urlparse(iframe_document_url_str)
|
||||
query_params = urllib.parse.parse_qs(parsed_url.query)
|
||||
karar_id_param_from_url_on_doc_page = query_params.get("KararId", [None])[0]
|
||||
default_error_response_data["karar_id_param_from_url"] = karar_id_param_from_url_on_doc_page
|
||||
if not karar_id_param_from_url_on_doc_page:
|
||||
default_error_response_data["error_message"]="KararId (KIK internal ID) not found in extracted iframe URL."
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
logger.info(f"Fetching KIK decision content from iframe URL using a new Playwright page: {iframe_document_url_str}")
|
||||
|
||||
doc_page_for_content = await self.context.new_page()
|
||||
try:
|
||||
# `goto` metoduna MUTLAK URL verilmeli. Loglanan URL'nin mutlak olduğundan emin olalım.
|
||||
await doc_page_for_content.goto(iframe_document_url_str, wait_until="domcontentloaded", timeout=self.request_timeout)
|
||||
document_html_content = await doc_page_for_content.content()
|
||||
except Exception as e_doc_page:
|
||||
logger.error(f"Error navigating or getting content from doc_page ({iframe_document_url_str}): {e_doc_page}")
|
||||
if doc_page_for_content and not doc_page_for_content.is_closed(): await doc_page_for_content.close()
|
||||
default_error_response_data["error_message"]=f"Failed to load decision detail page: {e_doc_page}"
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
finally:
|
||||
if doc_page_for_content and not doc_page_for_content.is_closed():
|
||||
await doc_page_for_content.close()
|
||||
|
||||
soup_decision_detail = BeautifulSoup(document_html_content, "html.parser")
|
||||
karar_content_span = soup_decision_detail.find("span", {"id": "ctl00_ContentPlaceHolder1_lblKarar"})
|
||||
actual_decision_html = karar_content_span.decode_contents() if karar_content_span else document_html_content
|
||||
full_markdown_content = self._convert_html_to_markdown_internal(actual_decision_html)
|
||||
|
||||
if not full_markdown_content:
|
||||
default_error_response_data["error_message"]="Markdown conversion failed or returned empty content."
|
||||
try:
|
||||
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=1000):
|
||||
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
|
||||
except: pass
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
|
||||
content_length = len(full_markdown_content); total_pages = math.ceil(content_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE) or 1
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
start_index = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
markdown_chunk = full_markdown_content[start_index : start_index + self.DOCUMENT_MARKDOWN_CHUNK_SIZE]
|
||||
|
||||
try:
|
||||
if await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).is_visible(timeout=2000):
|
||||
await current_main_page.locator(self.MODAL_CLOSE_BUTTON_SELECTOR).click()
|
||||
await current_main_page.wait_for_selector(f"div#detayPopUp:not(.in)", timeout=5000)
|
||||
except: pass
|
||||
|
||||
return KikDocumentMarkdown(
|
||||
retrieved_with_karar_id=karar_id_b64,
|
||||
retrieved_karar_no=karar_no_for_search,
|
||||
retrieved_karar_tipi=original_karar_tipi,
|
||||
kararIdParam=karar_id_param_from_url_on_doc_page,
|
||||
markdown_chunk=markdown_chunk, source_url=iframe_document_url_str,
|
||||
current_page=current_page_clamped, total_pages=total_pages,
|
||||
is_paginated=(total_pages > 1), full_content_char_count=content_length
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Error in get_decision_document_as_markdown for Karar ID {karar_id_b64}: {e}", exc_info=True)
|
||||
default_error_response_data["error_message"] = f"General error: {str(e)}"
|
||||
return KikDocumentMarkdown(**default_error_response_data)
|
||||
@@ -0,0 +1,507 @@
|
||||
# kik_mcp_module/client_v2.py
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import httpx
|
||||
import logging
|
||||
import uuid
|
||||
import ssl
|
||||
import os
|
||||
from typing import Optional
|
||||
from datetime import datetime
|
||||
|
||||
# Cryptography imports for AES-256-CBC encryption of document IDs
|
||||
try:
|
||||
from cryptography.hazmat.primitives.ciphers import Cipher, algorithms, modes
|
||||
from cryptography.hazmat.backends import default_backend
|
||||
HAS_CRYPTOGRAPHY = True
|
||||
except ImportError:
|
||||
HAS_CRYPTOGRAPHY = False
|
||||
|
||||
from .models_v2 import (
|
||||
KikV2DecisionType, KikV2SearchPayload, KikV2SearchPayloadDk, KikV2SearchPayloadMk,
|
||||
KikV2RequestData, KikV2QueryRequest, KikV2KeyValuePair,
|
||||
KikV2SearchResponse, KikV2SearchResponseDk, KikV2SearchResponseMk,
|
||||
KikV2SearchResult, KikV2CompactDecision, KikV2DocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class KikV2ApiClient:
|
||||
"""
|
||||
New KIK v2 API Client for https://ekapv2.kik.gov.tr
|
||||
|
||||
This client uses the modern JSON-based API endpoint that provides
|
||||
better structured data compared to the legacy form-based API.
|
||||
"""
|
||||
|
||||
BASE_URL = "https://ekapv2.kik.gov.tr"
|
||||
|
||||
# Endpoint mappings for different decision types
|
||||
ENDPOINTS = {
|
||||
KikV2DecisionType.UYUSMAZLIK: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlari",
|
||||
KikV2DecisionType.DUZENLEYICI: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariDk",
|
||||
KikV2DecisionType.MAHKEME: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariMk"
|
||||
}
|
||||
|
||||
# AES-256-CBC encryption key for document ID encryption (reverse engineered from ekapv2.kik.gov.tr Angular app)
|
||||
# This key is used to encrypt numeric gundemMaddesiId values to 64-character hex hashes for document URLs
|
||||
DOCUMENT_ID_ENCRYPTION_KEY = bytes([
|
||||
236, 193, 164, 43, 12, 135, 121, 170, 4, 244, 123, 219, 82, 158, 124, 174,
|
||||
174, 228, 219, 174, 208, 104, 174, 120, 32, 76, 250, 4, 143, 159, 211, 176
|
||||
])
|
||||
|
||||
# AES-192-CBC key (environment.r8fact) used by the Angular HTTP interceptor to sign every
|
||||
# request. The server decrypts X-Custom-Request-Ts and rejects stale timestamps with
|
||||
# HTTP 401 "İstek zaman aşımına uğradı.", so these headers MUST be generated per-request
|
||||
# with the current timestamp (see _generate_security_headers).
|
||||
REQUEST_SIGNING_KEY = b"Qm2LtXR0aByP69vZNKef4wMJ" # UTF-8 bytes, 24 chars -> AES-192
|
||||
|
||||
@staticmethod
|
||||
def encrypt_document_id(numeric_id: str) -> str:
|
||||
"""
|
||||
Encrypt a numeric KİK gundemMaddesiId to the 64-character hex hash
|
||||
used in document URLs.
|
||||
|
||||
Algorithm: AES-256-CBC with PKCS7 padding
|
||||
Output format: IV (16 bytes hex) + Ciphertext (16 bytes hex) = 64 chars
|
||||
|
||||
Args:
|
||||
numeric_id: The numeric document ID from search results (e.g., "177280")
|
||||
|
||||
Returns:
|
||||
64-character hex string for use in document URL KararId parameter
|
||||
"""
|
||||
if not HAS_CRYPTOGRAPHY:
|
||||
raise ImportError("cryptography library required for document ID encryption")
|
||||
|
||||
# Generate random IV (16 bytes)
|
||||
iv = os.urandom(16)
|
||||
|
||||
# Create AES-CBC cipher with the encryption key
|
||||
cipher = Cipher(
|
||||
algorithms.AES(KikV2ApiClient.DOCUMENT_ID_ENCRYPTION_KEY),
|
||||
modes.CBC(iv),
|
||||
backend=default_backend()
|
||||
)
|
||||
encryptor = cipher.encryptor()
|
||||
|
||||
# Encode plaintext and apply PKCS7 padding
|
||||
plaintext = numeric_id.encode('utf-8')
|
||||
block_size = 16
|
||||
padding_len = block_size - (len(plaintext) % block_size)
|
||||
padded_plaintext = plaintext + bytes([padding_len] * padding_len)
|
||||
|
||||
# Encrypt
|
||||
ciphertext = encryptor.update(padded_plaintext) + encryptor.finalize()
|
||||
|
||||
# Return IV + ciphertext as lowercase hex (64 characters total)
|
||||
return iv.hex() + ciphertext.hex()
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
# Create SSL context with legacy server support
|
||||
ssl_context = ssl.create_default_context()
|
||||
ssl_context.check_hostname = False
|
||||
ssl_context.verify_mode = ssl.CERT_NONE
|
||||
|
||||
# Enable legacy server connect option for older SSL implementations (Python 3.12+)
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
|
||||
# Set broader cipher suite support including legacy ciphers
|
||||
ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
verify=ssl_context,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Language": "tr",
|
||||
"Content-Type": "application/json",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": f"{self.BASE_URL}/sorgulamalar/kurul-kararlari",
|
||||
"Sec-Fetch-Dest": "empty",
|
||||
"Sec-Fetch-Mode": "cors",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36",
|
||||
"api-version": "v1",
|
||||
"sec-ch-ua": '"Not;A=Brand";v="99", "Google Chrome";v="139", "Chromium";v="139"',
|
||||
"sec-ch-ua-mobile": "?0",
|
||||
"sec-ch-ua-platform": '"macOS"'
|
||||
},
|
||||
timeout=request_timeout
|
||||
)
|
||||
|
||||
# Generate security headers (these might need to be updated based on API requirements)
|
||||
self.security_headers = self._generate_security_headers()
|
||||
|
||||
def _sign_request_value(self, plaintext: str, iv: bytes) -> str:
|
||||
"""AES-192-CBC encrypt a value with the request signing key, return base64 ciphertext."""
|
||||
cipher = Cipher(
|
||||
algorithms.AES(self.REQUEST_SIGNING_KEY),
|
||||
modes.CBC(iv),
|
||||
backend=default_backend()
|
||||
)
|
||||
encryptor = cipher.encryptor()
|
||||
data = plaintext.encode("utf-8")
|
||||
block_size = 16
|
||||
padding_len = block_size - (len(data) % block_size)
|
||||
padded = data + bytes([padding_len] * padding_len)
|
||||
ciphertext = encryptor.update(padded) + encryptor.finalize()
|
||||
return base64.b64encode(ciphertext).decode("ascii")
|
||||
|
||||
def _generate_security_headers(self) -> dict:
|
||||
"""
|
||||
Generate the custom security headers required by the KIK v2 API.
|
||||
|
||||
Mirrors the Angular HTTP interceptor on ekapv2.kik.gov.tr: a random GUID and a
|
||||
current-timestamp (epoch milliseconds) are AES-192-CBC encrypted with environment.r8fact
|
||||
using a fresh random IV. The IV is sent as -Siv, the encrypted timestamp as -Ts, and the
|
||||
encrypted GUID as -R8id. The server validates the decrypted timestamp's freshness, so these
|
||||
MUST be regenerated on every request; stale values yield HTTP 401 "İstek zaman aşımına uğradı.".
|
||||
"""
|
||||
if not HAS_CRYPTOGRAPHY:
|
||||
raise ImportError("cryptography library required for KIK v2 request signing")
|
||||
|
||||
request_guid = str(uuid.uuid4())
|
||||
iv = os.urandom(16)
|
||||
timestamp_ms = str(int(datetime.now().timestamp() * 1000))
|
||||
|
||||
return {
|
||||
"X-Custom-Request-Guid": request_guid,
|
||||
"X-Custom-Request-R8id": self._sign_request_value(request_guid, iv),
|
||||
"X-Custom-Request-Siv": base64.b64encode(iv).decode("ascii"),
|
||||
"X-Custom-Request-Ts": self._sign_request_value(timestamp_ms, iv),
|
||||
}
|
||||
|
||||
def _build_search_payload(self,
|
||||
decision_type: KikV2DecisionType,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = ""):
|
||||
"""Build the search payload for KIK v2 API."""
|
||||
|
||||
key_value_pairs = []
|
||||
|
||||
# Add non-empty search criteria
|
||||
if karar_metni:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=karar_metni))
|
||||
|
||||
if karar_no:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararNo", value=karar_no))
|
||||
|
||||
if basvuran:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BasvuranAdi", value=basvuran))
|
||||
|
||||
if idare_adi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="IdareAdi", value=idare_adi))
|
||||
|
||||
if baslangic_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BaslangicTarihi", value=baslangic_tarihi))
|
||||
|
||||
if bitis_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BitisTarihi", value=bitis_tarihi))
|
||||
|
||||
# If no search criteria provided, use a generic search
|
||||
if not key_value_pairs:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=""))
|
||||
|
||||
query_request = KikV2QueryRequest(keyValueOfstringanyType=key_value_pairs)
|
||||
request_data = KikV2RequestData(keyValuePairs=query_request)
|
||||
|
||||
# Return appropriate payload based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
return KikV2SearchPayload(sorgulaKurulKararlari=request_data)
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
return KikV2SearchPayloadDk(sorgulaKurulKararlariDk=request_data)
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
return KikV2SearchPayloadMk(sorgulaKurulKararlariMk=request_data)
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
async def search_decisions(self,
|
||||
decision_type: KikV2DecisionType = KikV2DecisionType.UYUSMAZLIK,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = "") -> KikV2SearchResult:
|
||||
"""
|
||||
Search KIK decisions using the v2 API.
|
||||
|
||||
Args:
|
||||
decision_type: Type of decision to search (uyusmazlik/duzenleyici/mahkeme)
|
||||
karar_metni: Decision text search
|
||||
karar_no: Decision number (e.g., "2025/UH.II-1801")
|
||||
basvuran: Applicant name
|
||||
idare_adi: Administration name
|
||||
baslangic_tarihi: Start date (YYYY-MM-DD format)
|
||||
bitis_tarihi: End date (YYYY-MM-DD format)
|
||||
|
||||
Returns:
|
||||
KikV2SearchResult with compact decision list
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Searching {decision_type.value} decisions with criteria - karar_metni: '{karar_metni}', karar_no: '{karar_no}', basvuran: '{basvuran}'")
|
||||
|
||||
try:
|
||||
# Build request payload
|
||||
payload = self._build_search_payload(
|
||||
decision_type=decision_type,
|
||||
karar_metni=karar_metni,
|
||||
karar_no=karar_no,
|
||||
basvuran=basvuran,
|
||||
idare_adi=idare_adi,
|
||||
baslangic_tarihi=baslangic_tarihi,
|
||||
bitis_tarihi=bitis_tarihi
|
||||
)
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Get the appropriate endpoint for this decision type
|
||||
endpoint = self.ENDPOINTS[decision_type]
|
||||
|
||||
# Make API request
|
||||
response = await self.http_client.post(
|
||||
endpoint,
|
||||
json=payload.model_dump(),
|
||||
headers=headers
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
response_data = response.json()
|
||||
|
||||
logger.debug(f"KikV2ApiClient: Raw API response structure: {type(response_data)}")
|
||||
|
||||
# Parse the API response based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
api_response = KikV2SearchResponse(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariResponse.SorgulaKurulKararlariResult
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
api_response = KikV2SearchResponseDk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariDkResponse.SorgulaKurulKararlariDkResult
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
api_response = KikV2SearchResponseMk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariMkResponse.SorgulaKurulKararlariMkResult
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
# Check for API errors
|
||||
if result_data.hataKodu and result_data.hataKodu != "0":
|
||||
logger.warning(f"KikV2ApiClient: API returned error - Code: {result_data.hataKodu}, Message: {result_data.hataMesaji}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code=result_data.hataKodu,
|
||||
error_message=result_data.hataMesaji
|
||||
)
|
||||
|
||||
# Convert to compact format
|
||||
compact_decisions = []
|
||||
total_count = 0
|
||||
|
||||
for decision_group in result_data.KurulKararTutanakDetayListesi:
|
||||
for decision_detail in decision_group.KurulKararTutanakDetayi:
|
||||
compact_decision = KikV2CompactDecision(
|
||||
kararNo=decision_detail.kararNo,
|
||||
kararTarihi=decision_detail.kararTarihi,
|
||||
basvuran=decision_detail.basvuran,
|
||||
idareAdi=decision_detail.idareAdi,
|
||||
basvuruKonusu=decision_detail.basvuruKonusu,
|
||||
gundemMaddesiId=decision_detail.gundemMaddesiId,
|
||||
decision_type=decision_type.value
|
||||
)
|
||||
compact_decisions.append(compact_decision)
|
||||
total_count += 1
|
||||
|
||||
logger.info(f"KikV2ApiClient: Found {total_count} decisions")
|
||||
|
||||
return KikV2SearchResult(
|
||||
decisions=compact_decisions,
|
||||
total_records=total_count,
|
||||
page=1,
|
||||
error_code="0",
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"KikV2ApiClient: HTTP error during search: {e.response.status_code} - {e.response.text}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="HTTP_ERROR",
|
||||
error_message=f"HTTP {e.response.status_code}: {e.response.text}"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Unexpected error during search: {str(e)}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="UNEXPECTED_ERROR",
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def get_document_markdown(self, document_id: str) -> KikV2DocumentMarkdown:
|
||||
"""
|
||||
Get KİK decision document content in Markdown format.
|
||||
|
||||
This method uses a two-step process:
|
||||
1. Call GetSorgulamaUrl endpoint to get the actual document URL
|
||||
2. Use httpx to fetch the document content
|
||||
|
||||
Args:
|
||||
document_id: The gundemMaddesiId from search results
|
||||
|
||||
Returns:
|
||||
KikV2DocumentMarkdown with document content converted to Markdown
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Getting document for ID: {document_id}")
|
||||
|
||||
if not document_id or not document_id.strip():
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Document ID is required"
|
||||
)
|
||||
|
||||
try:
|
||||
# Step 1: Get the actual document URL using GetSorgulamaUrl endpoint
|
||||
logger.info(f"KikV2ApiClient: Step 1 - Getting document URL for ID: {document_id}")
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Call GetSorgulamaUrl to get the real document URL
|
||||
url_payload = {"sorguSayfaTipi": 2} # As shown in curl example
|
||||
|
||||
url_response = await self.http_client.post(
|
||||
"/b_ihalearaclari/api/KurulKararlari/GetSorgulamaUrl",
|
||||
json=url_payload,
|
||||
headers=headers
|
||||
)
|
||||
|
||||
url_response.raise_for_status()
|
||||
url_data = url_response.json()
|
||||
|
||||
# Get the base document URL from API response
|
||||
base_document_url = url_data.get("sorgulamaUrl", "")
|
||||
if not base_document_url:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Could not get document URL from GetSorgulamaUrl API"
|
||||
)
|
||||
|
||||
# If document_id is numeric, encrypt it to get the KararId hash
|
||||
# The web interface uses AES-256-CBC encrypted hashes for document URLs
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID {document_id} to hash: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt document ID, using as-is: {enc_error}")
|
||||
|
||||
# Construct full document URL with the encrypted KararId
|
||||
document_url = f"{base_document_url}?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Retrieved document URL: {document_url}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error getting document URL for ID {document_id}: {str(e)}")
|
||||
# Fallback to old method if GetSorgulamaUrl fails
|
||||
# Also encrypt numeric IDs in fallback path
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID in fallback: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt in fallback: {enc_error}")
|
||||
document_url = f"https://ekap.kik.gov.tr/EKAP/Vatandas/KurulKararGoster.aspx?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Falling back to direct URL: {document_url}")
|
||||
|
||||
try:
|
||||
# Step 2: Use httpx to get the document content
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Using httpx to retrieve document from: {document_url}")
|
||||
|
||||
# Create a separate httpx client for document retrieval with HTML headers
|
||||
doc_ssl_context = ssl.create_default_context()
|
||||
doc_ssl_context.check_hostname = False
|
||||
doc_ssl_context.verify_mode = ssl.CERT_NONE
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
doc_ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
doc_ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
verify=doc_ssl_context,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "tr,en-US;q=0.5",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=60.0,
|
||||
follow_redirects=True
|
||||
) as doc_client:
|
||||
response = await doc_client.get(document_url)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
logger.info(f"KikV2ApiClient: Retrieved content via httpx, length: {len(html_content)}")
|
||||
|
||||
# Convert HTML to Markdown using MarkItDown with BytesIO
|
||||
try:
|
||||
from markitdown import MarkItDown
|
||||
from io import BytesIO
|
||||
|
||||
md = MarkItDown()
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = BytesIO(html_bytes)
|
||||
|
||||
# markitdown is sync; offload to thread so HTML parsing doesn't
|
||||
# block the event-loop / other in-flight MCP requests.
|
||||
result = await asyncio.to_thread(md.convert_stream, html_stream, file_extension=".html")
|
||||
markdown_content = result.text_content
|
||||
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content=markdown_content,
|
||||
source_url=document_url,
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except ImportError:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="MarkItDown library not available",
|
||||
source_url=document_url,
|
||||
error_message="MarkItDown library not installed"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error retrieving document {document_id}: {str(e)}")
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url=document_url,
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("KikV2ApiClient: HTTP client session closed.")
|
||||
@@ -1,75 +0,0 @@
|
||||
# kik_mcp_module/models.py
|
||||
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
import base64 # Base64 encoding/decoding için
|
||||
|
||||
class KikKararTipi(str, Enum):
|
||||
"""Enum for KIK (Public Procurement Authority) Decision Types."""
|
||||
UYUSMAZLIK = "rbUyusmazlik"
|
||||
DUZENLEYICI = "rbDuzenleyici"
|
||||
MAHKEME = "rbMahkeme"
|
||||
|
||||
class KikSearchRequest(BaseModel):
|
||||
"""Model for KIK Decision search criteria."""
|
||||
karar_tipi: KikKararTipi = Field(KikKararTipi.UYUSMAZLIK, description="Type of KIK Decision.")
|
||||
karar_no: Optional[str] = Field(None, description="Decision Number (e.g., '2024/UH.II-1766').")
|
||||
karar_tarihi_baslangic: Optional[str] = Field(None, description="Decision Date Start (DD.MM.YYYY).", pattern=r"^\d{2}\.\d{2}\.\d{4}$")
|
||||
karar_tarihi_bitis: Optional[str] = Field(None, description="Decision Date End (DD.MM.YYYY).", pattern=r"^\d{2}\.\d{2}\.\d{4}$")
|
||||
resmi_gazete_sayisi: Optional[str] = Field(None, description="Official Gazette Number.")
|
||||
resmi_gazete_tarihi: Optional[str] = Field(None, description="Official Gazette Date (DD.MM.YYYY).", pattern=r"^\d{2}\.\d{2}\.\d{4}$")
|
||||
basvuru_konusu_ihale: Optional[str] = Field(None, description="Tender subject of the application.")
|
||||
basvuru_sahibi: Optional[str] = Field(None, description="Applicant.")
|
||||
ihaleyi_yapan_idare: Optional[str] = Field(None, description="Procuring Entity.")
|
||||
yil: Optional[str] = Field(None, description="Year of the decision.")
|
||||
karar_metni: Optional[str] = Field(None, description="Keyword/phrase in decision text.")
|
||||
page: int = Field(1, ge=1, description="Results page number.")
|
||||
|
||||
class KikDecisionEntry(BaseModel):
|
||||
"""Represents a single decision entry from KIK search results."""
|
||||
preview_event_target: str = Field(..., description="Internal event target for fetching details.")
|
||||
karar_no_str: str = Field(..., alias="kararNo", description="Raw decision number as extracted from KIK (e.g., '2024/UH.II-1766').")
|
||||
karar_tipi: KikKararTipi = Field(..., description="The type of decision this entry belongs to.")
|
||||
|
||||
karar_tarihi_str: str = Field(..., alias="kararTarihi", description="Decision date.")
|
||||
idare_str: Optional[str] = Field(None, alias="idare", description="Procuring entity.")
|
||||
basvuru_sahibi_str: Optional[str] = Field(None, alias="basvuruSahibi", description="Applicant.")
|
||||
ihale_konusu_str: Optional[str] = Field(None, alias="ihaleKonusu", description="Tender subject.")
|
||||
|
||||
@computed_field
|
||||
@property
|
||||
def karar_id(self) -> str:
|
||||
"""
|
||||
A Base64 encoded unique ID for the decision, combining decision type and number.
|
||||
Format before encoding: "{karar_tipi.value}|{karar_no_str}"
|
||||
"""
|
||||
combined_key = f"{self.karar_tipi.value}|{self.karar_no_str}"
|
||||
return base64.b64encode(combined_key.encode('utf-8')).decode('utf-8')
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikSearchResult(BaseModel):
|
||||
"""Model for KIK search results."""
|
||||
decisions: List[KikDecisionEntry]
|
||||
total_records: int = 0
|
||||
current_page: int = 1
|
||||
|
||||
class KikDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
KIK decision document, with Markdown content potentially paginated.
|
||||
"""
|
||||
retrieved_with_karar_id: Optional[str] = Field(None, description="The Base64 encoded karar_id that was used to request this document.")
|
||||
# Decode edilmiş karar no ve tipini de yanıt olarak ekleyelim, Claude için faydalı olabilir.
|
||||
retrieved_karar_no: Optional[str] = Field(None, description="The raw KIK Decision Number (e.g., '2024/UH.II-1766') this document pertains to.")
|
||||
retrieved_karar_tipi: Optional[KikKararTipi] = Field(None, description="The KIK Decision Type this document pertains to.")
|
||||
|
||||
karar_id_param_from_url: Optional[str] = Field(None, alias="kararIdParam", description="The KIK system's internal KararId parameter from the document's display URL (KurulKararGoster.aspx).")
|
||||
markdown_chunk: Optional[str] = Field(None, description="The requested chunk of the decision content converted to Markdown.")
|
||||
source_url: Optional[str] = Field(None, description="The source URL of the original document (KurulKararGoster.aspx).")
|
||||
error_message: Optional[str] = Field(None, description="Error message if document retrieval or processing failed.")
|
||||
current_page: int = Field(1, description="The current page number of the markdown chunk being returned.")
|
||||
total_pages: int = Field(1, description="The total number of pages the full markdown content is divided into.")
|
||||
is_paginated: bool = Field(False, description="True if the full markdown content is split into multiple pages.")
|
||||
full_content_char_count: Optional[int] = Field(None, description="Total character count of the full markdown content before chunking.")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
@@ -0,0 +1,147 @@
|
||||
# kik_mcp_module/models_v2.py
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
|
||||
# New KIK v2 API Models
|
||||
|
||||
class KikV2DecisionType(str, Enum):
|
||||
"""KIK v2 Decision Types with corresponding endpoints."""
|
||||
UYUSMAZLIK = "uyusmazlik" # Disputes - GetKurulKararlari
|
||||
DUZENLEYICI = "duzenleyici" # Regulatory - GetKurulKararlariDk
|
||||
MAHKEME = "mahkeme" # Court - GetKurulKararlariMk
|
||||
|
||||
class KikV2SearchRequest(BaseModel):
|
||||
"""Model for KIK v2 API search request."""
|
||||
KararMetni: str = Field("", description="Decision text search query")
|
||||
KararNo: str = Field("", description="Decision number (e.g., '2025/UH.II-1801')")
|
||||
BasvuranAdi: str = Field("", description="Applicant name")
|
||||
IdareAdi: str = Field("", description="Administration name")
|
||||
BaslangicTarihi: str = Field("", description="Start date (YYYY-MM-DD)")
|
||||
BitisTarihi: str = Field("", description="End date (YYYY-MM-DD)")
|
||||
|
||||
class KikV2KeyValuePair(BaseModel):
|
||||
"""Key-value pair for KIK v2 API request."""
|
||||
key: str
|
||||
value: str
|
||||
|
||||
class KikV2QueryRequest(BaseModel):
|
||||
"""Nested query structure for KIK v2 API."""
|
||||
keyValueOfstringanyType: List[KikV2KeyValuePair]
|
||||
|
||||
class KikV2RequestData(BaseModel):
|
||||
"""Main request data structure for KIK v2 API."""
|
||||
keyValuePairs: KikV2QueryRequest
|
||||
|
||||
# Request Payloads for different decision types
|
||||
class KikV2SearchPayload(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Uyuşmazlık (Disputes)."""
|
||||
sorgulaKurulKararlari: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadDk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Düzenleyici (Regulatory)."""
|
||||
sorgulaKurulKararlariDk: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadMk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Mahkeme (Court)."""
|
||||
sorgulaKurulKararlariMk: KikV2RequestData
|
||||
|
||||
# Response Models
|
||||
|
||||
class KikV2DecisionDetail(BaseModel):
|
||||
"""Individual decision detail from KIK v2 API response."""
|
||||
resmiGazeteMukerrerSayi: str = Field("", description="Official Gazette duplicate number")
|
||||
itiraz: str = Field("", description="Objection")
|
||||
yayinlanmaTarihi: str = Field("", description="Publication date")
|
||||
idareAdi: str = Field("", description="Administration name")
|
||||
uzmanTCKN: str = Field("", description="Expert TCKN")
|
||||
resmiGazeteTarihi: str = Field("", description="Official Gazette date")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
kararTurKod: str = Field("", description="Decision type code")
|
||||
kararTurAciklama: str = Field("", description="Decision type description")
|
||||
karar: str = Field("", description="Decision text")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
resmiGazeteSayisi: str = Field("", description="Official Gazette number")
|
||||
inceleme: str = Field("", description="Review")
|
||||
basvuruTarihi: str = Field("", description="Application date")
|
||||
kararNitelikKod: str = Field("", description="Decision nature code")
|
||||
resmiGazeteMukerrer: str = Field("", description="Official Gazette duplicate")
|
||||
basvuruSayisi: str = Field("", description="Application number")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
kararNitelik: str = Field("", description="Decision nature")
|
||||
uyusmazlikKararNo: str = Field("", description="Dispute decision number")
|
||||
kurulNo: str = Field("", description="Board number")
|
||||
gundemMaddesiSiraNo: str = Field("", description="Agenda item sequence")
|
||||
kararTarihi: str = Field("", description="Decision date (ISO format)")
|
||||
dosyaBirimKodu: str = Field("", description="File unit code")
|
||||
gundemMaddesiId: str = Field("", description="Agenda item ID")
|
||||
|
||||
class KikV2DecisionGroup(BaseModel):
|
||||
"""Group of decision details."""
|
||||
KurulKararTutanakDetayi: List[KikV2DecisionDetail] = Field(alias="kurulKararTutanakDetayi")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultData(BaseModel):
|
||||
"""Search result data structure."""
|
||||
hataKodu: str = Field("", description="Error code")
|
||||
hataMesaji: str = Field("", description="Error message")
|
||||
KurulKararTutanakDetayListesi: List[KikV2DecisionGroup]
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultWrapper(BaseModel):
|
||||
"""Wrapper for search result."""
|
||||
SorgulaKurulKararlariResult: KikV2SearchResultData
|
||||
|
||||
# Base Response Models
|
||||
class KikV2SearchResponse(BaseModel):
|
||||
"""Complete KIK v2 API search response for Uyuşmazlık (Disputes)."""
|
||||
SorgulaKurulKararlariResponse: KikV2SearchResultWrapper
|
||||
|
||||
# Düzenleyici Kararlar (Regulatory Decisions) Response Models
|
||||
class KikV2SearchResultWrapperDk(BaseModel):
|
||||
"""Wrapper for regulatory decisions search result."""
|
||||
SorgulaKurulKararlariDkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseDk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Düzenleyici (Regulatory) decisions."""
|
||||
SorgulaKurulKararlariDkResponse: KikV2SearchResultWrapperDk
|
||||
|
||||
# Mahkeme Kararlar (Court Decisions) Response Models
|
||||
class KikV2SearchResultWrapperMk(BaseModel):
|
||||
"""Wrapper for court decisions search result."""
|
||||
SorgulaKurulKararlariMkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseMk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Mahkeme (Court) decisions."""
|
||||
SorgulaKurulKararlariMkResponse: KikV2SearchResultWrapperMk
|
||||
|
||||
# Simplified Models for MCP Tools
|
||||
|
||||
class KikV2CompactDecision(BaseModel):
|
||||
"""Compact decision format for MCP tool responses."""
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
kararTarihi: str = Field("", description="Decision date")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
idareAdi: str = Field("", description="Administration")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
gundemMaddesiId: str = Field("", description="Document ID for retrieval")
|
||||
decision_type: str = Field("", description="Decision type (uyusmazlik/duzenleyici/mahkeme)")
|
||||
|
||||
class KikV2SearchResult(BaseModel):
|
||||
"""Compact search results for MCP tools."""
|
||||
decisions: List[KikV2CompactDecision]
|
||||
total_records: int = Field(0, description="Total number of decisions found")
|
||||
page: int = Field(1, description="Current page number")
|
||||
error_code: str = Field("", description="API error code")
|
||||
error_message: str = Field("", description="API error message")
|
||||
|
||||
class KikV2DocumentMarkdown(BaseModel):
|
||||
"""Document content in Markdown format."""
|
||||
document_id: str = Field("", description="Document ID")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
markdown_content: str = Field("", description="Decision content in Markdown")
|
||||
source_url: str = Field("", description="Source URL")
|
||||
error_message: str = Field("", description="Error message if retrieval failed")
|
||||
@@ -0,0 +1 @@
|
||||
# kvkk_mcp_module/__init__.py
|
||||
@@ -0,0 +1,373 @@
|
||||
# kvkk_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Dict, Any
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from urllib.parse import urljoin, urlparse, parse_qs
|
||||
from markitdown import MarkItDown
|
||||
from pydantic import HttpUrl
|
||||
|
||||
from .models import (
|
||||
KvkkSearchRequest,
|
||||
KvkkDecisionSummary,
|
||||
KvkkSearchResult,
|
||||
KvkkDocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
class KvkkApiClient:
|
||||
"""
|
||||
API client for searching and retrieving KVKK (Personal Data Protection Authority) decisions
|
||||
using Brave Search API for discovery and direct HTTP requests for content retrieval.
|
||||
"""
|
||||
|
||||
BRAVE_API_URL = "https://api.search.brave.com/res/v1/web/search"
|
||||
KVKK_BASE_URL = "https://www.kvkk.gov.tr"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000 # Character limit per page
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
"""Initialize the KVKK API client."""
|
||||
self.brave_api_token = os.getenv("BRAVE_API_TOKEN")
|
||||
if not self.brave_api_token:
|
||||
# Fallback to provided free token
|
||||
self.brave_api_token = "BSAuaRKB-dvSDSQxIN0ft1p2k6N82Kq"
|
||||
logger.info("Using fallback Brave API token (limited free token)")
|
||||
else:
|
||||
logger.info("Using Brave API token from environment variable")
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True
|
||||
)
|
||||
|
||||
def _construct_search_query(self, keywords: str) -> str:
|
||||
"""Construct the search query for Brave API."""
|
||||
base_query = 'site:kvkk.gov.tr "karar özeti"'
|
||||
if keywords.strip():
|
||||
return f"{base_query} {keywords.strip()}"
|
||||
return base_query
|
||||
|
||||
def _extract_decision_id_from_url(self, url: str) -> Optional[str]:
|
||||
"""Extract decision ID from KVKK decision URL."""
|
||||
try:
|
||||
# Example URL: https://www.kvkk.gov.tr/Icerik/7288/2021-1303
|
||||
parsed_url = urlparse(url)
|
||||
path_parts = parsed_url.path.strip('/').split('/')
|
||||
|
||||
if len(path_parts) >= 3 and path_parts[0] == 'Icerik':
|
||||
# Extract the decision ID from the path
|
||||
decision_id = '/'.join(path_parts[1:]) # e.g., "7288/2021-1303"
|
||||
return decision_id
|
||||
|
||||
except Exception as e:
|
||||
logger.debug(f"Could not extract decision ID from URL {url}: {e}")
|
||||
|
||||
return None
|
||||
|
||||
def _extract_decision_metadata_from_title(self, title: str) -> Dict[str, Optional[str]]:
|
||||
"""Extract decision metadata from title string."""
|
||||
metadata = {
|
||||
"decision_date": None,
|
||||
"decision_number": None
|
||||
}
|
||||
|
||||
if not title:
|
||||
return metadata
|
||||
|
||||
# Extract decision date (DD/MM/YYYY format)
|
||||
date_match = re.search(r'(\d{1,2}/\d{1,2}/\d{4})', title)
|
||||
if date_match:
|
||||
metadata["decision_date"] = date_match.group(1)
|
||||
|
||||
# Extract decision number (YYYY/XXXX format)
|
||||
number_match = re.search(r'(\d{4}/\d+)', title)
|
||||
if number_match:
|
||||
metadata["decision_number"] = number_match.group(1)
|
||||
|
||||
return metadata
|
||||
|
||||
async def search_decisions(self, params: KvkkSearchRequest) -> KvkkSearchResult:
|
||||
"""Search for KVKK decisions using Brave API."""
|
||||
|
||||
search_query = self._construct_search_query(params.keywords)
|
||||
logger.info(f"KvkkApiClient: Searching with query: {search_query}")
|
||||
|
||||
try:
|
||||
# Calculate offset for pagination
|
||||
offset = (params.page - 1) * params.pageSize
|
||||
|
||||
response = await self.http_client.get(
|
||||
self.BRAVE_API_URL,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Encoding": "gzip",
|
||||
"x-subscription-token": self.brave_api_token
|
||||
},
|
||||
params={
|
||||
"q": search_query,
|
||||
"country": "TR",
|
||||
"search_lang": "tr",
|
||||
"ui_lang": "tr-TR",
|
||||
"offset": offset,
|
||||
"count": params.pageSize
|
||||
}
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
# Extract search results
|
||||
decisions = []
|
||||
web_results = data.get("web", {}).get("results", [])
|
||||
|
||||
for result in web_results:
|
||||
title = result.get("title", "")
|
||||
url = result.get("url", "")
|
||||
description = result.get("description", "")
|
||||
|
||||
# Extract metadata from title
|
||||
metadata = self._extract_decision_metadata_from_title(title)
|
||||
|
||||
# Extract decision ID from URL
|
||||
decision_id = self._extract_decision_id_from_url(url)
|
||||
|
||||
decision = KvkkDecisionSummary(
|
||||
title=title,
|
||||
url=HttpUrl(url) if url else None,
|
||||
description=description,
|
||||
decision_id=decision_id,
|
||||
publication_date=metadata.get("decision_date"),
|
||||
decision_number=metadata.get("decision_number")
|
||||
)
|
||||
decisions.append(decision)
|
||||
|
||||
# Get total results if available
|
||||
total_results = None
|
||||
query_info = data.get("query", {})
|
||||
if "total_results" in query_info:
|
||||
total_results = query_info["total_results"]
|
||||
|
||||
return KvkkSearchResult(
|
||||
decisions=decisions,
|
||||
total_results=total_results,
|
||||
page=params.page,
|
||||
pageSize=params.pageSize,
|
||||
query=search_query
|
||||
)
|
||||
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"KvkkApiClient: HTTP request error during search: {e}")
|
||||
return KvkkSearchResult(
|
||||
decisions=[],
|
||||
total_results=0,
|
||||
page=params.page,
|
||||
pageSize=params.pageSize,
|
||||
query=search_query
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"KvkkApiClient: Unexpected error during search: {e}")
|
||||
return KvkkSearchResult(
|
||||
decisions=[],
|
||||
total_results=0,
|
||||
page=params.page,
|
||||
pageSize=params.pageSize,
|
||||
query=search_query
|
||||
)
|
||||
|
||||
def _extract_decision_content_from_html(self, html: str, url: str) -> Dict[str, Any]:
|
||||
"""Extract decision content from KVKK decision page HTML."""
|
||||
try:
|
||||
soup = BeautifulSoup(html, 'html.parser')
|
||||
|
||||
# Extract title
|
||||
title = None
|
||||
title_element = soup.find('h3', class_='blog-post-title')
|
||||
if title_element:
|
||||
title = title_element.get_text(strip=True)
|
||||
elif soup.title:
|
||||
title = soup.title.get_text(strip=True)
|
||||
|
||||
# Extract decision content from the main content div
|
||||
content_div = soup.find('div', class_='blog-post-inner')
|
||||
if not content_div:
|
||||
# Fallback to other possible content containers
|
||||
content_div = soup.find('div', style='text-align:justify;')
|
||||
if not content_div:
|
||||
logger.warning(f"Could not find decision content div in {url}")
|
||||
return {
|
||||
"title": title,
|
||||
"decision_date": None,
|
||||
"decision_number": None,
|
||||
"subject_summary": None,
|
||||
"html_content": None
|
||||
}
|
||||
|
||||
# Extract decision metadata from table
|
||||
decision_date = None
|
||||
decision_number = None
|
||||
subject_summary = None
|
||||
|
||||
table = content_div.find('table')
|
||||
if table:
|
||||
rows = table.find_all('tr')
|
||||
for row in rows:
|
||||
cells = row.find_all('td')
|
||||
if len(cells) >= 3:
|
||||
field_name = cells[0].get_text(strip=True)
|
||||
field_value = cells[2].get_text(strip=True)
|
||||
|
||||
if 'Karar Tarihi' in field_name:
|
||||
decision_date = field_value
|
||||
elif 'Karar No' in field_name:
|
||||
decision_number = field_value
|
||||
elif 'Konu Özeti' in field_name:
|
||||
subject_summary = field_value
|
||||
|
||||
return {
|
||||
"title": title,
|
||||
"decision_date": decision_date,
|
||||
"decision_number": decision_number,
|
||||
"subject_summary": subject_summary,
|
||||
"html_content": str(content_div)
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error extracting content from HTML for {url}: {e}")
|
||||
return {
|
||||
"title": None,
|
||||
"decision_date": None,
|
||||
"decision_number": None,
|
||||
"subject_summary": None,
|
||||
"html_content": None
|
||||
}
|
||||
|
||||
def _convert_html_to_markdown(self, html_content: str) -> Optional[str]:
|
||||
"""Convert HTML content to Markdown using MarkItDown with BytesIO to avoid filename length issues."""
|
||||
if not html_content:
|
||||
return None
|
||||
|
||||
try:
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown(enable_plugins=False)
|
||||
result = md_converter.convert(html_stream)
|
||||
return result.text_content
|
||||
except Exception as e:
|
||||
logger.error(f"Error converting HTML to Markdown: {e}")
|
||||
return None
|
||||
|
||||
async def get_decision_document(self, decision_url: str, page_number: int = 1) -> KvkkDocumentMarkdown:
|
||||
"""Retrieve and convert a KVKK decision document to paginated Markdown."""
|
||||
logger.info(f"KvkkApiClient: Getting decision document from: {decision_url}, page: {page_number}")
|
||||
|
||||
try:
|
||||
# Fetch the decision page
|
||||
response = await self.http_client.get(decision_url)
|
||||
response.raise_for_status()
|
||||
|
||||
# Extract content from HTML
|
||||
extracted_data = self._extract_decision_content_from_html(response.text, decision_url)
|
||||
|
||||
# Convert HTML content to Markdown
|
||||
full_markdown_content = None
|
||||
if extracted_data["html_content"]:
|
||||
full_markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, extracted_data["html_content"])
|
||||
|
||||
if not full_markdown_content:
|
||||
return KvkkDocumentMarkdown(
|
||||
source_url=HttpUrl(decision_url),
|
||||
title=extracted_data["title"],
|
||||
decision_date=extracted_data["decision_date"],
|
||||
decision_number=extracted_data["decision_number"],
|
||||
subject_summary=extracted_data["subject_summary"],
|
||||
markdown_chunk=None,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="Could not convert document content to Markdown"
|
||||
)
|
||||
|
||||
# Calculate pagination
|
||||
content_length = len(full_markdown_content)
|
||||
total_pages = math.ceil(content_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
if total_pages == 0:
|
||||
total_pages = 1
|
||||
|
||||
# Clamp page number to valid range
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
|
||||
# Extract the requested chunk
|
||||
start_index = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_index = start_index + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
markdown_chunk = full_markdown_content[start_index:end_index]
|
||||
|
||||
return KvkkDocumentMarkdown(
|
||||
source_url=HttpUrl(decision_url),
|
||||
title=extracted_data["title"],
|
||||
decision_date=extracted_data["decision_date"],
|
||||
decision_number=extracted_data["decision_number"],
|
||||
subject_summary=extracted_data["subject_summary"],
|
||||
markdown_chunk=markdown_chunk,
|
||||
current_page=current_page_clamped,
|
||||
total_pages=total_pages,
|
||||
is_paginated=(total_pages > 1),
|
||||
error_message=None
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
error_msg = f"HTTP error {e.response.status_code} when fetching decision document"
|
||||
logger.error(f"KvkkApiClient: {error_msg}")
|
||||
return KvkkDocumentMarkdown(
|
||||
source_url=HttpUrl(decision_url),
|
||||
title=None,
|
||||
decision_date=None,
|
||||
decision_number=None,
|
||||
subject_summary=None,
|
||||
markdown_chunk=None,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=error_msg
|
||||
)
|
||||
except Exception as e:
|
||||
error_msg = f"Unexpected error when fetching decision document: {str(e)}"
|
||||
logger.error(f"KvkkApiClient: {error_msg}")
|
||||
return KvkkDocumentMarkdown(
|
||||
source_url=HttpUrl(decision_url),
|
||||
title=None,
|
||||
decision_date=None,
|
||||
decision_number=None,
|
||||
subject_summary=None,
|
||||
markdown_chunk=None,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=error_msg
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close the HTTP client session."""
|
||||
if hasattr(self, 'http_client') and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("KvkkApiClient: HTTP client session closed.")
|
||||
@@ -0,0 +1,49 @@
|
||||
# kvkk_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional, Any
|
||||
|
||||
class KvkkSearchRequest(BaseModel):
|
||||
"""Model for KVKK (Personal Data Protection Authority) search request via Brave API."""
|
||||
keywords: str = Field(..., description="""
|
||||
Keywords to search for in KVKK decisions.
|
||||
The search will automatically include 'site:kvkk.gov.tr "karar özeti"' to target KVKK decision summaries.
|
||||
Examples: "açık rıza", "veri güvenliği", "kişisel veri işleme"
|
||||
""")
|
||||
page: int = Field(1, ge=1, le=50, description="Page number for search results (1-50).")
|
||||
pageSize: int = Field(10, ge=1, le=10, description="Number of results per page (1-10).")
|
||||
|
||||
class KvkkDecisionSummary(BaseModel):
|
||||
"""Model for a single KVKK decision summary from Brave search results."""
|
||||
title: Optional[str] = Field(None, description="Decision title from search results.")
|
||||
url: Optional[HttpUrl] = Field(None, description="URL to the KVKK decision page.")
|
||||
description: Optional[str] = Field(None, description="Brief description or snippet from search results.")
|
||||
decision_id: Optional[str] = Field(None, description="Value")
|
||||
publication_date: Optional[str] = Field(None, description="Value")
|
||||
decision_number: Optional[str] = Field(None, description="Value")
|
||||
|
||||
class KvkkSearchResult(BaseModel):
|
||||
"""Model for the overall search result for KVKK decisions."""
|
||||
decisions: List[KvkkDecisionSummary] = Field(default_factory=list, description="List of KVKK decisions found.")
|
||||
total_results: Optional[int] = Field(None, description="Value")
|
||||
page: int = Field(1, description="Current page number of results.")
|
||||
pageSize: int = Field(10, description="Number of results per page.")
|
||||
query: Optional[str] = Field(None, description="The actual search query sent to Brave API.")
|
||||
|
||||
class KvkkDocumentMarkdown(BaseModel):
|
||||
"""Model for KVKK decision document content converted to paginated Markdown."""
|
||||
source_url: HttpUrl = Field(description="URL of the original KVKK decision page.")
|
||||
title: Optional[str] = Field(None, description="Title of the KVKK decision.")
|
||||
decision_date: Optional[str] = Field(None, description="Decision date (Karar Tarihi).")
|
||||
decision_number: Optional[str] = Field(None, description="Decision number (Karar No).")
|
||||
subject_summary: Optional[str] = Field(None, description="Subject summary (Konu Özeti).")
|
||||
markdown_chunk: Optional[str] = Field(None, description="A 5,000 character chunk of the Markdown content.")
|
||||
current_page: int = Field(description="The current page number of the markdown chunk (1-indexed).")
|
||||
total_pages: int = Field(description="Total number of pages for the full markdown content.")
|
||||
is_paginated: bool = Field(description="True if the full markdown content is split into multiple pages.")
|
||||
error_message: Optional[str] = Field(None, description="Value")
|
||||
|
||||
class Config:
|
||||
json_encoders = {
|
||||
HttpUrl: str
|
||||
}
|
||||
@@ -1,28 +0,0 @@
|
||||
"""
|
||||
MCP Auth Toolkit - OAuth 2.1 + Authorization for Model Context Protocol Servers
|
||||
Integrated with Clerk Authentication
|
||||
"""
|
||||
|
||||
from .middleware import (
|
||||
AuthContext,
|
||||
FastMCPAuthWrapper,
|
||||
MCPAuthMiddleware,
|
||||
auth_required,
|
||||
)
|
||||
from .oauth import OAuthConfig, OAuthProvider
|
||||
from .policy import PolicyEngine, ToolPolicy, create_default_policies
|
||||
from .storage import PersistentStorage
|
||||
|
||||
__version__ = "0.1.0"
|
||||
__all__ = [
|
||||
"OAuthProvider",
|
||||
"OAuthConfig",
|
||||
"AuthContext",
|
||||
"auth_required",
|
||||
"create_default_policies",
|
||||
"MCPAuthMiddleware",
|
||||
"FastMCPAuthWrapper",
|
||||
"PolicyEngine",
|
||||
"ToolPolicy",
|
||||
"PersistentStorage",
|
||||
]
|
||||
@@ -1,73 +0,0 @@
|
||||
"""
|
||||
Clerk OAuth configuration for MCP Auth Toolkit
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
from .oauth import OAuthConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def create_clerk_oauth_config() -> OAuthConfig:
|
||||
"""Create OAuth configuration for Clerk integration using SDK"""
|
||||
|
||||
# Get Clerk configuration from environment
|
||||
clerk_domain = os.getenv("CLERK_DOMAIN", "accounts.yargimcp.com")
|
||||
clerk_publishable_key = os.getenv("CLERK_PUBLISHABLE_KEY")
|
||||
clerk_secret_key = os.getenv("CLERK_SECRET_KEY")
|
||||
|
||||
if not clerk_publishable_key or not clerk_secret_key:
|
||||
raise ValueError("CLERK_PUBLISHABLE_KEY and CLERK_SECRET_KEY are required")
|
||||
|
||||
# For Clerk with custom domains, we use our adapter endpoints
|
||||
# This allows us to handle the custom domain flow properly
|
||||
base_url = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
config = OAuthConfig(
|
||||
client_id=clerk_publishable_key,
|
||||
client_secret=clerk_secret_key,
|
||||
# Use our adapter endpoints instead of Clerk's direct endpoints
|
||||
authorization_endpoint=f"{base_url}/authorize",
|
||||
token_endpoint=f"{base_url}/token",
|
||||
# Keep Clerk's JWKS for token validation
|
||||
jwks_uri=f"https://{clerk_domain}/.well-known/jwks.json",
|
||||
issuer=base_url, # We're the issuer for MCP tokens
|
||||
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
|
||||
)
|
||||
|
||||
logger.info(f"Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info(f"Clerk domain: {clerk_domain}")
|
||||
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
|
||||
logger.debug(f"Token endpoint: {config.token_endpoint}")
|
||||
|
||||
return config
|
||||
|
||||
|
||||
def get_jwt_secret() -> str:
|
||||
"""Get JWT secret for token signing"""
|
||||
jwt_secret = os.getenv("JWT_SECRET_KEY")
|
||||
|
||||
if not jwt_secret:
|
||||
raise ValueError("JWT_SECRET_KEY environment variable is required")
|
||||
|
||||
return jwt_secret
|
||||
|
||||
|
||||
def create_mcp_server_config():
|
||||
"""Create complete MCP server configuration for Clerk integration"""
|
||||
|
||||
try:
|
||||
oauth_config = create_clerk_oauth_config()
|
||||
jwt_secret = get_jwt_secret()
|
||||
|
||||
return {
|
||||
"oauth_config": oauth_config,
|
||||
"jwt_secret": jwt_secret,
|
||||
"base_url": os.getenv("BASE_URL", "https://yargi-mcp.fly.dev"),
|
||||
"auth_enabled": os.getenv("ENABLE_AUTH", "true").lower() == "true"
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create MCP server config: {e}")
|
||||
raise
|
||||
@@ -1,315 +0,0 @@
|
||||
"""
|
||||
MCP server middleware for OAuth authentication and authorization
|
||||
"""
|
||||
|
||||
import functools
|
||||
import logging
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from fastmcp import FastMCP
|
||||
FASTMCP_AVAILABLE = True
|
||||
except ImportError:
|
||||
FASTMCP_AVAILABLE = False
|
||||
FastMCP = None
|
||||
logger.warning("FastMCP not available, some features will be disabled")
|
||||
|
||||
from .oauth import OAuthProvider
|
||||
from .policy import PolicyEngine
|
||||
|
||||
|
||||
@dataclass
|
||||
class AuthContext:
|
||||
"""Authentication context passed to MCP tools"""
|
||||
|
||||
user_id: str
|
||||
scopes: list[str]
|
||||
claims: dict[str, Any]
|
||||
token: str
|
||||
|
||||
|
||||
class MCPAuthMiddleware:
|
||||
"""Authentication middleware for MCP servers"""
|
||||
|
||||
def __init__(self, oauth_provider: OAuthProvider, policy_engine: PolicyEngine):
|
||||
self.oauth_provider = oauth_provider
|
||||
self.policy_engine = policy_engine
|
||||
|
||||
def authenticate_request(self, authorization_header: str) -> AuthContext | None:
|
||||
"""Extract and validate auth token from request"""
|
||||
|
||||
if not authorization_header:
|
||||
logger.debug("No authorization header provided")
|
||||
return None
|
||||
|
||||
if not authorization_header.startswith("Bearer "):
|
||||
logger.debug("Authorization header does not start with 'Bearer '")
|
||||
return None
|
||||
|
||||
token = authorization_header[7:] # Remove 'Bearer ' prefix
|
||||
|
||||
token_info = self.oauth_provider.introspect_token(token)
|
||||
|
||||
if not token_info.get("active"):
|
||||
logger.warning("Token is not active")
|
||||
return None
|
||||
|
||||
logger.debug(f"Authenticated user: {token_info.get('sub', 'unknown')}")
|
||||
|
||||
return AuthContext(
|
||||
user_id=token_info.get("sub", "unknown"),
|
||||
scopes=token_info.get("mcp_tool_scopes", []),
|
||||
claims=token_info,
|
||||
token=token,
|
||||
)
|
||||
|
||||
def authorize_tool_call(
|
||||
self, tool_name: str, auth_context: AuthContext
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Check if user can call the specified tool"""
|
||||
|
||||
return self.policy_engine.authorize_tool_call(
|
||||
tool_name=tool_name,
|
||||
user_scopes=auth_context.scopes,
|
||||
user_claims=auth_context.claims,
|
||||
)
|
||||
|
||||
|
||||
def auth_required(
|
||||
oauth_provider: OAuthProvider,
|
||||
policy_engine: PolicyEngine,
|
||||
tool_name: str | None = None,
|
||||
):
|
||||
"""
|
||||
Decorator to require authentication for MCP tool functions
|
||||
|
||||
Usage:
|
||||
@auth_required(oauth_provider, policy_engine, "search_yargitay")
|
||||
def my_tool_function(context: AuthContext, ...):
|
||||
pass
|
||||
"""
|
||||
|
||||
def decorator(func: Callable) -> Callable:
|
||||
middleware = MCPAuthMiddleware(oauth_provider, policy_engine)
|
||||
|
||||
@functools.wraps(func)
|
||||
async def wrapper(*args, **kwargs):
|
||||
# Extract authorization header from kwargs
|
||||
auth_header = kwargs.pop("authorization", None)
|
||||
|
||||
# Also check in args if it's a Request object
|
||||
if not auth_header and args:
|
||||
for arg in args:
|
||||
if hasattr(arg, 'headers'):
|
||||
auth_header = arg.headers.get("Authorization")
|
||||
break
|
||||
|
||||
if not auth_header:
|
||||
logger.warning(f"No authorization header for tool '{tool_name or func.__name__}'")
|
||||
raise PermissionError("Authorization header required")
|
||||
|
||||
auth_context = middleware.authenticate_request(auth_header)
|
||||
|
||||
if not auth_context:
|
||||
logger.warning(f"Authentication failed for tool '{tool_name or func.__name__}'")
|
||||
raise PermissionError("Invalid or expired token")
|
||||
|
||||
actual_tool_name = tool_name or func.__name__
|
||||
|
||||
authorized, reason = middleware.authorize_tool_call(
|
||||
actual_tool_name, auth_context
|
||||
)
|
||||
|
||||
if not authorized:
|
||||
logger.warning(f"Authorization failed for tool '{actual_tool_name}': {reason}")
|
||||
raise PermissionError(f"Access denied: {reason}")
|
||||
|
||||
# Add auth context to function call
|
||||
return await func(auth_context, *args, **kwargs)
|
||||
|
||||
return wrapper
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
class FastMCPAuthWrapper:
|
||||
"""Wrapper for FastMCP servers to add authentication"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
mcp_server: "FastMCP",
|
||||
oauth_provider: OAuthProvider,
|
||||
policy_engine: PolicyEngine,
|
||||
):
|
||||
if not FASTMCP_AVAILABLE:
|
||||
raise ImportError("FastMCP is required for FastMCPAuthWrapper")
|
||||
|
||||
self.mcp_server = mcp_server
|
||||
self.middleware = MCPAuthMiddleware(oauth_provider, policy_engine)
|
||||
self.oauth_provider = oauth_provider
|
||||
logger.info("Initializing FastMCP authentication wrapper")
|
||||
self._wrap_tools()
|
||||
|
||||
def _wrap_tools(self):
|
||||
"""Wrap all existing tools with auth middleware"""
|
||||
|
||||
# Try different FastMCP tool storage locations
|
||||
tool_registry = None
|
||||
|
||||
if hasattr(self.mcp_server, '_tools'):
|
||||
tool_registry = self.mcp_server._tools
|
||||
elif hasattr(self.mcp_server, 'tools'):
|
||||
tool_registry = self.mcp_server.tools
|
||||
elif hasattr(self.mcp_server, '_tool_registry'):
|
||||
tool_registry = self.mcp_server._tool_registry
|
||||
elif hasattr(self.mcp_server, '_handlers') and hasattr(self.mcp_server._handlers, 'tools'):
|
||||
tool_registry = self.mcp_server._handlers.tools
|
||||
|
||||
if not tool_registry:
|
||||
logger.warning("FastMCP server tool registry not found, tools will not be automatically wrapped")
|
||||
logger.debug(f"Available server attributes: {dir(self.mcp_server)}")
|
||||
return
|
||||
|
||||
logger.debug(f"Found tool registry with {len(tool_registry)} tools")
|
||||
original_tools = dict(tool_registry)
|
||||
wrapped_count = 0
|
||||
|
||||
for tool_name, tool_func in original_tools.items():
|
||||
try:
|
||||
wrapped_func = self._create_auth_wrapper(tool_name, tool_func)
|
||||
tool_registry[tool_name] = wrapped_func
|
||||
wrapped_count += 1
|
||||
logger.debug(f"Wrapped tool: {tool_name}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to wrap tool {tool_name}: {e}")
|
||||
|
||||
logger.info(f"Successfully wrapped {wrapped_count} tools with authentication")
|
||||
|
||||
def _create_auth_wrapper(self, tool_name: str, original_func: Callable) -> Callable:
|
||||
"""Create auth wrapper for a specific tool"""
|
||||
|
||||
@functools.wraps(original_func)
|
||||
async def auth_wrapper(*args, **kwargs):
|
||||
# Extract authorization from various sources
|
||||
auth_header = None
|
||||
|
||||
# Check kwargs first
|
||||
auth_header = kwargs.pop("authorization", None)
|
||||
|
||||
# Check if first argument is a Request object
|
||||
if not auth_header and args:
|
||||
first_arg = args[0]
|
||||
if hasattr(first_arg, 'headers'):
|
||||
auth_header = first_arg.headers.get("Authorization")
|
||||
|
||||
if not auth_header:
|
||||
logger.warning(f"No authorization header for tool '{tool_name}'")
|
||||
raise PermissionError("Authorization required")
|
||||
|
||||
auth_context = self.middleware.authenticate_request(auth_header)
|
||||
|
||||
if not auth_context:
|
||||
logger.warning(f"Authentication failed for tool '{tool_name}'")
|
||||
raise PermissionError("Invalid token")
|
||||
|
||||
authorized, reason = self.middleware.authorize_tool_call(
|
||||
tool_name, auth_context
|
||||
)
|
||||
|
||||
if not authorized:
|
||||
logger.warning(f"Authorization failed for tool '{tool_name}': {reason}")
|
||||
raise PermissionError(f"Access denied: {reason}")
|
||||
|
||||
# Add auth context to kwargs
|
||||
kwargs["auth_context"] = auth_context
|
||||
logger.debug(f"Calling tool '{tool_name}' for user {auth_context.user_id}")
|
||||
|
||||
return await original_func(*args, **kwargs)
|
||||
|
||||
return auth_wrapper
|
||||
|
||||
def add_oauth_endpoints(self):
|
||||
"""Add OAuth endpoints to the MCP server"""
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Initiate OAuth 2.1 authorization flow with PKCE",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_authorize(redirect_uri: str, scopes: Optional[str] = None):
|
||||
"""OAuth authorization endpoint"""
|
||||
scope_list = scopes.split(" ") if scopes else None
|
||||
auth_url, pkce = self.oauth_provider.generate_authorization_url(
|
||||
redirect_uri=redirect_uri, scopes=scope_list
|
||||
)
|
||||
logger.info(f"Generated authorization URL for redirect_uri: {redirect_uri}")
|
||||
return {
|
||||
"authorization_url": auth_url,
|
||||
"code_verifier": pkce.verifier, # For PKCE flow
|
||||
"code_challenge": pkce.challenge,
|
||||
"instructions": "Use the authorization_url to complete OAuth flow, then exchange the returned code using oauth_token tool"
|
||||
}
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Exchange OAuth authorization code for access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_token(
|
||||
code: str,
|
||||
state: str,
|
||||
redirect_uri: str
|
||||
):
|
||||
"""OAuth token exchange endpoint"""
|
||||
try:
|
||||
result = await self.oauth_provider.exchange_code_for_token(
|
||||
code=code, state=state, redirect_uri=redirect_uri
|
||||
)
|
||||
logger.info("Successfully exchanged authorization code for token")
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.error(f"Token exchange failed: {e}")
|
||||
raise
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Validate and introspect OAuth access token",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_introspect(token: str):
|
||||
"""Token introspection endpoint"""
|
||||
result = self.oauth_provider.introspect_token(token)
|
||||
logger.debug(f"Token introspection: active={result.get('active', False)}")
|
||||
return result
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Revoke OAuth access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_revoke(token: str):
|
||||
"""Token revocation endpoint"""
|
||||
success = self.oauth_provider.revoke_token(token)
|
||||
logger.info(f"Token revocation: success={success}")
|
||||
return {"revoked": success}
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Get list of tools available to authenticated user",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_user_tools(authorization: str):
|
||||
"""Get user's allowed tools based on scopes"""
|
||||
auth_context = self.middleware.authenticate_request(authorization)
|
||||
if not auth_context:
|
||||
raise PermissionError("Invalid token")
|
||||
|
||||
allowed_patterns = self.middleware.policy_engine.get_allowed_tools(auth_context.scopes)
|
||||
|
||||
return {
|
||||
"user_id": auth_context.user_id,
|
||||
"scopes": auth_context.scopes,
|
||||
"allowed_tool_patterns": allowed_patterns,
|
||||
"message": "Use these patterns to determine which tools you can access"
|
||||
}
|
||||
|
||||
logger.info("Added OAuth endpoints: oauth_authorize, oauth_token, oauth_introspect, oauth_revoke, oauth_user_tools")
|
||||
@@ -1,304 +0,0 @@
|
||||
"""
|
||||
OAuth 2.1 + PKCE implementation for MCP servers with Clerk integration
|
||||
"""
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import secrets
|
||||
import time
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Any, Optional
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import httpx
|
||||
import jwt
|
||||
from jwt.exceptions import PyJWTError, InvalidTokenError
|
||||
|
||||
from .storage import PersistentStorage
|
||||
|
||||
# Try to import Clerk SDK
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class OAuthConfig:
|
||||
"""OAuth provider configuration for Clerk"""
|
||||
|
||||
client_id: str
|
||||
client_secret: str
|
||||
authorization_endpoint: str
|
||||
token_endpoint: str
|
||||
jwks_uri: str | None = None
|
||||
issuer: str = "mcp-auth"
|
||||
scopes: list[str] = None
|
||||
|
||||
def __post_init__(self):
|
||||
if self.scopes is None:
|
||||
self.scopes = ["mcp:tools:read", "mcp:tools:write"]
|
||||
|
||||
|
||||
class PKCEChallenge:
|
||||
"""PKCE challenge/verifier pair for OAuth 2.1"""
|
||||
|
||||
def __init__(self):
|
||||
self.verifier = (
|
||||
base64.urlsafe_b64encode(secrets.token_bytes(32))
|
||||
.decode("utf-8")
|
||||
.rstrip("=")
|
||||
)
|
||||
|
||||
challenge_bytes = hashlib.sha256(self.verifier.encode("utf-8")).digest()
|
||||
self.challenge = (
|
||||
base64.urlsafe_b64encode(challenge_bytes).decode("utf-8").rstrip("=")
|
||||
)
|
||||
|
||||
|
||||
class OAuthProvider:
|
||||
"""OAuth 2.1 provider with PKCE support and Clerk integration"""
|
||||
|
||||
def __init__(self, config: OAuthConfig, jwt_secret: str):
|
||||
self.config = config
|
||||
self.jwt_secret = jwt_secret
|
||||
# Use persistent storage instead of memory
|
||||
self.storage = PersistentStorage()
|
||||
|
||||
# Initialize Clerk SDK if available
|
||||
self.clerk = None
|
||||
if CLERK_AVAILABLE and config.client_secret:
|
||||
try:
|
||||
self.clerk = Clerk(bearer_auth=config.client_secret)
|
||||
logger.info("Clerk SDK initialized successfully")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to initialize Clerk SDK: {e}")
|
||||
|
||||
logger.info("OAuth provider initialized with persistent storage")
|
||||
|
||||
def generate_authorization_url(
|
||||
self,
|
||||
redirect_uri: str,
|
||||
state: str | None = None,
|
||||
scopes: list[str] | None = None,
|
||||
) -> tuple[str, PKCEChallenge]:
|
||||
"""Generate OAuth authorization URL with PKCE for Clerk"""
|
||||
|
||||
pkce = PKCEChallenge()
|
||||
session_id = secrets.token_urlsafe(32)
|
||||
|
||||
if state is None:
|
||||
state = secrets.token_urlsafe(16)
|
||||
|
||||
if scopes is None:
|
||||
scopes = self.config.scopes
|
||||
|
||||
# Store session data with expiration
|
||||
session_data = {
|
||||
"pkce_verifier": pkce.verifier,
|
||||
"state": state,
|
||||
"redirect_uri": redirect_uri,
|
||||
"scopes": scopes,
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=10)).timestamp(),
|
||||
}
|
||||
self.storage.set_session(session_id, session_data)
|
||||
|
||||
# Build Clerk OAuth URL
|
||||
# Check if this is a custom domain (sign-in endpoint)
|
||||
if self.config.authorization_endpoint.endswith('/sign-in'):
|
||||
# For custom domains, Clerk expects redirect_url parameter
|
||||
params = {
|
||||
"redirect_url": redirect_uri,
|
||||
"state": f"{state}:{session_id}",
|
||||
}
|
||||
auth_url = f"{self.config.authorization_endpoint}?{urlencode(params)}"
|
||||
else:
|
||||
# Standard OAuth flow with PKCE
|
||||
params = {
|
||||
"response_type": "code",
|
||||
"client_id": self.config.client_id,
|
||||
"redirect_uri": redirect_uri,
|
||||
"scope": " ".join(scopes),
|
||||
"state": f"{state}:{session_id}", # Combine state with session ID
|
||||
"code_challenge": pkce.challenge,
|
||||
"code_challenge_method": "S256",
|
||||
}
|
||||
auth_url = f"{self.config.authorization_endpoint}?{urlencode(params)}"
|
||||
|
||||
logger.info(f"Generated OAuth URL with session {session_id[:8]}...")
|
||||
logger.debug(f"Auth URL: {auth_url}")
|
||||
return auth_url, pkce
|
||||
|
||||
async def exchange_code_for_token(
|
||||
self, code: str, state: str, redirect_uri: str
|
||||
) -> dict[str, Any]:
|
||||
"""Exchange authorization code for access token with Clerk"""
|
||||
|
||||
try:
|
||||
original_state, session_id = state.split(":", 1)
|
||||
except ValueError as e:
|
||||
logger.error(f"Invalid state format: {state}")
|
||||
raise ValueError("Invalid state format") from e
|
||||
|
||||
session = self.storage.get_session(session_id)
|
||||
if not session:
|
||||
logger.error(f"Session {session_id} not found")
|
||||
raise ValueError("Invalid session")
|
||||
|
||||
# Check session expiration
|
||||
if datetime.utcnow().timestamp() > session.get("expires_at", 0):
|
||||
self.storage.delete_session(session_id)
|
||||
logger.error(f"Session {session_id} expired")
|
||||
raise ValueError("Session expired")
|
||||
|
||||
if session["state"] != original_state:
|
||||
logger.error(f"State mismatch: expected {session['state']}, got {original_state}")
|
||||
raise ValueError("State mismatch")
|
||||
|
||||
if session["redirect_uri"] != redirect_uri:
|
||||
logger.error(f"Redirect URI mismatch: expected {session['redirect_uri']}, got {redirect_uri}")
|
||||
raise ValueError("Redirect URI mismatch")
|
||||
|
||||
# Prepare token exchange request for Clerk
|
||||
token_data = {
|
||||
"grant_type": "authorization_code",
|
||||
"client_id": self.config.client_id,
|
||||
"client_secret": self.config.client_secret,
|
||||
"code": code,
|
||||
"redirect_uri": redirect_uri,
|
||||
"code_verifier": session["pkce_verifier"],
|
||||
}
|
||||
|
||||
logger.info(f"Exchanging code with Clerk for session {session_id[:8]}...")
|
||||
|
||||
async with httpx.AsyncClient() as client:
|
||||
response = await client.post(
|
||||
self.config.token_endpoint,
|
||||
data=token_data,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
||||
timeout=30.0,
|
||||
)
|
||||
|
||||
if response.status_code != 200:
|
||||
logger.error(f"Clerk token exchange failed: {response.status_code} - {response.text}")
|
||||
raise ValueError(f"Token exchange failed: {response.text}")
|
||||
|
||||
token_response = response.json()
|
||||
logger.info("Successfully exchanged code for Clerk token")
|
||||
|
||||
# Create MCP-scoped JWT token
|
||||
access_token = self._create_mcp_token(
|
||||
session["scopes"], token_response.get("access_token"), session_id
|
||||
)
|
||||
|
||||
# Store token for introspection
|
||||
token_id = secrets.token_urlsafe(16)
|
||||
token_data = {
|
||||
"access_token": access_token,
|
||||
"scopes": session["scopes"],
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(hours=1)).timestamp(),
|
||||
"session_id": session_id,
|
||||
"clerk_token": token_response.get("access_token"),
|
||||
}
|
||||
self.storage.set_token(token_id, token_data)
|
||||
|
||||
# Clean up session
|
||||
self.storage.delete_session(session_id)
|
||||
|
||||
return {
|
||||
"access_token": access_token,
|
||||
"token_type": "bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": " ".join(session["scopes"]),
|
||||
}
|
||||
|
||||
def validate_pkce(self, code_verifier: str, code_challenge: str) -> bool:
|
||||
"""Validate PKCE code challenge (RFC 7636)"""
|
||||
# S256 method
|
||||
verifier_hash = hashlib.sha256(code_verifier.encode()).digest()
|
||||
expected_challenge = base64.urlsafe_b64encode(verifier_hash).decode().rstrip('=')
|
||||
return expected_challenge == code_challenge
|
||||
|
||||
def _create_mcp_token(
|
||||
self, scopes: list[str], upstream_token: str, session_id: str
|
||||
) -> str:
|
||||
"""Create MCP-scoped JWT token with Clerk token embedded"""
|
||||
|
||||
now = int(time.time())
|
||||
payload = {
|
||||
"iss": self.config.issuer,
|
||||
"sub": session_id,
|
||||
"aud": "mcp-server",
|
||||
"iat": now,
|
||||
"exp": now + 3600, # 1 hour expiration
|
||||
"mcp_tool_scopes": scopes,
|
||||
"upstream_token": upstream_token,
|
||||
"clerk_integration": True,
|
||||
}
|
||||
|
||||
return jwt.encode(payload, self.jwt_secret, algorithm="HS256")
|
||||
|
||||
def introspect_token(self, token: str) -> dict[str, Any]:
|
||||
"""Introspect and validate MCP token"""
|
||||
|
||||
try:
|
||||
payload = jwt.decode(token, self.jwt_secret, algorithms=["HS256"])
|
||||
|
||||
# Check if token is expired
|
||||
if payload.get("exp", 0) < time.time():
|
||||
return {"active": False, "error": "token_expired"}
|
||||
|
||||
return {
|
||||
"active": True,
|
||||
"sub": payload.get("sub"),
|
||||
"aud": payload.get("aud"),
|
||||
"iss": payload.get("iss"),
|
||||
"exp": payload.get("exp"),
|
||||
"iat": payload.get("iat"),
|
||||
"mcp_tool_scopes": payload.get("mcp_tool_scopes", []),
|
||||
"upstream_token": payload.get("upstream_token"),
|
||||
"clerk_integration": payload.get("clerk_integration", False),
|
||||
}
|
||||
|
||||
except PyJWTError as e:
|
||||
logger.warning(f"Token validation failed: {e}")
|
||||
return {"active": False, "error": "invalid_token"}
|
||||
|
||||
def revoke_token(self, token: str) -> bool:
|
||||
"""Revoke a token"""
|
||||
|
||||
try:
|
||||
payload = jwt.decode(token, self.jwt_secret, algorithms=["HS256"])
|
||||
session_id = payload.get("sub")
|
||||
|
||||
# Remove all tokens associated with this session
|
||||
all_tokens = self.storage.get_tokens()
|
||||
tokens_to_remove = [
|
||||
token_id
|
||||
for token_id, token_data in all_tokens.items()
|
||||
if token_data.get("session_id") == session_id
|
||||
]
|
||||
|
||||
for token_id in tokens_to_remove:
|
||||
self.storage.delete_token(token_id)
|
||||
|
||||
logger.info(f"Revoked {len(tokens_to_remove)} tokens for session {session_id}")
|
||||
return True
|
||||
|
||||
except InvalidTokenError as e:
|
||||
logger.warning(f"Token revocation failed: {e}")
|
||||
return False
|
||||
|
||||
def cleanup_expired_sessions(self):
|
||||
"""Clean up expired sessions and tokens"""
|
||||
# This is now handled automatically by persistent storage
|
||||
self.storage.cleanup_expired_sessions()
|
||||
logger.debug("Cleanup completed via persistent storage")
|
||||
@@ -1,201 +0,0 @@
|
||||
"""
|
||||
Authorization policy engine for MCP tools
|
||||
"""
|
||||
|
||||
import re
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PolicyAction(Enum):
|
||||
ALLOW = "allow"
|
||||
DENY = "deny"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ToolPolicy:
|
||||
"""Policy rule for MCP tool access"""
|
||||
|
||||
tool_pattern: str # regex pattern for tool names
|
||||
required_scopes: list[str]
|
||||
action: PolicyAction = PolicyAction.ALLOW
|
||||
conditions: dict[str, Any] | None = None
|
||||
|
||||
def matches_tool(self, tool_name: str) -> bool:
|
||||
"""Check if the policy applies to given tool"""
|
||||
return bool(re.match(self.tool_pattern, tool_name))
|
||||
|
||||
def evaluate_scopes(self, user_scopes: list[str]) -> bool:
|
||||
"""Check if user has required scopes"""
|
||||
return all(scope in user_scopes for scope in self.required_scopes)
|
||||
|
||||
|
||||
class PolicyEngine:
|
||||
"""Authorization policy engine for Turkish legal database tools"""
|
||||
|
||||
def __init__(self):
|
||||
self.policies: list[ToolPolicy] = []
|
||||
self.default_action = PolicyAction.DENY
|
||||
|
||||
def add_policy(self, policy: ToolPolicy):
|
||||
"""Add a policy rule"""
|
||||
self.policies.append(policy)
|
||||
logger.debug(f"Added policy: {policy.tool_pattern} -> {policy.required_scopes}")
|
||||
|
||||
def add_tool_scope_policy(
|
||||
self,
|
||||
tool_pattern: str,
|
||||
required_scopes: str | list[str],
|
||||
action: PolicyAction = PolicyAction.ALLOW,
|
||||
):
|
||||
"""Convenience method to add tool-scope policy"""
|
||||
if isinstance(required_scopes, str):
|
||||
required_scopes = [required_scopes]
|
||||
|
||||
policy = ToolPolicy(
|
||||
tool_pattern=tool_pattern, required_scopes=required_scopes, action=action
|
||||
)
|
||||
self.add_policy(policy)
|
||||
|
||||
def authorize_tool_call(
|
||||
self,
|
||||
tool_name: str,
|
||||
user_scopes: list[str],
|
||||
user_claims: dict[str, Any] | None = None,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""
|
||||
Authorize a tool call
|
||||
|
||||
Returns:
|
||||
(authorized: bool, reason: Optional[str])
|
||||
"""
|
||||
|
||||
logger.debug(f"Authorizing tool '{tool_name}' for user with scopes: {user_scopes}")
|
||||
|
||||
matching_policies = [
|
||||
policy for policy in self.policies if policy.matches_tool(tool_name)
|
||||
]
|
||||
|
||||
if not matching_policies:
|
||||
if self.default_action == PolicyAction.ALLOW:
|
||||
logger.debug(f"No policies found for '{tool_name}', allowing by default")
|
||||
return True, None
|
||||
else:
|
||||
logger.warning(f"No policies found for '{tool_name}', denying by default")
|
||||
return False, f"No policy found for tool '{tool_name}', default deny"
|
||||
|
||||
# Check for explicit deny policies first
|
||||
for policy in matching_policies:
|
||||
if policy.action == PolicyAction.DENY:
|
||||
if policy.evaluate_scopes(user_scopes):
|
||||
logger.warning(f"Explicit deny policy matched for '{tool_name}'")
|
||||
return False, f"Explicit deny policy for tool '{tool_name}'"
|
||||
|
||||
# Check allow policies
|
||||
allow_policies = [
|
||||
p for p in matching_policies if p.action == PolicyAction.ALLOW
|
||||
]
|
||||
|
||||
if not allow_policies:
|
||||
logger.warning(f"No allow policies found for '{tool_name}'")
|
||||
return False, f"No allow policies found for tool '{tool_name}'"
|
||||
|
||||
for policy in allow_policies:
|
||||
if policy.evaluate_scopes(user_scopes):
|
||||
if self._evaluate_conditions(policy.conditions, user_claims):
|
||||
logger.debug(f"Authorization granted for '{tool_name}'")
|
||||
return True, None
|
||||
|
||||
logger.warning(f"Insufficient scopes for '{tool_name}'. Required: {[p.required_scopes for p in allow_policies]}, User has: {user_scopes}")
|
||||
return False, f"Insufficient scopes for tool '{tool_name}'"
|
||||
|
||||
def _evaluate_conditions(
|
||||
self,
|
||||
conditions: dict[str, Any] | None,
|
||||
user_claims: dict[str, Any] | None,
|
||||
) -> bool:
|
||||
"""Evaluate additional policy conditions"""
|
||||
|
||||
if not conditions:
|
||||
return True
|
||||
|
||||
if not user_claims:
|
||||
logger.debug("No user claims provided, conditions evaluation failed")
|
||||
return False
|
||||
|
||||
for key, expected_value in conditions.items():
|
||||
user_value = user_claims.get(key)
|
||||
|
||||
if isinstance(expected_value, list):
|
||||
if user_value not in expected_value:
|
||||
logger.debug(f"Condition failed: {key} = {user_value} not in {expected_value}")
|
||||
return False
|
||||
elif user_value != expected_value:
|
||||
logger.debug(f"Condition failed: {key} = {user_value} != {expected_value}")
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def get_allowed_tools(self, user_scopes: list[str]) -> list[str]:
|
||||
"""Get list of tool patterns user is allowed to call"""
|
||||
|
||||
allowed_tools = []
|
||||
|
||||
for policy in self.policies:
|
||||
if policy.action == PolicyAction.ALLOW and policy.evaluate_scopes(
|
||||
user_scopes
|
||||
):
|
||||
allowed_tools.append(policy.tool_pattern)
|
||||
|
||||
return allowed_tools
|
||||
|
||||
|
||||
def create_turkish_legal_policies() -> PolicyEngine:
|
||||
"""Create policy set for Turkish legal database MCP server"""
|
||||
|
||||
engine = PolicyEngine()
|
||||
|
||||
# Administrative tools (full access)
|
||||
engine.add_tool_scope_policy(".*", ["mcp:tools:admin"])
|
||||
|
||||
# Search tools - require read access
|
||||
engine.add_tool_scope_policy("search.*", ["mcp:tools:read"])
|
||||
|
||||
# Fetch/get document tools - require read access
|
||||
engine.add_tool_scope_policy("get_.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("fetch.*", ["mcp:tools:read"])
|
||||
|
||||
# Specific Turkish legal database tools
|
||||
engine.add_tool_scope_policy("search_yargitay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_danistay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_anayasa.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_rekabet.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_kik.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_emsal.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_uyusmazlik.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_sayistay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_.*_bedesten", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_yerel_hukuk.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_istinaf_hukuk.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_kyb.*", ["mcp:tools:read"])
|
||||
|
||||
# Document retrieval tools
|
||||
engine.add_tool_scope_policy("get_.*_document.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("get_.*_markdown", ["mcp:tools:read"])
|
||||
|
||||
# Write operations (if any future tools need them)
|
||||
engine.add_tool_scope_policy("create_.*", ["mcp:tools:write"])
|
||||
engine.add_tool_scope_policy("update_.*", ["mcp:tools:write"])
|
||||
engine.add_tool_scope_policy("delete_.*", ["mcp:tools:write"])
|
||||
|
||||
logger.info("Created Turkish legal database policy engine")
|
||||
return engine
|
||||
|
||||
|
||||
def create_default_policies() -> PolicyEngine:
|
||||
"""Create a default policy set for MCP servers (backwards compatibility)"""
|
||||
return create_turkish_legal_policies()
|
||||
@@ -1,112 +0,0 @@
|
||||
"""
|
||||
Persistent storage for OAuth sessions and tokens
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Dict, Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PersistentStorage:
|
||||
"""File-based persistent storage for OAuth data"""
|
||||
|
||||
def __init__(self, storage_dir: str = None):
|
||||
if storage_dir is None:
|
||||
# Use system temp directory or environment variable
|
||||
storage_dir = os.environ.get('TEMP', tempfile.gettempdir())
|
||||
|
||||
self.storage_dir = os.path.join(storage_dir, 'mcp_oauth_storage')
|
||||
os.makedirs(self.storage_dir, exist_ok=True)
|
||||
|
||||
self.sessions_file = os.path.join(self.storage_dir, 'oauth_sessions.json')
|
||||
self.tokens_file = os.path.join(self.storage_dir, 'oauth_tokens.json')
|
||||
|
||||
logger.info(f"Persistent OAuth storage initialized at: {self.storage_dir}")
|
||||
|
||||
def _load_json(self, filepath: str) -> Dict:
|
||||
"""Load JSON data from file"""
|
||||
try:
|
||||
if os.path.exists(filepath):
|
||||
with open(filepath, 'r', encoding='utf-8') as f:
|
||||
return json.load(f)
|
||||
except Exception as e:
|
||||
logger.error(f"Error loading {filepath}: {e}")
|
||||
return {}
|
||||
|
||||
def _save_json(self, filepath: str, data: Dict):
|
||||
"""Save JSON data to file"""
|
||||
try:
|
||||
with open(filepath, 'w', encoding='utf-8') as f:
|
||||
json.dump(data, f, indent=2, default=str)
|
||||
except Exception as e:
|
||||
logger.error(f"Error saving {filepath}: {e}")
|
||||
|
||||
def get_sessions(self) -> Dict[str, Dict[str, Any]]:
|
||||
"""Get all OAuth sessions"""
|
||||
data = self._load_json(self.sessions_file)
|
||||
# Clean expired sessions
|
||||
now = datetime.utcnow().timestamp()
|
||||
valid_sessions = {k: v for k, v in data.items()
|
||||
if v.get('expires_at', 0) > now}
|
||||
if len(valid_sessions) != len(data):
|
||||
self._save_json(self.sessions_file, valid_sessions)
|
||||
return valid_sessions
|
||||
|
||||
def set_session(self, session_id: str, data: Dict[str, Any]):
|
||||
"""Set OAuth session data"""
|
||||
sessions = self.get_sessions()
|
||||
sessions[session_id] = data
|
||||
self._save_json(self.sessions_file, sessions)
|
||||
|
||||
def get_session(self, session_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Get specific OAuth session data"""
|
||||
sessions = self.get_sessions()
|
||||
return sessions.get(session_id)
|
||||
|
||||
def delete_session(self, session_id: str):
|
||||
"""Delete OAuth session"""
|
||||
sessions = self.get_sessions()
|
||||
if session_id in sessions:
|
||||
del sessions[session_id]
|
||||
self._save_json(self.sessions_file, sessions)
|
||||
|
||||
def get_tokens(self) -> Dict[str, Dict[str, Any]]:
|
||||
"""Get all OAuth tokens"""
|
||||
data = self._load_json(self.tokens_file)
|
||||
# Clean expired tokens
|
||||
now = datetime.utcnow().timestamp()
|
||||
valid_tokens = {k: v for k, v in data.items()
|
||||
if v.get('expires_at', 0) > now}
|
||||
if len(valid_tokens) != len(data):
|
||||
self._save_json(self.tokens_file, valid_tokens)
|
||||
return valid_tokens
|
||||
|
||||
def set_token(self, token_id: str, token_data: Dict[str, Any]):
|
||||
"""Set OAuth token data"""
|
||||
tokens = self.get_tokens()
|
||||
tokens[token_id] = token_data
|
||||
self._save_json(self.tokens_file, tokens)
|
||||
|
||||
def get_token(self, token_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Get specific OAuth token data"""
|
||||
tokens = self.get_tokens()
|
||||
return tokens.get(token_id)
|
||||
|
||||
def delete_token(self, token_id: str):
|
||||
"""Delete OAuth token"""
|
||||
tokens = self.get_tokens()
|
||||
if token_id in tokens:
|
||||
del tokens[token_id]
|
||||
self._save_json(self.tokens_file, tokens)
|
||||
|
||||
def cleanup_expired_sessions(self):
|
||||
"""Clean up expired sessions and tokens"""
|
||||
# This is handled automatically in get_sessions() and get_tokens()
|
||||
sessions = self.get_sessions()
|
||||
tokens = self.get_tokens()
|
||||
logger.debug(f"Cleanup: {len(sessions)} active sessions, {len(tokens)} active tokens")
|
||||
@@ -1,193 +0,0 @@
|
||||
"""
|
||||
Factory for creating FastMCP app with MCP Auth Toolkit integration
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from fastmcp import FastMCP
|
||||
FASTMCP_AVAILABLE = True
|
||||
except ImportError:
|
||||
FASTMCP_AVAILABLE = False
|
||||
FastMCP = None
|
||||
|
||||
from mcp_auth import (
|
||||
OAuthProvider,
|
||||
PolicyEngine,
|
||||
FastMCPAuthWrapper,
|
||||
create_default_policies
|
||||
)
|
||||
from mcp_auth.clerk_config import create_mcp_server_config
|
||||
|
||||
|
||||
def create_auth_enabled_app(app_name: str = "Yargı MCP Server") -> FastMCP:
|
||||
"""Create FastMCP app with authentication enabled"""
|
||||
|
||||
if not FASTMCP_AVAILABLE:
|
||||
raise ImportError("FastMCP is required for authenticated MCP server")
|
||||
|
||||
logger.info("Creating FastMCP app with MCP Auth Toolkit integration")
|
||||
|
||||
# Create base FastMCP app
|
||||
app = FastMCP(app_name)
|
||||
|
||||
# Check if authentication is enabled
|
||||
auth_enabled = os.getenv("ENABLE_AUTH", "true").lower() == "true"
|
||||
|
||||
if not auth_enabled:
|
||||
logger.info("Authentication disabled, returning basic FastMCP app")
|
||||
return app
|
||||
|
||||
try:
|
||||
# Get configuration
|
||||
logger.info("Getting MCP server configuration...")
|
||||
config = create_mcp_server_config()
|
||||
logger.info("Configuration loaded successfully")
|
||||
|
||||
# Create OAuth provider with Clerk config
|
||||
logger.info("Creating OAuth provider...")
|
||||
oauth_provider = OAuthProvider(
|
||||
config=config["oauth_config"],
|
||||
jwt_secret=config["jwt_secret"]
|
||||
)
|
||||
logger.info("OAuth provider created successfully")
|
||||
|
||||
# Create policy engine for Turkish legal database
|
||||
policy_engine = create_default_policies()
|
||||
|
||||
# Store auth components for later wrapping (after tools are defined)
|
||||
app._oauth_provider = oauth_provider
|
||||
app._policy_engine = policy_engine
|
||||
app._auth_config = config
|
||||
|
||||
# Add OAuth endpoints immediately
|
||||
@app.tool(
|
||||
description="Initiate OAuth 2.1 authorization flow with PKCE",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_authorize(redirect_uri: str, scopes: str = None):
|
||||
"""OAuth authorization endpoint"""
|
||||
scope_list = scopes.split(" ") if scopes else ["mcp:tools:read", "mcp:tools:write"]
|
||||
auth_url, pkce = oauth_provider.generate_authorization_url(
|
||||
redirect_uri=redirect_uri, scopes=scope_list
|
||||
)
|
||||
logger.info(f"Generated authorization URL for redirect_uri: {redirect_uri}")
|
||||
return {
|
||||
"authorization_url": auth_url,
|
||||
"code_verifier": pkce.verifier,
|
||||
"code_challenge": pkce.challenge,
|
||||
"instructions": "Use the authorization_url to complete OAuth flow, then exchange the returned code using oauth_token tool"
|
||||
}
|
||||
|
||||
@app.tool(
|
||||
description="Exchange OAuth authorization code for access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_token(code: str, state: str, redirect_uri: str):
|
||||
"""OAuth token exchange endpoint"""
|
||||
try:
|
||||
result = await oauth_provider.exchange_code_for_token(
|
||||
code=code, state=state, redirect_uri=redirect_uri
|
||||
)
|
||||
logger.info("Successfully exchanged authorization code for token")
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.error(f"Token exchange failed: {e}")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
description="Validate and introspect OAuth access token",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_introspect(token: str):
|
||||
"""Token introspection endpoint"""
|
||||
result = oauth_provider.introspect_token(token)
|
||||
logger.debug(f"Token introspection: active={result.get('active', False)}")
|
||||
return result
|
||||
|
||||
@app.tool(
|
||||
description="Revoke OAuth access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_revoke(token: str):
|
||||
"""Token revocation endpoint"""
|
||||
success = oauth_provider.revoke_token(token)
|
||||
logger.info(f"Token revocation: success={success}")
|
||||
return {"revoked": success}
|
||||
|
||||
logger.info("Successfully created authenticated FastMCP app")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create authenticated app: {e}")
|
||||
logger.info("Falling back to non-authenticated FastMCP app")
|
||||
# Return basic app if auth setup fails
|
||||
return app
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def create_app() -> FastMCP:
|
||||
"""Create FastMCP app (backwards compatible with mcp_factory.py)"""
|
||||
return create_auth_enabled_app()
|
||||
|
||||
|
||||
def get_auth_wrapper(app: FastMCP) -> Optional[FastMCPAuthWrapper]:
|
||||
"""Get auth wrapper from app if available"""
|
||||
return getattr(app, '_auth_wrapper', None)
|
||||
|
||||
|
||||
def get_oauth_provider(app: FastMCP) -> Optional[OAuthProvider]:
|
||||
"""Get OAuth provider from app if available"""
|
||||
return getattr(app, '_oauth_provider', None)
|
||||
|
||||
|
||||
def get_policy_engine(app: FastMCP) -> Optional[PolicyEngine]:
|
||||
"""Get policy engine from app if available"""
|
||||
return getattr(app, '_policy_engine', None)
|
||||
|
||||
|
||||
def is_auth_enabled(app: FastMCP) -> bool:
|
||||
"""Check if authentication is enabled for the app"""
|
||||
return hasattr(app, '_oauth_provider') or hasattr(app, '_auth_wrapper')
|
||||
|
||||
|
||||
def enable_tool_authentication(app: FastMCP):
|
||||
"""Enable authentication on all existing tools (call after tools are defined)"""
|
||||
if not is_auth_enabled(app):
|
||||
logger.debug("Authentication not enabled, skipping tool authentication")
|
||||
return
|
||||
|
||||
oauth_provider = get_oauth_provider(app)
|
||||
policy_engine = get_policy_engine(app)
|
||||
|
||||
if not oauth_provider or not policy_engine:
|
||||
logger.warning("OAuth provider or policy engine not available")
|
||||
return
|
||||
|
||||
try:
|
||||
# Create auth wrapper and wrap tools
|
||||
auth_wrapper = FastMCPAuthWrapper(
|
||||
mcp_server=app,
|
||||
oauth_provider=oauth_provider,
|
||||
policy_engine=policy_engine
|
||||
)
|
||||
|
||||
# Store wrapper for reference
|
||||
app._auth_wrapper = auth_wrapper
|
||||
|
||||
logger.info("Tool authentication enabled successfully")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to enable tool authentication: {e}")
|
||||
|
||||
|
||||
def cleanup_auth_sessions(app: FastMCP):
|
||||
"""Clean up expired auth sessions and tokens"""
|
||||
oauth_provider = get_oauth_provider(app)
|
||||
if oauth_provider:
|
||||
oauth_provider.cleanup_expired_sessions()
|
||||
logger.debug("Cleaned up expired OAuth sessions")
|
||||
@@ -1,338 +0,0 @@
|
||||
"""
|
||||
HTTP adapter for MCP Auth Toolkit OAuth endpoints
|
||||
Exposes MCP OAuth tools as HTTP endpoints for Claude.ai integration
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
import secrets
|
||||
import time
|
||||
from typing import Optional
|
||||
from urllib.parse import urlencode, quote
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
from fastapi import APIRouter, Request, Query, HTTPException
|
||||
from fastapi.responses import RedirectResponse, JSONResponse
|
||||
|
||||
# Try to import Clerk SDK
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError as e:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
# OAuth configuration
|
||||
BASE_URL = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
|
||||
@router.get("/.well-known/oauth-authorization-server")
|
||||
async def get_oauth_metadata():
|
||||
"""OAuth 2.0 Authorization Server Metadata (RFC 8414)"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": f"{BASE_URL}/authorize",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"token_endpoint_auth_methods_supported": ["none"],
|
||||
"scopes_supported": ["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"],
|
||||
"service_documentation": f"{BASE_URL}/mcp/"
|
||||
})
|
||||
|
||||
|
||||
@router.get("/.well-known/oauth-protected-resource")
|
||||
async def get_protected_resource_metadata():
|
||||
"""OAuth Protected Resource Metadata (RFC 9728)"""
|
||||
return JSONResponse({
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [BASE_URL],
|
||||
"bearer_methods_supported": ["header"],
|
||||
"scopes_supported": ["mcp:tools:read", "mcp:tools:write"],
|
||||
"resource_documentation": f"{BASE_URL}/docs"
|
||||
})
|
||||
|
||||
|
||||
@router.get("/authorize")
|
||||
async def authorize_endpoint(
|
||||
response_type: str = Query(...),
|
||||
client_id: str = Query(...),
|
||||
redirect_uri: str = Query(...),
|
||||
code_challenge: str = Query(...),
|
||||
code_challenge_method: str = Query("S256"),
|
||||
state: Optional[str] = Query(None),
|
||||
scope: Optional[str] = Query(None)
|
||||
):
|
||||
"""OAuth 2.1 Authorization Endpoint - Uses Clerk SDK for custom domains"""
|
||||
|
||||
logger.info(f"OAuth authorize request - client_id: {client_id}, redirect_uri: {redirect_uri}")
|
||||
|
||||
if not CLERK_AVAILABLE:
|
||||
logger.error("Clerk SDK not available")
|
||||
raise HTTPException(status_code=500, detail="Clerk SDK not available")
|
||||
|
||||
# Store OAuth session for later validation
|
||||
try:
|
||||
from mcp_server_main import app as mcp_app
|
||||
from mcp_auth_factory import get_oauth_provider
|
||||
|
||||
oauth_provider = get_oauth_provider(mcp_app)
|
||||
if not oauth_provider:
|
||||
raise HTTPException(status_code=500, detail="OAuth provider not configured")
|
||||
|
||||
# Generate session and store PKCE
|
||||
session_id = secrets.token_urlsafe(32)
|
||||
if state is None:
|
||||
state = secrets.token_urlsafe(16)
|
||||
|
||||
# Create PKCE challenge
|
||||
from mcp_auth.oauth import PKCEChallenge
|
||||
pkce = PKCEChallenge()
|
||||
|
||||
# Store session data
|
||||
session_data = {
|
||||
"pkce_verifier": pkce.verifier,
|
||||
"pkce_challenge": code_challenge, # Store the client's challenge
|
||||
"state": state,
|
||||
"redirect_uri": redirect_uri,
|
||||
"client_id": client_id,
|
||||
"scopes": scope.split(" ") if scope else ["mcp:tools:read", "mcp:tools:write"],
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=10)).timestamp(),
|
||||
}
|
||||
oauth_provider.storage.set_session(session_id, session_data)
|
||||
|
||||
# For Clerk with custom domains, we need to use their hosted sign-in page
|
||||
# We'll pass our callback URL and session info in the state
|
||||
callback_url = f"{BASE_URL}/auth/callback"
|
||||
|
||||
# Encode session info in state for retrieval after Clerk auth
|
||||
combined_state = f"{state}:{session_id}"
|
||||
|
||||
# Use Clerk's sign-in URL with proper parameters
|
||||
clerk_domain = os.getenv("CLERK_DOMAIN", "accounts.yargimcp.com")
|
||||
sign_in_params = {
|
||||
"redirect_url": f"{callback_url}?state={quote(combined_state)}",
|
||||
}
|
||||
|
||||
sign_in_url = f"https://{clerk_domain}/sign-in?{urlencode(sign_in_params)}"
|
||||
|
||||
logger.info(f"Redirecting to Clerk sign-in: {sign_in_url}")
|
||||
|
||||
return RedirectResponse(url=sign_in_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Authorization failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
@router.get("/auth/callback")
|
||||
async def oauth_callback(
|
||||
request: Request,
|
||||
state: Optional[str] = Query(None)
|
||||
):
|
||||
"""Handle OAuth callback from Clerk - simplified for custom domains"""
|
||||
|
||||
logger.info(f"OAuth callback received - state: {state}")
|
||||
logger.info(f"Query params: {dict(request.query_params)}")
|
||||
logger.info(f"Cookies: {dict(request.cookies)}")
|
||||
|
||||
# For Clerk custom domains, we'll assume authentication succeeded
|
||||
# if Clerk redirected the user to our callback URL
|
||||
|
||||
# For custom domains, we'll skip complex session verification
|
||||
# and rely on the fact that Clerk only redirects here after successful auth
|
||||
|
||||
try:
|
||||
if not state:
|
||||
logger.error("No state parameter provided")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Missing state parameter"}
|
||||
)
|
||||
|
||||
# Parse state to get original state and session ID
|
||||
try:
|
||||
if ":" in state:
|
||||
original_state, session_id = state.rsplit(":", 1)
|
||||
else:
|
||||
original_state = state
|
||||
session_id = state # Fallback
|
||||
except ValueError:
|
||||
logger.error(f"Invalid state format: {state}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Invalid state format"}
|
||||
)
|
||||
|
||||
# Get OAuth provider
|
||||
from mcp_server_main import app as mcp_app
|
||||
from mcp_auth_factory import get_oauth_provider
|
||||
|
||||
oauth_provider = get_oauth_provider(mcp_app)
|
||||
if not oauth_provider:
|
||||
raise HTTPException(status_code=500, detail="OAuth provider not configured")
|
||||
|
||||
# Get stored session
|
||||
oauth_session = oauth_provider.storage.get_session(session_id)
|
||||
|
||||
if not oauth_session:
|
||||
logger.error(f"OAuth session not found for ID: {session_id}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "OAuth session expired or not found"}
|
||||
)
|
||||
|
||||
# Generate simple authorization code for custom domain flow
|
||||
auth_code = f"clerk_custom_{session_id}_{int(time.time())}"
|
||||
|
||||
# Store the code mapping for token exchange
|
||||
code_data = {
|
||||
"session_id": session_id,
|
||||
"clerk_authenticated": True,
|
||||
"custom_domain_flow": True,
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=5)).timestamp(),
|
||||
}
|
||||
oauth_provider.storage.set_session(f"code_{auth_code}", code_data)
|
||||
|
||||
# Build redirect URL back to Claude
|
||||
redirect_params = {
|
||||
"code": auth_code,
|
||||
"state": original_state
|
||||
}
|
||||
|
||||
redirect_url = f"{oauth_session['redirect_uri']}?{urlencode(redirect_params)}"
|
||||
logger.info(f"Redirecting back to Claude: {redirect_url}")
|
||||
|
||||
return RedirectResponse(url=redirect_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Callback processing failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
|
||||
|
||||
@router.post("/register")
|
||||
async def register_client(request: Request):
|
||||
"""Dynamic Client Registration (RFC 7591)"""
|
||||
|
||||
data = await request.json()
|
||||
logger.info(f"Client registration request: {data}")
|
||||
|
||||
# Simple dynamic registration - accept any client
|
||||
client_id = f"mcp-client-{os.urandom(8).hex()}"
|
||||
|
||||
return JSONResponse({
|
||||
"client_id": client_id,
|
||||
"client_secret": None, # Public client
|
||||
"redirect_uris": data.get("redirect_uris", []),
|
||||
"grant_types": ["authorization_code", "refresh_token"],
|
||||
"response_types": ["code"],
|
||||
"client_name": data.get("client_name", "MCP Client"),
|
||||
"token_endpoint_auth_method": "none",
|
||||
"client_id_issued_at": int(datetime.now().timestamp())
|
||||
})
|
||||
|
||||
|
||||
@router.post("/token")
|
||||
async def token_endpoint(request: Request):
|
||||
"""OAuth 2.1 Token Endpoint"""
|
||||
|
||||
# Parse form data
|
||||
form_data = await request.form()
|
||||
grant_type = form_data.get("grant_type")
|
||||
code = form_data.get("code")
|
||||
redirect_uri = form_data.get("redirect_uri")
|
||||
client_id = form_data.get("client_id")
|
||||
code_verifier = form_data.get("code_verifier")
|
||||
|
||||
logger.info(f"Token exchange - grant_type: {grant_type}, code: {code[:20] if code else 'None'}...")
|
||||
|
||||
if grant_type != "authorization_code":
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "unsupported_grant_type"}
|
||||
)
|
||||
|
||||
try:
|
||||
# Import here to avoid circular imports
|
||||
from mcp_server_main import app as mcp_app
|
||||
from mcp_auth_factory import get_oauth_provider
|
||||
|
||||
# Get OAuth provider
|
||||
oauth_provider = get_oauth_provider(mcp_app)
|
||||
if not oauth_provider:
|
||||
raise HTTPException(status_code=500, detail="OAuth provider not configured")
|
||||
|
||||
# Extract session info from code
|
||||
code_session = None
|
||||
if code.startswith("clerk_"):
|
||||
# Get the code mapping
|
||||
code_session = oauth_provider.storage.get_session(f"code_{code}")
|
||||
if code_session:
|
||||
session_id = code_session.get("session_id")
|
||||
else:
|
||||
logger.error(f"Code mapping not found for: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid authorization code"}
|
||||
)
|
||||
else:
|
||||
session_id = code
|
||||
|
||||
session = oauth_provider.storage.get_session(session_id)
|
||||
|
||||
if not session:
|
||||
logger.error(f"Session {session_id} not found for token exchange")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid authorization code"}
|
||||
)
|
||||
|
||||
# Validate PKCE if present
|
||||
if "pkce_challenge" in session and code_verifier:
|
||||
# Validate PKCE challenge
|
||||
if not oauth_provider.validate_pkce(code_verifier, session["pkce_challenge"]):
|
||||
logger.error("PKCE challenge validation failed")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid code verifier"}
|
||||
)
|
||||
logger.info("PKCE validation successful")
|
||||
else:
|
||||
logger.info("No PKCE validation required")
|
||||
|
||||
# Create JWT token
|
||||
access_token = oauth_provider._create_mcp_token(
|
||||
session["scopes"],
|
||||
session.get("clerk_token", ""),
|
||||
session_id
|
||||
)
|
||||
|
||||
# Clean up sessions
|
||||
oauth_provider.storage.delete_session(session_id)
|
||||
if code_session:
|
||||
oauth_provider.storage.delete_session(f"code_{code}")
|
||||
|
||||
return JSONResponse({
|
||||
"access_token": access_token,
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": " ".join(session["scopes"])
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Token exchange failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
+1966
-1935
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,57 @@
|
||||
"""
|
||||
Migration stub for the deprecated Yargı MCP endpoint.
|
||||
|
||||
Exposes a single tool that informs the MCP client the server has moved
|
||||
and the user must update their configuration.
|
||||
|
||||
Entrypoint variable `app` is a FastMCP instance so it works with
|
||||
Dokploy's FastMCP build pipeline (`fastmcp inspect`, `fastmcp run`).
|
||||
|
||||
Run with:
|
||||
fastmcp run migration_app.py:app --transport http --port 8000
|
||||
"""
|
||||
|
||||
from starlette.responses import JSONResponse
|
||||
from fastmcp import FastMCP
|
||||
|
||||
NEW_URL = "https://yargimcp.surucu.dev/mcp"
|
||||
|
||||
app = FastMCP(
|
||||
name="Yargı MCP (taşındı / moved)",
|
||||
instructions=(
|
||||
f"Bu Yargı MCP endpoint'i kullanımdan kaldırıldı. "
|
||||
f"Sunucu yeni adrese taşındı: {NEW_URL}. "
|
||||
f"Kullanıcıya MCP istemci ayarlarındaki sunucu URL'sini güncellemesini söyle."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@app.tool(
|
||||
description=(
|
||||
"DEPRECATED ENDPOINT — Yargı MCP sunucusu yeni adrese taşındı. "
|
||||
"Bu endpoint'teki eski araçlar (Yargıtay, Danıştay, Anayasa Mahkemesi, "
|
||||
"KVKK, BDDK, Rekabet Kurumu, Sayıştay, KİK, Emsal, Uyuşmazlık, GİB, "
|
||||
"Sigorta Tahkim, Bedesten vb.) artık burada çalışmıyor. "
|
||||
"Bu aracı çağırarak yeni sunucu adresini ve migrasyon talimatını al, "
|
||||
"ardından kullanıcıya MCP istemci (Claude Desktop, 5ire, ChatGPT vb.) "
|
||||
"ayarlarındaki sunucu URL'sini güncellemesi gerektiğini açıkça söyle."
|
||||
)
|
||||
)
|
||||
def migration_notice() -> dict:
|
||||
"""Yargı MCP sunucusunun yeni adresini ve migrasyon talimatını döner."""
|
||||
return {
|
||||
"status": "moved",
|
||||
"new_url": NEW_URL,
|
||||
"message": (
|
||||
f"Yargı MCP sunucusu yeni adrese taşındı: {NEW_URL}\n\n"
|
||||
f"Lütfen MCP istemcinin (Claude Desktop, 5ire, ChatGPT vb.) "
|
||||
f"ayarlarındaki sunucu URL'sini yukarıdaki yeni adresle güncelleyin. "
|
||||
f"Mevcut endpoint artık kullanım dışıdır ve sadece bu uyarıyı döner."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@app.custom_route("/health", methods=["GET"])
|
||||
async def health(request):
|
||||
"""Health check endpoint for monitoring services."""
|
||||
return JSONResponse({"status": "deprecated", "new_url": NEW_URL})
|
||||
-94
@@ -1,94 +0,0 @@
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
upstream yargi_mcp {
|
||||
server yargi-mcp:8000;
|
||||
}
|
||||
|
||||
# Rate limiting
|
||||
limit_req_zone $binary_remote_addr zone=api_limit:10m rate=10r/s;
|
||||
limit_req_zone $binary_remote_addr zone=mcp_limit:10m rate=100r/s;
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name localhost;
|
||||
|
||||
# Redirect HTTP to HTTPS in production
|
||||
# return 301 https://$server_name$request_uri;
|
||||
|
||||
# Security headers
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header X-Frame-Options DENY;
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header Referrer-Policy "strict-origin-when-cross-origin";
|
||||
|
||||
# API endpoints
|
||||
location /api/ {
|
||||
limit_req zone=api_limit burst=20 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts
|
||||
proxy_connect_timeout 60s;
|
||||
proxy_send_timeout 60s;
|
||||
proxy_read_timeout 60s;
|
||||
}
|
||||
|
||||
# MCP endpoint (higher rate limit)
|
||||
location /mcp-server/mcp/ {
|
||||
limit_req zone=mcp_limit burst=50 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# WebSocket support
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
|
||||
# Longer timeouts for MCP operations
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
}
|
||||
|
||||
# Health check (no rate limit)
|
||||
location /health {
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
# Root and other paths
|
||||
location / {
|
||||
limit_req zone=api_limit burst=10 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
}
|
||||
|
||||
# SSL configuration (uncomment for production)
|
||||
# server {
|
||||
# listen 443 ssl http2;
|
||||
# server_name your-domain.com;
|
||||
#
|
||||
# ssl_certificate /etc/nginx/ssl/cert.pem;
|
||||
# ssl_certificate_key /etc/nginx/ssl/key.pem;
|
||||
# ssl_protocols TLSv1.2 TLSv1.3;
|
||||
# ssl_ciphers HIGH:!aNULL:!MD5;
|
||||
#
|
||||
# # Include all location blocks from above
|
||||
# }
|
||||
}
|
||||
+9
-12
@@ -1,12 +1,12 @@
|
||||
[project]
|
||||
name = "yargi-mcp"
|
||||
version = "0.1.1"
|
||||
version = "0.2.2"
|
||||
description = "MCP Server For Turkish Legal Databases"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
license = {text = "MIT"}
|
||||
authors = [{name = "Said Surucu", email = "saidsrc@gmail.com"}]
|
||||
keywords = ["mcp", "turkish-law", "legal", "yargitay", "danistay", "turkish", "law", "court", "decisions"]
|
||||
keywords = ["mcp", "turkish-law", "legal", "yargitay", "danistay", "bddk", "btk", "kvkk", "turkish", "law", "court", "decisions"]
|
||||
classifiers = [
|
||||
"Development Status :: 4 - Beta",
|
||||
"Intended Audience :: Legal Industry",
|
||||
@@ -14,7 +14,7 @@ classifiers = [
|
||||
"License :: OSI Approved :: MIT License",
|
||||
"Programming Language :: Python :: 3.11",
|
||||
"Programming Language :: Python :: 3.12",
|
||||
"Topic :: Legal",
|
||||
"Topic :: Software Development :: Libraries :: Python Modules",
|
||||
"Topic :: Text Processing :: Markup :: Markdown",
|
||||
"Operating System :: OS Independent",
|
||||
]
|
||||
@@ -25,11 +25,12 @@ dependencies = [
|
||||
"markitdown[pdf]>=0.1.1",
|
||||
"pydantic>=2.11.4",
|
||||
"aiohttp>=3.11.18",
|
||||
"playwright>=1.52.0",
|
||||
"fastmcp>=2.9.2",
|
||||
"fastmcp>=2.10.5",
|
||||
"pypdf>=5.5.0",
|
||||
"fastapi>=0.115.14",
|
||||
"PyJWT>=2.8.0",
|
||||
"cryptography>=44.0.0",
|
||||
"openai>=1.0.0",
|
||||
"numpy>=1.24.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
@@ -45,19 +46,15 @@ production = [
|
||||
"gunicorn>=22.0.0",
|
||||
"uvicorn[standard]>=0.30.0",
|
||||
]
|
||||
saas = [
|
||||
"clerk-backend-api>=3.0.0",
|
||||
"stripe>=9.1.0",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
yargi-mcp = "mcp_server_main:main"
|
||||
|
||||
[tool.setuptools]
|
||||
py-modules = ["mcp_server_main", "mcp_auth_factory", "mcp_auth_http_adapter", "asgi_app", "fastapi_app", "starlette_app", "run_asgi", "stripe_webhook"]
|
||||
py-modules = ["mcp_server_main", "asgi_app"]
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["*_mcp_module", "mcp_auth"]
|
||||
include = ["*_mcp_module", "semantic_search"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=65.0", "wheel"]
|
||||
|
||||
@@ -0,0 +1,464 @@
|
||||
"""
|
||||
Redis Session Store for OAuth Authorization Codes and User Sessions
|
||||
|
||||
This module provides Redis-based storage for OAuth authorization codes and user sessions,
|
||||
enabling multi-machine deployment support by replacing in-memory storage.
|
||||
|
||||
Uses Upstash Redis via REST API for serverless-friendly operation.
|
||||
"""
|
||||
|
||||
import os
|
||||
import json
|
||||
import time
|
||||
import logging
|
||||
from typing import Optional, Dict, Any, Union
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from upstash_redis import Redis
|
||||
UPSTASH_AVAILABLE = True
|
||||
except ImportError:
|
||||
UPSTASH_AVAILABLE = False
|
||||
Redis = None
|
||||
|
||||
# Use standard Python exceptions for Redis connection errors
|
||||
import socket
|
||||
from requests.exceptions import ConnectionError as RequestsConnectionError, Timeout as RequestsTimeout
|
||||
|
||||
class RedisSessionStore:
|
||||
"""
|
||||
Redis-based session store for OAuth flows and user sessions.
|
||||
|
||||
Uses Upstash Redis REST API for connection-free operation suitable for
|
||||
multi-instance deployments on platforms like Fly.io.
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize Redis connection using environment variables."""
|
||||
if not UPSTASH_AVAILABLE:
|
||||
raise ImportError("upstash-redis package is required. Install with: pip install upstash-redis")
|
||||
|
||||
# Initialize Upstash Redis client from environment with optimized connection settings
|
||||
try:
|
||||
# Get Upstash Redis configuration
|
||||
redis_url = os.getenv("UPSTASH_REDIS_REST_URL")
|
||||
redis_token = os.getenv("UPSTASH_REDIS_REST_TOKEN")
|
||||
|
||||
if not redis_url or not redis_token:
|
||||
raise ValueError("UPSTASH_REDIS_REST_URL and UPSTASH_REDIS_REST_TOKEN must be set")
|
||||
|
||||
logger.info(f"Connecting to Upstash Redis at {redis_url[:30]}...")
|
||||
|
||||
# Initialize with explicit configuration for better SSL handling
|
||||
self.redis = Redis(
|
||||
url=redis_url,
|
||||
token=redis_token
|
||||
)
|
||||
|
||||
logger.info("Upstash Redis client created")
|
||||
|
||||
# Skip connection test during initialization to prevent server hang
|
||||
# Connection will be tested during first actual operation
|
||||
logger.info("Redis client initialized - connection will be tested on first use")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to initialize Upstash Redis: {e}")
|
||||
raise
|
||||
|
||||
# TTL values (in seconds)
|
||||
self.oauth_code_ttl = int(os.getenv("OAUTH_CODE_TTL", "600")) # 10 minutes
|
||||
self.session_ttl = int(os.getenv("SESSION_TTL", "3600")) # 1 hour
|
||||
|
||||
def _serialize_data(self, data: Dict[str, Any]) -> Dict[str, str]:
|
||||
"""Convert data to Redis-compatible string format."""
|
||||
serialized = {}
|
||||
for key, value in data.items():
|
||||
if isinstance(value, (dict, list)):
|
||||
serialized[key] = json.dumps(value)
|
||||
elif isinstance(value, (int, float)):
|
||||
serialized[key] = str(value)
|
||||
elif isinstance(value, bool):
|
||||
serialized[key] = "true" if value else "false"
|
||||
else:
|
||||
serialized[key] = str(value)
|
||||
return serialized
|
||||
|
||||
def _deserialize_data(self, data: Dict[str, str]) -> Dict[str, Any]:
|
||||
"""Convert Redis string data back to original types."""
|
||||
if not data:
|
||||
return {}
|
||||
|
||||
deserialized = {}
|
||||
for key, value in data.items():
|
||||
if not isinstance(value, str):
|
||||
deserialized[key] = value
|
||||
continue
|
||||
|
||||
# Try to deserialize JSON
|
||||
if value.startswith(('[', '{')):
|
||||
try:
|
||||
deserialized[key] = json.loads(value)
|
||||
continue
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
# Try to convert numbers
|
||||
if value.isdigit():
|
||||
deserialized[key] = int(value)
|
||||
continue
|
||||
|
||||
if value.replace('.', '').isdigit():
|
||||
try:
|
||||
deserialized[key] = float(value)
|
||||
continue
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
# Handle booleans
|
||||
if value in ("true", "false"):
|
||||
deserialized[key] = value == "true"
|
||||
continue
|
||||
|
||||
# Keep as string
|
||||
deserialized[key] = value
|
||||
|
||||
return deserialized
|
||||
|
||||
# OAuth Authorization Code Methods
|
||||
|
||||
def set_oauth_code(self, code: str, data: Dict[str, Any]) -> bool:
|
||||
"""
|
||||
Store OAuth authorization code with automatic expiration.
|
||||
|
||||
Args:
|
||||
code: Authorization code string
|
||||
data: Code data including user_id, client_id, etc.
|
||||
|
||||
Returns:
|
||||
True if stored successfully, False otherwise
|
||||
"""
|
||||
try:
|
||||
key = f"oauth:code:{code}"
|
||||
|
||||
# Add timestamp for debugging
|
||||
data_with_timestamp = data.copy()
|
||||
data_with_timestamp.update({
|
||||
"created_at": time.time(),
|
||||
"expires_at": time.time() + self.oauth_code_ttl
|
||||
})
|
||||
|
||||
# Serialize and store - Upstash Redis doesn't support mapping parameter
|
||||
serialized_data = self._serialize_data(data_with_timestamp)
|
||||
|
||||
# Use individual hset calls for each field with retry logic
|
||||
max_retries = 3
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
# Clear any existing data first
|
||||
self.redis.delete(key)
|
||||
|
||||
# Set all fields in a pipeline-like manner
|
||||
for field, value in serialized_data.items():
|
||||
self.redis.hset(key, field, value)
|
||||
|
||||
# Set expiration
|
||||
self.redis.expire(key, self.oauth_code_ttl)
|
||||
|
||||
logger.info(f"Stored OAuth code {code[:10]}... with TTL {self.oauth_code_ttl}s (attempt {attempt + 1})")
|
||||
return True
|
||||
|
||||
except (RequestsConnectionError, RequestsTimeout, OSError, socket.error) as e:
|
||||
logger.warning(f"Redis connection error on attempt {attempt + 1}: {e}")
|
||||
if attempt == max_retries - 1:
|
||||
raise # Re-raise on final attempt
|
||||
time.sleep(0.5 * (attempt + 1)) # Exponential backoff
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to store OAuth code {code[:10]}... after {max_retries} attempts: {e}")
|
||||
return False
|
||||
|
||||
def get_oauth_code(self, code: str, delete_after_use: bool = True) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Retrieve OAuth authorization code data.
|
||||
|
||||
Args:
|
||||
code: Authorization code string
|
||||
delete_after_use: If True, delete the code after retrieval (one-time use)
|
||||
|
||||
Returns:
|
||||
Code data dictionary or None if not found/expired
|
||||
"""
|
||||
max_retries = 3
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
key = f"oauth:code:{code}"
|
||||
|
||||
# Get all hash fields with retry
|
||||
data = self.redis.hgetall(key)
|
||||
|
||||
if not data:
|
||||
logger.warning(f"OAuth code {code[:10]}... not found or expired (attempt {attempt + 1})")
|
||||
return None
|
||||
|
||||
# Deserialize data
|
||||
deserialized_data = self._deserialize_data(data)
|
||||
|
||||
# Check manual expiration (in case Redis TTL failed)
|
||||
expires_at = deserialized_data.get("expires_at", 0)
|
||||
if expires_at and time.time() > expires_at:
|
||||
logger.warning(f"OAuth code {code[:10]}... manually expired")
|
||||
try:
|
||||
self.redis.delete(key)
|
||||
except Exception as del_error:
|
||||
logger.warning(f"Failed to delete expired code: {del_error}")
|
||||
return None
|
||||
|
||||
# Delete after use for security (one-time use)
|
||||
if delete_after_use:
|
||||
try:
|
||||
self.redis.delete(key)
|
||||
logger.info(f"Retrieved and deleted OAuth code {code[:10]}... (attempt {attempt + 1})")
|
||||
except Exception as del_error:
|
||||
logger.warning(f"Failed to delete code after use: {del_error}")
|
||||
# Continue anyway since we got the data
|
||||
else:
|
||||
logger.info(f"Retrieved OAuth code {code[:10]}... (not deleted, attempt {attempt + 1})")
|
||||
|
||||
return deserialized_data
|
||||
|
||||
except (RequestsConnectionError, RequestsTimeout, OSError, socket.error) as e:
|
||||
logger.warning(f"Redis connection error on retrieval attempt {attempt + 1}: {e}")
|
||||
if attempt == max_retries - 1:
|
||||
logger.error(f"Failed to retrieve OAuth code {code[:10]}... after {max_retries} attempts: {e}")
|
||||
return None
|
||||
time.sleep(0.5 * (attempt + 1)) # Exponential backoff
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to retrieve OAuth code {code[:10]}... on attempt {attempt + 1}: {e}")
|
||||
if attempt == max_retries - 1:
|
||||
return None
|
||||
time.sleep(0.5 * (attempt + 1))
|
||||
|
||||
return None
|
||||
|
||||
# User Session Methods
|
||||
|
||||
def set_session(self, session_id: str, user_data: Dict[str, Any]) -> bool:
|
||||
"""
|
||||
Store user session data with sliding expiration.
|
||||
|
||||
Args:
|
||||
session_id: Unique session identifier
|
||||
user_data: User session data (user_id, email, scopes, etc.)
|
||||
|
||||
Returns:
|
||||
True if stored successfully, False otherwise
|
||||
"""
|
||||
try:
|
||||
key = f"session:{session_id}"
|
||||
|
||||
# Add session metadata
|
||||
session_data = user_data.copy()
|
||||
session_data.update({
|
||||
"session_id": session_id,
|
||||
"created_at": time.time(),
|
||||
"last_accessed": time.time()
|
||||
})
|
||||
|
||||
# Serialize and store - Upstash Redis doesn't support mapping parameter
|
||||
serialized_data = self._serialize_data(session_data)
|
||||
|
||||
# Use individual hset calls for each field (Upstash compatibility)
|
||||
for field, value in serialized_data.items():
|
||||
self.redis.hset(key, field, value)
|
||||
self.redis.expire(key, self.session_ttl)
|
||||
|
||||
logger.info(f"Stored session {session_id[:10]}... with TTL {self.session_ttl}s")
|
||||
return True
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to store session {session_id[:10]}...: {e}")
|
||||
return False
|
||||
|
||||
def get_session(self, session_id: str, refresh_ttl: bool = True) -> Optional[Dict[str, Any]]:
|
||||
"""
|
||||
Retrieve user session data.
|
||||
|
||||
Args:
|
||||
session_id: Session identifier
|
||||
refresh_ttl: If True, extend session TTL on access
|
||||
|
||||
Returns:
|
||||
Session data dictionary or None if not found/expired
|
||||
"""
|
||||
try:
|
||||
key = f"session:{session_id}"
|
||||
|
||||
# Get session data
|
||||
data = self.redis.hgetall(key)
|
||||
|
||||
if not data:
|
||||
logger.warning(f"Session {session_id[:10]}... not found or expired")
|
||||
return None
|
||||
|
||||
# Deserialize data
|
||||
session_data = self._deserialize_data(data)
|
||||
|
||||
# Update last accessed time and refresh TTL
|
||||
if refresh_ttl:
|
||||
session_data["last_accessed"] = time.time()
|
||||
self.redis.hset(key, "last_accessed", str(time.time()))
|
||||
self.redis.expire(key, self.session_ttl)
|
||||
logger.debug(f"Refreshed session {session_id[:10]}... TTL")
|
||||
|
||||
return session_data
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to retrieve session {session_id[:10]}...: {e}")
|
||||
return None
|
||||
|
||||
def delete_session(self, session_id: str) -> bool:
|
||||
"""
|
||||
Delete user session (logout).
|
||||
|
||||
Args:
|
||||
session_id: Session identifier
|
||||
|
||||
Returns:
|
||||
True if deleted successfully, False otherwise
|
||||
"""
|
||||
try:
|
||||
key = f"session:{session_id}"
|
||||
result = self.redis.delete(key)
|
||||
|
||||
if result:
|
||||
logger.info(f"Deleted session {session_id[:10]}...")
|
||||
return True
|
||||
else:
|
||||
logger.warning(f"Session {session_id[:10]}... not found for deletion")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to delete session {session_id[:10]}...: {e}")
|
||||
return False
|
||||
|
||||
# Health Check Methods
|
||||
|
||||
def health_check(self) -> Dict[str, Any]:
|
||||
"""
|
||||
Perform Redis health check.
|
||||
|
||||
Returns:
|
||||
Health status dictionary
|
||||
"""
|
||||
try:
|
||||
# Test basic operations
|
||||
test_key = f"health:check:{int(time.time())}"
|
||||
test_value = {"timestamp": time.time(), "test": True}
|
||||
|
||||
# Test set - Use individual hset calls for Upstash compatibility
|
||||
serialized_test = self._serialize_data(test_value)
|
||||
for field, value in serialized_test.items():
|
||||
self.redis.hset(test_key, field, value)
|
||||
|
||||
# Test get
|
||||
retrieved = self.redis.hgetall(test_key)
|
||||
|
||||
# Test delete
|
||||
self.redis.delete(test_key)
|
||||
|
||||
return {
|
||||
"status": "healthy",
|
||||
"redis_connected": True,
|
||||
"operations_working": bool(retrieved),
|
||||
"timestamp": datetime.utcnow().isoformat()
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Redis health check failed: {e}")
|
||||
return {
|
||||
"status": "unhealthy",
|
||||
"redis_connected": False,
|
||||
"error": str(e),
|
||||
"timestamp": datetime.utcnow().isoformat()
|
||||
}
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""
|
||||
Get Redis usage statistics.
|
||||
|
||||
Returns:
|
||||
Statistics dictionary
|
||||
"""
|
||||
try:
|
||||
# Get basic info (not all Upstash plans support INFO command)
|
||||
stats = {
|
||||
"oauth_codes_pattern": "oauth:code:*",
|
||||
"sessions_pattern": "session:*",
|
||||
"timestamp": datetime.utcnow().isoformat()
|
||||
}
|
||||
|
||||
try:
|
||||
# Try to get counts (may fail on some Upstash plans)
|
||||
oauth_keys = self.redis.keys("oauth:code:*")
|
||||
session_keys = self.redis.keys("session:*")
|
||||
|
||||
stats.update({
|
||||
"active_oauth_codes": len(oauth_keys) if oauth_keys else 0,
|
||||
"active_sessions": len(session_keys) if session_keys else 0
|
||||
})
|
||||
except Exception as e:
|
||||
logger.warning(f"Could not get detailed stats: {e}")
|
||||
stats["warning"] = "Detailed stats not available on this Redis plan"
|
||||
|
||||
return stats
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to get Redis stats: {e}")
|
||||
return {"error": str(e), "timestamp": datetime.utcnow().isoformat()}
|
||||
|
||||
# Global instance for easy importing
|
||||
redis_store = None
|
||||
|
||||
def get_redis_store() -> Optional[RedisSessionStore]:
|
||||
"""
|
||||
Get global Redis store instance (singleton pattern).
|
||||
|
||||
Returns:
|
||||
RedisSessionStore instance or None if initialization fails
|
||||
"""
|
||||
global redis_store
|
||||
|
||||
if redis_store is None:
|
||||
try:
|
||||
logger.info("Initializing Redis store...")
|
||||
redis_store = RedisSessionStore()
|
||||
logger.info("Redis store initialized successfully")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to initialize Redis store: {e}")
|
||||
redis_store = None
|
||||
|
||||
return redis_store
|
||||
|
||||
def init_redis_store() -> RedisSessionStore:
|
||||
"""
|
||||
Initialize Redis store and perform health check.
|
||||
|
||||
Returns:
|
||||
RedisSessionStore instance
|
||||
|
||||
Raises:
|
||||
Exception if Redis is not available or unhealthy
|
||||
"""
|
||||
store = get_redis_store()
|
||||
|
||||
# Perform health check
|
||||
health = store.health_check()
|
||||
|
||||
if health["status"] != "healthy":
|
||||
raise Exception(f"Redis health check failed: {health}")
|
||||
|
||||
logger.info("Redis session store initialized and healthy")
|
||||
return store
|
||||
@@ -1,5 +1,6 @@
|
||||
# rekabet_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple, Dict, Any
|
||||
@@ -141,12 +142,12 @@ class RekabetKurumuApiClient:
|
||||
|
||||
# Row 1: Publication Date, Decision Number, Related Cases Link
|
||||
td_elements_r1 = rows[0].find_all("td")
|
||||
pub_date = td_elements_r1[0].get_text(strip=True) if len(td_elements_r1) > 0 else None
|
||||
dec_num = td_elements_r1[1].get_text(strip=True) if len(td_elements_r1) > 1 else None
|
||||
pub_date = td_elements_r1[0].get_text(strip=True) if len(td_elements_r1) > 0 else ""
|
||||
dec_num = td_elements_r1[1].get_text(strip=True) if len(td_elements_r1) > 1 else ""
|
||||
|
||||
related_cases_link_tag = td_elements_r1[2].find("a", href=True) if len(td_elements_r1) > 2 else None
|
||||
related_cases_url_str: Optional[str] = None
|
||||
karar_id_from_related: Optional[str] = None
|
||||
related_cases_url_str: str = ""
|
||||
karar_id_from_related: str = ""
|
||||
if related_cases_link_tag and related_cases_link_tag.has_attr('href'):
|
||||
related_cases_url_str = urljoin(self.BASE_URL, related_cases_link_tag['href'])
|
||||
qs_related = parse_qs(urlparse(related_cases_link_tag['href']).query)
|
||||
@@ -155,16 +156,16 @@ class RekabetKurumuApiClient:
|
||||
|
||||
# Row 2: Decision Date, Decision Type
|
||||
td_elements_r2 = rows[1].find_all("td")
|
||||
dec_date = td_elements_r2[0].get_text(strip=True) if len(td_elements_r2) > 0 else None
|
||||
dec_type_text = td_elements_r2[1].get_text(strip=True) if len(td_elements_r2) > 1 else None
|
||||
dec_date = td_elements_r2[0].get_text(strip=True) if len(td_elements_r2) > 0 else ""
|
||||
dec_type_text = td_elements_r2[1].get_text(strip=True) if len(td_elements_r2) > 1 else ""
|
||||
|
||||
# Row 3: Title and Main Decision Link
|
||||
title_cell = rows[2].find("td", colspan="5")
|
||||
decision_link_tag = title_cell.find("a", href=True) if title_cell else None
|
||||
|
||||
title_text: Optional[str] = None
|
||||
decision_landing_url_str: Optional[str] = None
|
||||
karar_id_from_main_link: Optional[str] = None
|
||||
title_text: str = ""
|
||||
decision_landing_url_str: str = ""
|
||||
karar_id_from_main_link: str = ""
|
||||
|
||||
if decision_link_tag and decision_link_tag.has_attr('href'):
|
||||
title_text = decision_link_tag.get_text(strip=True)
|
||||
@@ -185,16 +186,12 @@ class RekabetKurumuApiClient:
|
||||
logger.warning(f"Table {idx+1} Karar ID not found. Skipping. Title (if any): {title_text}")
|
||||
continue
|
||||
|
||||
# Convert string URLs to HttpUrl for the model
|
||||
final_decision_url = HttpUrl(decision_landing_url_str) if decision_landing_url_str else None
|
||||
final_related_cases_url = HttpUrl(related_cases_url_str) if related_cases_url_str else None
|
||||
|
||||
processed_decisions.append(RekabetDecisionSummary(
|
||||
publication_date=pub_date, decision_number=dec_num, decision_date=dec_date,
|
||||
decision_type_text=dec_type_text, title=title_text,
|
||||
decision_url=final_decision_url,
|
||||
decision_url=decision_landing_url_str,
|
||||
karar_id=current_karar_id,
|
||||
related_cases_url=final_related_cases_url
|
||||
related_cases_url=related_cases_url_str
|
||||
))
|
||||
logger.debug(f"Table {idx+1} parsed successfully: Karar ID '{current_karar_id}', Title '{title_text[:50] if title_text else 'N/A'}...'")
|
||||
|
||||
@@ -357,7 +354,7 @@ class RekabetKurumuApiClient:
|
||||
total_pdf_pages = total_pdf_pages_from_extraction
|
||||
|
||||
if single_page_pdf_bytes:
|
||||
markdown_for_requested_page = self._convert_pdf_bytes_to_markdown(single_page_pdf_bytes, str(pdf_url_to_report or full_landing_page_url))
|
||||
markdown_for_requested_page = await asyncio.to_thread(self._convert_pdf_bytes_to_markdown, single_page_pdf_bytes, str(pdf_url_to_report or full_landing_page_url))
|
||||
if not markdown_for_requested_page:
|
||||
error_message = (error_message or "") + f"; Could not convert page {page_number} of PDF to Markdown."
|
||||
elif total_pdf_pages > 0 :
|
||||
|
||||
@@ -25,35 +25,31 @@ class RekabetKararTuruAdiEnum(str, Enum):
|
||||
|
||||
class RekabetKurumuSearchRequest(BaseModel):
|
||||
"""Model for Rekabet Kurumu (Turkish Competition Authority) search request."""
|
||||
sayfaAdi: Optional[str] = Field(None, description="Search in decision title (Başlık).")
|
||||
YayinlanmaTarihi: Optional[str] = Field(None, description="Publication date (Yayım Tarihi), e.g., DD.MM.YYYY.")
|
||||
PdfText: Optional[str] = Field(
|
||||
None,
|
||||
description='Search in decision text (Metin). For an exact phrase match, enclose the phrase in double quotes (e.g., "\\"vertical agreement\\" competition). The website indicates that using "" provides more precise results for phrases.'
|
||||
)
|
||||
# This field uses the GUID enum as it's used by the client to make the actual web request.
|
||||
KararTuruID: Optional[RekabetKararTuruGuidEnum] = Field(RekabetKararTuruGuidEnum.TUMU, description="Decision type (Karar Türü) GUID for internal client use, corresponding to the website's values.")
|
||||
KararSayisi: Optional[str] = Field(None, description="Decision number (Karar Sayısı).")
|
||||
KararTarihi: Optional[str] = Field(None, description="Decision date (Karar Tarihi), e.g., DD.MM.YYYY.")
|
||||
page: int = Field(1, ge=1, description="Page number to fetch for results list.")
|
||||
sayfaAdi: str = Field("", description="Title")
|
||||
YayinlanmaTarihi: str = Field("", description="Date")
|
||||
PdfText: str = Field("", description="Text")
|
||||
KararTuruID: RekabetKararTuruGuidEnum = Field(RekabetKararTuruGuidEnum.TUMU, description="Type")
|
||||
KararSayisi: str = Field("", description="No")
|
||||
KararTarihi: str = Field("", description="Date")
|
||||
page: int = Field(1, ge=1, description="Page")
|
||||
|
||||
class RekabetDecisionSummary(BaseModel):
|
||||
"""Model for a single Rekabet Kurumu decision summary from search results."""
|
||||
publication_date: Optional[str] = Field(None, description="Publication Date (Yayımlanma Tarihi).")
|
||||
decision_number: Optional[str] = Field(None, description="Decision Number (Karar Sayısı).")
|
||||
decision_date: Optional[str] = Field(None, description="Decision Date (Karar Tarihi).")
|
||||
decision_type_text: Optional[str] = Field(None, description="Decision Type as text (Karar Türü - metin olarak).")
|
||||
title: Optional[str] = Field(None, description="Decision title or summary text.")
|
||||
decision_url: Optional[HttpUrl] = Field(None, description="URL to the decision's landing page (e.g., /Karar?kararId=...).")
|
||||
karar_id: Optional[str] = Field(None, description="GUID of the decision, extracted from its URL.")
|
||||
related_cases_url: Optional[HttpUrl] = Field(None, description="URL to related court cases page, if available.")
|
||||
publication_date: str = Field("", description="Pub date")
|
||||
decision_number: str = Field("", description="Number")
|
||||
decision_date: str = Field("", description="Date")
|
||||
decision_type_text: str = Field("", description="Type")
|
||||
title: str = Field("", description="Title")
|
||||
decision_url: str = Field("", description="URL")
|
||||
karar_id: str = Field("", description="ID")
|
||||
related_cases_url: str = Field("", description="Cases URL")
|
||||
|
||||
class RekabetSearchResult(BaseModel):
|
||||
"""Model for the overall search result for Rekabet Kurumu decisions."""
|
||||
decisions: List[RekabetDecisionSummary]
|
||||
total_records_found: Optional[int] = Field(None, description="Total number of records found matching the query.")
|
||||
retrieved_page_number: int = Field(description="The page number of the results that were retrieved.")
|
||||
total_pages: Optional[int] = Field(None, description="Total number of pages available for the query.")
|
||||
total_records_found: int = Field(0, description="Total")
|
||||
retrieved_page_number: int = Field(description="Page")
|
||||
total_pages: int = Field(0, description="Pages")
|
||||
|
||||
class RekabetDocument(BaseModel):
|
||||
"""
|
||||
@@ -61,16 +57,15 @@ class RekabetDocument(BaseModel):
|
||||
Contains metadata from the landing page, a link to the PDF,
|
||||
and the PDF's content converted to paginated Markdown.
|
||||
"""
|
||||
source_landing_page_url: HttpUrl = Field(description="The URL of the decision's landing page from which the PDF was identified.")
|
||||
karar_id: str = Field(description="GUID of the decision.")
|
||||
source_landing_page_url: HttpUrl = Field(description="Source URL")
|
||||
karar_id: str = Field(description="ID")
|
||||
|
||||
title_on_landing_page: Optional[str] = Field(None, description="Title as found on the landing page (e.g., from <title> tag or a main heading). Could be a generic title if direct PDF.")
|
||||
pdf_url: Optional[HttpUrl] = Field(None, description="Direct URL to the decision PDF document, if successfully found and resolved.")
|
||||
title_on_landing_page: Optional[str] = Field(None, description="Title")
|
||||
pdf_url: Optional[HttpUrl] = Field(None, description="PDF URL")
|
||||
|
||||
# Fields for Markdown content derived from the PDF
|
||||
markdown_chunk: Optional[str] = Field(None, description="A 5,000 character chunk of the Markdown content derived from the decision PDF.")
|
||||
current_page: int = Field(1, description="The current page number of the PDF-derived markdown chunk (1-indexed).")
|
||||
total_pages: int = Field(1, description="Total number of pages for the full PDF-derived markdown content. Will be 0 if content could not be processed.")
|
||||
is_paginated: bool = Field(False, description="True if the full PDF-derived markdown content is split into multiple pages.")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Content")
|
||||
current_page: int = Field(1, description="Page")
|
||||
total_pages: int = Field(1, description="Total pages")
|
||||
is_paginated: bool = Field(False, description="Paginated")
|
||||
|
||||
error_message: Optional[str] = Field(None, description="Contains an error message if the document retrieval or processing failed at any stage.")
|
||||
error_message: Optional[str] = Field(None, description="Error")
|
||||
@@ -1,11 +0,0 @@
|
||||
fastmcp
|
||||
httpx
|
||||
beautifulsoup4
|
||||
markitdown[pdf]
|
||||
pydantic
|
||||
aiohttp
|
||||
playwright
|
||||
pypdf
|
||||
fastapi>=0.115.14
|
||||
uvicorn[standard]>=0.30.0
|
||||
starlette>=0.37.0
|
||||
-119
@@ -1,119 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Standalone ASGI server runner for Yargı MCP
|
||||
|
||||
This script provides a simple way to run the Yargı MCP server
|
||||
as a web service using uvicorn.
|
||||
|
||||
Usage:
|
||||
python run_asgi.py
|
||||
python run_asgi.py --host 0.0.0.0 --port 8080
|
||||
python run_asgi.py --reload # For development
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
# Add project root to Python path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
try:
|
||||
import uvicorn
|
||||
except ImportError:
|
||||
print("Error: uvicorn is not installed.")
|
||||
print("Please install it with: pip install uvicorn")
|
||||
sys.exit(1)
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run Yargı MCP server as an ASGI web service"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--host",
|
||||
type=str,
|
||||
default=os.getenv("HOST", "127.0.0.1"),
|
||||
help="Host to bind to (default: 127.0.0.1)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--port",
|
||||
type=int,
|
||||
default=int(os.getenv("PORT", "8000")),
|
||||
help="Port to bind to (default: 8000)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--reload",
|
||||
action="store_true",
|
||||
help="Enable auto-reload for development"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--transport",
|
||||
choices=["http", "sse"],
|
||||
default="http",
|
||||
help="Transport type (default: http)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--log-level",
|
||||
choices=["debug", "info", "warning", "error"],
|
||||
default=os.getenv("LOG_LEVEL", "info").lower(),
|
||||
help="Log level (default: info)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--workers",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of worker processes (default: 1)"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Select app based on transport
|
||||
app_name = "asgi_app:app" if args.transport == "http" else "asgi_app:sse_app"
|
||||
|
||||
# Configure uvicorn
|
||||
config = {
|
||||
"app": app_name,
|
||||
"host": args.host,
|
||||
"port": args.port,
|
||||
"log_level": args.log_level,
|
||||
"reload": args.reload,
|
||||
"access_log": True,
|
||||
}
|
||||
|
||||
# Add workers only if not in reload mode
|
||||
if not args.reload and args.workers > 1:
|
||||
config["workers"] = args.workers
|
||||
|
||||
# Print startup information
|
||||
print(f"Starting Yargı MCP server...")
|
||||
print(f"Host: {args.host}")
|
||||
print(f"Port: {args.port}")
|
||||
print(f"Transport: {args.transport}")
|
||||
print(f"Log level: {args.log_level}")
|
||||
if args.reload:
|
||||
print("Auto-reload: enabled")
|
||||
else:
|
||||
print(f"Workers: {args.workers}")
|
||||
print(f"\nServer will be available at: http://{args.host}:{args.port}")
|
||||
print(f"MCP endpoint: http://{args.host}:{args.port}/mcp/")
|
||||
print(f"Health check: http://{args.host}:{args.port}/health")
|
||||
print(f"API status: http://{args.host}:{args.port}/status")
|
||||
print("\nPress CTRL+C to stop the server\n")
|
||||
|
||||
# Run uvicorn
|
||||
try:
|
||||
uvicorn.run(**config)
|
||||
except KeyboardInterrupt:
|
||||
print("\nShutting down server...")
|
||||
sys.exit(0)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,13 +1,13 @@
|
||||
# sayistay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
import re
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
import logging
|
||||
import html
|
||||
import tempfile
|
||||
import os
|
||||
import io
|
||||
from urllib.parse import urlencode, urljoin
|
||||
from markitdown import MarkItDown
|
||||
|
||||
@@ -17,7 +17,7 @@ from .models import (
|
||||
DaireSearchRequest, DaireSearchResponse, DaireDecision,
|
||||
SayistayDocumentMarkdown
|
||||
)
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum, WEB_KARAR_KONUSU_MAPPING
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
@@ -49,6 +49,12 @@ class SayistayApiClient:
|
||||
TEMYIZ_KURULU_ENDPOINT = "/KararlarTemyiz/DataTablesList"
|
||||
DAIRE_ENDPOINT = "/KararlarDaire/DataTablesList"
|
||||
|
||||
# Marker present in the upstream WAF block page (also returns HTTP 418).
|
||||
# Verified 2026-05-03 against real Chrome — the block targets POSTs to
|
||||
# the DataTablesList endpoints regardless of headers/cookies/CSRF, so
|
||||
# we surface a specific error instead of the generic "I'm a teapot".
|
||||
_WAF_BLOCK_MARKER = "Bilgi Güvenliği Politikaları Gereği Kısıtlanmıştır"
|
||||
|
||||
# Page endpoints for session initialization and document access
|
||||
GENEL_KURUL_PAGE = "/KararlarGenelKurul"
|
||||
TEMYIZ_KURULU_PAGE = "/KararlarTemyiz"
|
||||
@@ -135,8 +141,30 @@ class SayistayApiClient:
|
||||
return "Tüm Kurumlar"
|
||||
elif enum_type == "web_karar_konusu":
|
||||
return "Tüm Konular"
|
||||
|
||||
# Apply web_karar_konusu mapping
|
||||
if enum_type == "web_karar_konusu":
|
||||
return WEB_KARAR_KONUSU_MAPPING.get(enum_value, enum_value)
|
||||
|
||||
return enum_value
|
||||
|
||||
def _raise_if_waf_blocked(self, response: httpx.Response, endpoint_label: str) -> None:
|
||||
"""
|
||||
Sayıştay's upstream WAF returns HTTP 418 with a Turkish HTML block
|
||||
page for POSTs to the DataTablesList endpoints. This affects every
|
||||
client (verified with real Chrome on 2026-05-03), so there is no
|
||||
client-side workaround. Detect it and raise a clear error.
|
||||
"""
|
||||
if response.status_code == 418 or self._WAF_BLOCK_MARKER in response.text:
|
||||
raise RuntimeError(
|
||||
f"Sayıştay upstream WAF blocked the {endpoint_label} request "
|
||||
f"(HTTP {response.status_code} from {response.request.url}). "
|
||||
"This is a server-side restriction at sayistay.gov.tr — affects "
|
||||
"all clients including a real browser — and cannot be worked "
|
||||
"around from yargi-mcp. Try again later or contact Sayıştay if "
|
||||
"the block persists."
|
||||
)
|
||||
|
||||
def _build_datatables_params(self, start: int, length: int, draw: int = 1) -> List[Tuple[str, str]]:
|
||||
"""Build standard DataTables parameters for all endpoints."""
|
||||
params = [
|
||||
@@ -380,6 +408,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Genel Kurul")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -439,6 +468,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Temyiz Kurulu")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -498,6 +528,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Daire")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -532,21 +563,18 @@ class SayistayApiClient:
|
||||
raise
|
||||
|
||||
def _convert_html_to_markdown(self, html_content: str) -> Optional[str]:
|
||||
"""Convert HTML content to Markdown using MarkItDown."""
|
||||
"""Convert HTML content to Markdown using MarkItDown with BytesIO to avoid filename length issues."""
|
||||
if not html_content:
|
||||
return None
|
||||
|
||||
temp_file_path = None
|
||||
try:
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
|
||||
# Write HTML to temp file
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp:
|
||||
tmp.write(html_content)
|
||||
temp_file_path = tmp.name
|
||||
|
||||
# Convert
|
||||
result = md_converter.convert(temp_file_path)
|
||||
result = md_converter.convert(html_stream)
|
||||
markdown_content = result.text_content
|
||||
|
||||
logger.info("Successfully converted HTML to Markdown")
|
||||
@@ -555,9 +583,6 @@ class SayistayApiClient:
|
||||
except Exception as e:
|
||||
logger.error(f"Error converting HTML to Markdown: {e}")
|
||||
return f"Error converting HTML content: {str(e)}"
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
|
||||
async def get_document_as_markdown(self, decision_id: str, decision_type: str) -> SayistayDocumentMarkdown:
|
||||
"""
|
||||
@@ -633,7 +658,7 @@ class SayistayApiClient:
|
||||
)
|
||||
|
||||
# Convert HTML to Markdown using existing method
|
||||
markdown_content = self._convert_html_to_markdown(html_content)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, html_content)
|
||||
|
||||
if markdown_content and "Error converting HTML content" not in markdown_content:
|
||||
logger.info(f"Successfully retrieved and converted document {decision_id} to Markdown")
|
||||
|
||||
@@ -28,18 +28,30 @@ KamuIdaresiTuruEnum = Literal[
|
||||
"Diğer" # Other
|
||||
]
|
||||
|
||||
# Decision Subject Categories (Web Karar Konusu)
|
||||
# Decision Subject Categories (Web Karar Konusu) - Shortened for token efficiency
|
||||
WebKararKonusuEnum = Literal[
|
||||
"ALL", # All subjects
|
||||
"Harcırah Mevzuatı ile İlgili Kararlar", # Travel Allowance Legislation Related Decisions
|
||||
"İhale Mevzuatı ile İlgili Kararlar", # Procurement Legislation Related Decisions
|
||||
"İş Mevzuatı ile İlgili Kararlar", # Labor Legislation Related Decisions
|
||||
"Personel Mevzuatı ile İlgili Kararlar", # Personnel Legislation Related Decisions
|
||||
"Sorumluluk ve Yargılama Usulleri ile İlgili Kararlar", # Liability and Trial Procedures Related Decisions
|
||||
"Vergi Resmi Harç ve Diğer Gelirlerle İlgili Kararlar", # Tax, Official Fee and Other Revenue Related Decisions
|
||||
"Çeşitli Konuları İlgilendiren Kararlar" # Decisions Concerning Various Topics
|
||||
"Harcırah Mevzuatı", # Travel Allowance Legislation
|
||||
"İhale Mevzuatı", # Procurement Legislation
|
||||
"İş Mevzuatı", # Labor Legislation
|
||||
"Personel Mevzuatı", # Personnel Legislation
|
||||
"Sorumluluk ve Yargılama Usulleri", # Liability and Trial Procedures
|
||||
"Vergi Resmi Harç ve Diğer Gelirler", # Tax, Official Fee and Other Revenue
|
||||
"Çeşitli Konular" # Various Topics
|
||||
]
|
||||
|
||||
# Mapping from shortened enum values to full API values
|
||||
WEB_KARAR_KONUSU_MAPPING = {
|
||||
"ALL": "ALL",
|
||||
"Harcırah Mevzuatı": "Harcırah Mevzuatı ile İlgili Kararlar",
|
||||
"İhale Mevzuatı": "İhale Mevzuatı ile İlgili Kararlar",
|
||||
"İş Mevzuatı": "İş Mevzuatı ile İlgili Kararlar",
|
||||
"Personel Mevzuatı": "Personel Mevzuatı ile İlgili Kararlar",
|
||||
"Sorumluluk ve Yargılama Usulleri": "Sorumluluk ve Yargılama Usulleri ile İlgili Kararlar",
|
||||
"Vergi Resmi Harç ve Diğer Gelirler": "Vergi Resmi Harç ve Diğer Gelirlerle İlgili Kararlar",
|
||||
"Çeşitli Konular": "Çeşitli Konuları İlgilendiren Kararlar"
|
||||
}
|
||||
|
||||
# Year ranges for different endpoints
|
||||
GENEL_KURUL_YEARS = [str(year) for year in range(2006, 2025)] # 2006-2024
|
||||
TEMYIZ_KURULU_YEARS = [str(year) for year in range(1993, 2023)] # 1993-2022
|
||||
|
||||
+89
-111
@@ -1,9 +1,16 @@
|
||||
# sayistay_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import Optional, List, Union
|
||||
from typing import Optional, List, Union, Dict, Any, Literal
|
||||
from enum import Enum
|
||||
from .enums import DaireEnum, KamuIdaresiTuruEnum, WebKararKonusuEnum
|
||||
|
||||
# --- Unified Enums ---
|
||||
class SayistayDecisionTypeEnum(str, Enum):
|
||||
GENEL_KURUL = "genel_kurul"
|
||||
TEMYIZ_KURULU = "temyiz_kurulu"
|
||||
DAIRE = "daire"
|
||||
|
||||
# ============================================================================
|
||||
# Genel Kurul (General Assembly) Models
|
||||
# ============================================================================
|
||||
@@ -16,30 +23,18 @@ class GenelKurulSearchRequest(BaseModel):
|
||||
of the Turkish Court of Accounts, typically addressing interpretation of
|
||||
audit and accountability regulations.
|
||||
"""
|
||||
karar_no: Optional[str] = Field(None, description="Decision number (e.g., '5415')")
|
||||
karar_ek: Optional[str] = Field(None, description="Decision appendix number (max 99)")
|
||||
karar_no: str = Field("", description="Decision no")
|
||||
karar_ek: str = Field("", description="Appendix no")
|
||||
|
||||
karar_tarih_baslangic: Optional[str] = Field(None, description="""
|
||||
Decision start year for date range filtering.
|
||||
Available years: 2006-2024. Format: 'YYYY' (e.g., '2020')
|
||||
Use with karar_tarih_bitis for date range filtering.
|
||||
""")
|
||||
karar_tarih_baslangic: str = Field("", description="Start year (YYYY)")
|
||||
|
||||
karar_tarih_bitis: Optional[str] = Field(None, description="""
|
||||
Decision end year for date range filtering.
|
||||
Available years: 2006-2024. Format: 'YYYY' (e.g., '2024')
|
||||
Use with karar_tarih_baslangic for date range filtering.
|
||||
""")
|
||||
karar_tarih_bitis: str = Field("", description="End year")
|
||||
|
||||
karar_tamami: Optional[str] = Field(None, description="""
|
||||
Content/text search within decision summaries (max 400 characters).
|
||||
Searches in decision abstracts and main content.
|
||||
Example: 'belediye taşınmaz tahsis'
|
||||
""")
|
||||
karar_tamami: str = Field("", description="Value")
|
||||
|
||||
# DataTables pagination
|
||||
start: int = Field(0, description="Starting record for pagination (0-based)")
|
||||
length: int = Field(10, description="Number of records per page (1-100)")
|
||||
length: int = Field(10, description="Number of records per page (1-10)")
|
||||
|
||||
class GenelKurulDecision(BaseModel):
|
||||
"""Single Genel Kurul decision entry from search results."""
|
||||
@@ -66,62 +61,27 @@ class TemyizKuruluSearchRequest(BaseModel):
|
||||
Temyiz Kurulu reviews appeals against audit chamber decisions,
|
||||
providing higher-level review of audit findings and sanctions.
|
||||
"""
|
||||
ilam_dairesi: DaireEnum = Field("ALL", description="""
|
||||
Chamber/Department filter for appeals board decisions.
|
||||
• ALL: All chambers (default)
|
||||
• 1-8: Specific chamber number (1. Daire through 8. Daire)
|
||||
Each chamber specializes in different types of public institutions.
|
||||
""")
|
||||
ilam_dairesi: DaireEnum = Field("ALL", description="Value")
|
||||
|
||||
yili: Optional[str] = Field(None, description="""
|
||||
Account year filter (Hesap Yılı).
|
||||
Available years: 1993-2022. Format: 'YYYY' (e.g., '2020')
|
||||
Refers to the fiscal year being audited, not decision date.
|
||||
""")
|
||||
yili: str = Field("", description="Value")
|
||||
|
||||
karar_tarih_baslangic: Optional[str] = Field(None, description="""
|
||||
Decision start year for date range filtering.
|
||||
Available years: 2000, 2006-2024. Format: 'YYYY' (e.g., '2020')
|
||||
Use with karar_tarih_bitis for date range filtering.
|
||||
""")
|
||||
karar_tarih_baslangic: str = Field("", description="Value")
|
||||
|
||||
karar_tarih_bitis: Optional[str] = Field(None, description="""
|
||||
Decision end year for date range filtering.
|
||||
Available years: 2000, 2006-2024. Format: 'YYYY' (e.g., '2024')
|
||||
Use with karar_tarih_baslangic for date range filtering.
|
||||
""")
|
||||
karar_tarih_bitis: str = Field("", description="End year")
|
||||
|
||||
kamu_idaresi_turu: KamuIdaresiTuruEnum = Field("ALL", description="""
|
||||
Public administration type filter:
|
||||
• ALL: All institutions (default)
|
||||
• Genel Bütçe Kapsamındaki İdareler: General budget administrations
|
||||
• Yüksek Öğretim Kurumları: Higher education institutions
|
||||
• Belediyeler ve Bağlı İdareler: Municipalities and affiliates
|
||||
• Other specific institution types
|
||||
""")
|
||||
kamu_idaresi_turu: KamuIdaresiTuruEnum = Field("ALL", description="Value")
|
||||
|
||||
ilam_no: Optional[str] = Field(None, description="Audit report number (İlam No, max 50 chars)")
|
||||
dosya_no: Optional[str] = Field(None, description="File number for the case")
|
||||
temyiz_tutanak_no: Optional[str] = Field(None, description="Appeals board meeting minutes number")
|
||||
ilam_no: str = Field("", description="Audit report number (İlam No, max 50 chars)")
|
||||
dosya_no: str = Field("", description="File number for the case")
|
||||
temyiz_tutanak_no: str = Field("", description="Appeals board meeting minutes number")
|
||||
|
||||
temyiz_karar: Optional[str] = Field(None, description="""
|
||||
Content search within appeals decisions.
|
||||
Searches decision text and reasoning.
|
||||
Example: 'araç kiralama kasko'
|
||||
""")
|
||||
temyiz_karar: str = Field("", description="Value")
|
||||
|
||||
web_karar_konusu: WebKararKonusuEnum = Field("ALL", description="""
|
||||
Decision subject category filter:
|
||||
• ALL: All subjects (default)
|
||||
• İhale Mevzuatı ile İlgili Kararlar: Procurement legislation
|
||||
• Personel Mevzuatı ile İlgili Kararlar: Personnel legislation
|
||||
• Harcırah Mevzuatı ile İlgili Kararlar: Travel allowance legislation
|
||||
• Other specialized legal areas
|
||||
""")
|
||||
web_karar_konusu: WebKararKonusuEnum = Field("ALL", description="Value")
|
||||
|
||||
# DataTables pagination
|
||||
start: int = Field(0, description="Starting record for pagination (0-based)")
|
||||
length: int = Field(10, description="Number of records per page (1-100)")
|
||||
length: int = Field(10, description="Number of records per page (1-10)")
|
||||
|
||||
class TemyizKuruluDecision(BaseModel):
|
||||
"""Single Temyiz Kurulu decision entry from search results."""
|
||||
@@ -148,60 +108,25 @@ class DaireSearchRequest(BaseModel):
|
||||
Daire decisions are first-instance audit findings and sanctions
|
||||
issued by individual audit chambers before potential appeals.
|
||||
"""
|
||||
yargilama_dairesi: DaireEnum = Field("ALL", description="""
|
||||
Audit chamber filter:
|
||||
• ALL: All chambers (default)
|
||||
• 1-8: Specific chamber number (1. Daire through 8. Daire)
|
||||
Each chamber audits different types of public institutions.
|
||||
""")
|
||||
yargilama_dairesi: DaireEnum = Field("ALL", description="Value")
|
||||
|
||||
karar_tarih_baslangic: Optional[str] = Field(None, description="""
|
||||
Decision start year for date range filtering.
|
||||
Available years: 2012-2025. Format: 'YYYY' (e.g., '2020')
|
||||
Use with karar_tarih_bitis for date range filtering.
|
||||
""")
|
||||
karar_tarih_baslangic: str = Field("", description="Value")
|
||||
|
||||
karar_tarih_bitis: Optional[str] = Field(None, description="""
|
||||
Decision end year for date range filtering.
|
||||
Available years: 2012-2025. Format: 'YYYY' (e.g., '2024')
|
||||
Use with karar_tarih_baslangic for date range filtering.
|
||||
""")
|
||||
karar_tarih_bitis: str = Field("", description="End year")
|
||||
|
||||
ilam_no: Optional[str] = Field(None, description="Audit report number (İlam No, max 50 chars)")
|
||||
ilam_no: str = Field("", description="Audit report number (İlam No, max 50 chars)")
|
||||
|
||||
kamu_idaresi_turu: KamuIdaresiTuruEnum = Field("ALL", description="""
|
||||
Public administration type filter:
|
||||
• ALL: All institutions (default)
|
||||
• Genel Bütçe Kapsamındaki İdareler: General budget administrations
|
||||
• Yüksek Öğretim Kurumları: Higher education institutions
|
||||
• Belediyeler ve Bağlı İdareler: Municipalities and affiliates
|
||||
• Other specific institution types
|
||||
""")
|
||||
kamu_idaresi_turu: KamuIdaresiTuruEnum = Field("ALL", description="Value")
|
||||
|
||||
hesap_yili: Optional[str] = Field(None, description="""
|
||||
Account year filter (Hesap Yılı).
|
||||
Available years: 2005, 2008-2023. Format: 'YYYY' (e.g., '2020')
|
||||
Refers to the fiscal year being audited, not decision date.
|
||||
""")
|
||||
hesap_yili: str = Field("", description="Value")
|
||||
|
||||
web_karar_konusu: WebKararKonusuEnum = Field("ALL", description="""
|
||||
Decision subject category filter:
|
||||
• ALL: All subjects (default)
|
||||
• İhale Mevzuatı ile İlgili Kararlar: Procurement legislation
|
||||
• Personel Mevzuatı ile İlgili Kararlar: Personnel legislation
|
||||
• Vergi Resmi Harç ve Diğer Gelirlerle İlgili Kararlar: Tax and fee legislation
|
||||
• Other specialized legal areas
|
||||
""")
|
||||
web_karar_konusu: WebKararKonusuEnum = Field("ALL", description="Value")
|
||||
|
||||
web_karar_metni: Optional[str] = Field(None, description="""
|
||||
Content search within chamber decisions.
|
||||
Searches decision text and audit findings.
|
||||
Example: 'birim fiyat revize edilmemesi'
|
||||
""")
|
||||
web_karar_metni: str = Field("", description="Value")
|
||||
|
||||
# DataTables pagination
|
||||
start: int = Field(0, description="Starting record for pagination (0-based)")
|
||||
length: int = Field(10, description="Number of records per page (1-100)")
|
||||
length: int = Field(10, description="Number of records per page (1-10)")
|
||||
|
||||
class DaireDecision(BaseModel):
|
||||
"""Single Daire decision entry from search results."""
|
||||
@@ -209,7 +134,7 @@ class DaireDecision(BaseModel):
|
||||
yargilama_dairesi: int = Field(..., description="Chamber number (1-8)")
|
||||
karar_tarih: str = Field(..., description="Decision date in DD.MM.YYYY format")
|
||||
karar_no: str = Field(..., description="Decision number")
|
||||
ilam_no: Optional[str] = Field(None, description="Audit report number (may be null)")
|
||||
ilam_no: str = Field("", description="Audit report number (may be null)")
|
||||
madde_no: int = Field(..., description="Article/item number within the decision")
|
||||
kamu_idaresi_turu: str = Field(..., description="Public administration type")
|
||||
hesap_yili: int = Field(..., description="Account year being audited")
|
||||
@@ -235,8 +160,61 @@ class SayistayDocumentMarkdown(BaseModel):
|
||||
decision types (Genel Kurul, Temyiz Kurulu, Daire).
|
||||
"""
|
||||
decision_id: str = Field(..., description="Unique decision identifier")
|
||||
decision_type: str = Field(..., description="Type of decision: 'genel_kurul', 'temyiz_kurulu', or 'daire'")
|
||||
decision_type: str = Field(..., description="Value")
|
||||
source_url: str = Field(..., description="Original URL where the document was retrieved")
|
||||
markdown_content: Optional[str] = Field(None, description="Full decision text converted to Markdown format")
|
||||
retrieval_date: Optional[str] = Field(None, description="Date when document was retrieved (ISO format)")
|
||||
error_message: Optional[str] = Field(None, description="Error message if document retrieval failed")
|
||||
|
||||
# ============================================================================
|
||||
# Unified Models
|
||||
# ============================================================================
|
||||
|
||||
class SayistayUnifiedSearchRequest(BaseModel):
|
||||
"""Unified search request for all Sayıştay decision types."""
|
||||
decision_type: Literal["genel_kurul", "temyiz_kurulu", "daire"] = Field(..., description="Decision type: genel_kurul, temyiz_kurulu, or daire")
|
||||
|
||||
# Common pagination parameters
|
||||
start: int = Field(0, ge=0, description="Starting record for pagination (0-based)")
|
||||
length: int = Field(10, ge=1, le=100, description="Number of records per page (1-100)")
|
||||
|
||||
# Common search parameters
|
||||
karar_tarih_baslangic: str = Field("", description="Start date (DD.MM.YYYY format)")
|
||||
karar_tarih_bitis: str = Field("", description="End date (DD.MM.YYYY format)")
|
||||
kamu_idaresi_turu: KamuIdaresiTuruEnum = Field("ALL", description="Public administration type filter")
|
||||
ilam_no: str = Field("", description="Audit report number (İlam No, max 50 chars)")
|
||||
web_karar_konusu: WebKararKonusuEnum = Field("ALL", description="Decision subject category filter")
|
||||
|
||||
# Genel Kurul specific parameters (ignored for other types)
|
||||
karar_no: str = Field("", description="Decision number (genel_kurul only)")
|
||||
karar_ek: str = Field("", description="Decision appendix number (genel_kurul only)")
|
||||
karar_tamami: str = Field("", description="Full text search (genel_kurul only)")
|
||||
|
||||
# Temyiz Kurulu specific parameters (ignored for other types)
|
||||
ilam_dairesi: DaireEnum = Field("ALL", description="Audit chamber selection (temyiz_kurulu only)")
|
||||
yili: str = Field("", description="Year (YYYY format, temyiz_kurulu only)")
|
||||
dosya_no: str = Field("", description="File number (temyiz_kurulu only)")
|
||||
temyiz_tutanak_no: str = Field("", description="Appeals board meeting minutes number (temyiz_kurulu only)")
|
||||
temyiz_karar: str = Field("", description="Appeals decision text search (temyiz_kurulu only)")
|
||||
|
||||
# Daire specific parameters (ignored for other types)
|
||||
yargilama_dairesi: DaireEnum = Field("ALL", description="Chamber selection (daire only)")
|
||||
hesap_yili: str = Field("", description="Account year (daire only)")
|
||||
web_karar_metni: str = Field("", description="Decision text search (daire only)")
|
||||
|
||||
class SayistayUnifiedSearchResult(BaseModel):
|
||||
"""Unified search result containing decisions from any Sayıştay decision type."""
|
||||
decision_type: Literal["genel_kurul", "temyiz_kurulu", "daire"] = Field(..., description="Type of decisions returned")
|
||||
decisions: List[Dict[str, Any]] = Field(default_factory=list, description="Decision list (structure varies by type)")
|
||||
total_records: int = Field(0, description="Total number of records found")
|
||||
total_filtered: int = Field(0, description="Number of records after filtering")
|
||||
draw: int = Field(1, description="DataTables draw counter")
|
||||
|
||||
class SayistayUnifiedDocumentMarkdown(BaseModel):
|
||||
"""Unified document model for all Sayıştay decision types."""
|
||||
decision_type: Literal["genel_kurul", "temyiz_kurulu", "daire"] = Field(..., description="Type of document")
|
||||
decision_id: str = Field(..., description="Decision ID")
|
||||
source_url: str = Field(..., description="Source URL of the document")
|
||||
document_data: Dict[str, Any] = Field(default_factory=dict, description="Document content and metadata")
|
||||
markdown_content: Optional[str] = Field(None, description="Markdown content")
|
||||
error_message: Optional[str] = Field(None, description="Error message if retrieval failed")
|
||||
@@ -0,0 +1,133 @@
|
||||
# sayistay_mcp_module/unified_client.py
|
||||
# Unified client for all three Sayıştay decision types
|
||||
|
||||
import logging
|
||||
from typing import Optional, Dict, Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from .models import (
|
||||
SayistayUnifiedSearchRequest,
|
||||
SayistayUnifiedSearchResult,
|
||||
SayistayUnifiedDocumentMarkdown,
|
||||
GenelKurulSearchRequest,
|
||||
TemyizKuruluSearchRequest,
|
||||
DaireSearchRequest
|
||||
)
|
||||
from .client import SayistayApiClient
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class SayistayUnifiedClient:
|
||||
"""Unified client that handles all three Sayıştay decision types."""
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.client = SayistayApiClient(request_timeout)
|
||||
|
||||
async def search_unified(self, params: SayistayUnifiedSearchRequest) -> SayistayUnifiedSearchResult:
|
||||
"""Unified search that routes to appropriate search method based on decision_type."""
|
||||
|
||||
if params.decision_type == "genel_kurul":
|
||||
# Convert to genel kurul request
|
||||
genel_kurul_params = GenelKurulSearchRequest(
|
||||
karar_no=params.karar_no,
|
||||
karar_ek=params.karar_ek,
|
||||
karar_tarih_baslangic=params.karar_tarih_baslangic,
|
||||
karar_tarih_bitis=params.karar_tarih_bitis,
|
||||
karar_tamami=params.karar_tamami,
|
||||
start=params.start,
|
||||
length=params.length
|
||||
)
|
||||
|
||||
result = await self.client.search_genel_kurul_decisions(genel_kurul_params)
|
||||
|
||||
# Convert to unified format
|
||||
decisions_list = [decision.model_dump() for decision in result.decisions]
|
||||
|
||||
return SayistayUnifiedSearchResult(
|
||||
decision_type="genel_kurul",
|
||||
decisions=decisions_list,
|
||||
total_records=result.total_records,
|
||||
total_filtered=result.total_filtered,
|
||||
draw=result.draw
|
||||
)
|
||||
|
||||
elif params.decision_type == "temyiz_kurulu":
|
||||
# Convert to temyiz kurulu request
|
||||
temyiz_params = TemyizKuruluSearchRequest(
|
||||
ilam_dairesi=params.ilam_dairesi,
|
||||
yili=params.yili,
|
||||
karar_tarih_baslangic=params.karar_tarih_baslangic,
|
||||
karar_tarih_bitis=params.karar_tarih_bitis,
|
||||
kamu_idaresi_turu=params.kamu_idaresi_turu,
|
||||
ilam_no=params.ilam_no,
|
||||
dosya_no=params.dosya_no,
|
||||
temyiz_tutanak_no=params.temyiz_tutanak_no,
|
||||
temyiz_karar=params.temyiz_karar,
|
||||
web_karar_konusu=params.web_karar_konusu,
|
||||
start=params.start,
|
||||
length=params.length
|
||||
)
|
||||
|
||||
result = await self.client.search_temyiz_kurulu_decisions(temyiz_params)
|
||||
|
||||
# Convert to unified format
|
||||
decisions_list = [decision.model_dump() for decision in result.decisions]
|
||||
|
||||
return SayistayUnifiedSearchResult(
|
||||
decision_type="temyiz_kurulu",
|
||||
decisions=decisions_list,
|
||||
total_records=result.total_records,
|
||||
total_filtered=result.total_filtered,
|
||||
draw=result.draw
|
||||
)
|
||||
|
||||
elif params.decision_type == "daire":
|
||||
# Convert to daire request
|
||||
daire_params = DaireSearchRequest(
|
||||
yargilama_dairesi=params.yargilama_dairesi,
|
||||
karar_tarih_baslangic=params.karar_tarih_baslangic,
|
||||
karar_tarih_bitis=params.karar_tarih_bitis,
|
||||
ilam_no=params.ilam_no,
|
||||
kamu_idaresi_turu=params.kamu_idaresi_turu,
|
||||
hesap_yili=params.hesap_yili,
|
||||
web_karar_konusu=params.web_karar_konusu,
|
||||
web_karar_metni=params.web_karar_metni,
|
||||
start=params.start,
|
||||
length=params.length
|
||||
)
|
||||
|
||||
result = await self.client.search_daire_decisions(daire_params)
|
||||
|
||||
# Convert to unified format
|
||||
decisions_list = [decision.model_dump() for decision in result.decisions]
|
||||
|
||||
return SayistayUnifiedSearchResult(
|
||||
decision_type="daire",
|
||||
decisions=decisions_list,
|
||||
total_records=result.total_records,
|
||||
total_filtered=result.total_filtered,
|
||||
draw=result.draw
|
||||
)
|
||||
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {params.decision_type}")
|
||||
|
||||
async def get_document_unified(self, decision_id: str, decision_type: str) -> SayistayUnifiedDocumentMarkdown:
|
||||
"""Unified document retrieval for all Sayıştay decision types."""
|
||||
|
||||
# Use existing client method (decision_type is already a string)
|
||||
result = await self.client.get_document_as_markdown(decision_id, decision_type)
|
||||
|
||||
return SayistayUnifiedDocumentMarkdown(
|
||||
decision_type=decision_type,
|
||||
decision_id=result.decision_id,
|
||||
source_url=result.source_url,
|
||||
document_data=result.model_dump(),
|
||||
markdown_content=result.markdown_content,
|
||||
error_message=result.error_message
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close the underlying client session."""
|
||||
if hasattr(self.client, 'close_client_session'):
|
||||
await self.client.close_client_session()
|
||||
@@ -0,0 +1,23 @@
|
||||
# semantic_search/__init__.py
|
||||
|
||||
from .embedder import (
|
||||
OpenRouterEmbedder,
|
||||
LocalEmbedder,
|
||||
get_embedder,
|
||||
is_openrouter_available,
|
||||
is_local_embedding_configured,
|
||||
is_semantic_search_available,
|
||||
)
|
||||
from .vector_store import VectorStore
|
||||
from .processor import DocumentProcessor
|
||||
|
||||
__all__ = [
|
||||
'OpenRouterEmbedder',
|
||||
'LocalEmbedder',
|
||||
'get_embedder',
|
||||
'is_openrouter_available',
|
||||
'is_local_embedding_configured',
|
||||
'is_semantic_search_available',
|
||||
'VectorStore',
|
||||
'DocumentProcessor',
|
||||
]
|
||||
@@ -0,0 +1,348 @@
|
||||
# semantic_search/embedder.py
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Dict, List, Optional
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# OpenRouter defaults (preserve backward compatibility)
|
||||
DEFAULT_MODEL = "google/gemini-embedding-001"
|
||||
DEFAULT_DIMENSION = 3072
|
||||
|
||||
# Local provider defaults — Ollama with nomic-embed-text out of the box.
|
||||
# Override via LOCAL_EMBEDDING_BASE_URL / LOCAL_EMBEDDING_MODEL /
|
||||
# LOCAL_EMBEDDING_DIMENSION when using a different server or model.
|
||||
# For Turkish, intfloat/multilingual-e5-large (1024 dims, prompt_style=e5)
|
||||
# served via HuggingFace TEI is the recommended setup — see README.
|
||||
LOCAL_DEFAULT_BASE_URL = "http://localhost:11434/v1"
|
||||
LOCAL_DEFAULT_MODEL = "nomic-embed-text"
|
||||
LOCAL_DEFAULT_DIMENSION = 768
|
||||
|
||||
# Prompt-template styles. Embedding models are trained with specific
|
||||
# prefixes — using the wrong style silently degrades retrieval quality.
|
||||
# - "gemini": "task: {task} | query: {text}" / "title: {title} | text: {text}"
|
||||
# (matches google/gemini-embedding-001, the OpenRouter default)
|
||||
# - "e5": "query: {text}" / "passage: {text}"
|
||||
# (matches intfloat/multilingual-e5-* models — best for Turkish)
|
||||
# - "raw": no prefix; pass text through as-is
|
||||
PROMPT_STYLES = ("gemini", "e5", "raw")
|
||||
DEFAULT_PROMPT_STYLE = "gemini"
|
||||
|
||||
|
||||
def _format_query(prompt_style: str, query: str, task: str) -> str:
|
||||
if prompt_style == "e5":
|
||||
return f"query: {query}"
|
||||
if prompt_style == "raw":
|
||||
return query
|
||||
# gemini (default)
|
||||
return f"task: {task} | query: {query}"
|
||||
|
||||
|
||||
def _format_document(prompt_style: str, doc: str, title: str) -> str:
|
||||
if prompt_style == "e5":
|
||||
return f"passage: {doc}"
|
||||
if prompt_style == "raw":
|
||||
return doc
|
||||
# gemini (default)
|
||||
return f"title: {title} | text: {doc}"
|
||||
|
||||
|
||||
def _resolve_prompt_style(explicit: Optional[str], default: str) -> str:
|
||||
style = (explicit or os.getenv("EMBEDDING_PROMPT_STYLE") or default).strip().lower()
|
||||
if style not in PROMPT_STYLES:
|
||||
raise ValueError(
|
||||
f"Unknown EMBEDDING_PROMPT_STYLE {style!r}; expected one of {PROMPT_STYLES}"
|
||||
)
|
||||
return style
|
||||
|
||||
|
||||
def is_openrouter_available() -> bool:
|
||||
"""Check if OpenRouter API key is available."""
|
||||
return bool(os.getenv("OPENROUTER_API_KEY"))
|
||||
|
||||
|
||||
def is_local_embedding_configured() -> bool:
|
||||
"""Check if the user opted into a local embedding endpoint."""
|
||||
return os.getenv("EMBEDDING_PROVIDER", "").strip().lower() == "local"
|
||||
|
||||
|
||||
def is_semantic_search_available() -> bool:
|
||||
"""Returns True if any embedding provider is configured."""
|
||||
return is_local_embedding_configured() or is_openrouter_available()
|
||||
|
||||
|
||||
def _coerce_dimension(value, env_name: str, default: int) -> int:
|
||||
"""Parse a dimension value (int or str) with clear error messages."""
|
||||
if value is None:
|
||||
return default
|
||||
try:
|
||||
parsed = int(value)
|
||||
except (TypeError, ValueError) as e:
|
||||
raise ValueError(
|
||||
f"{env_name} must be an integer, got {value!r}"
|
||||
) from e
|
||||
if parsed <= 0:
|
||||
raise ValueError(f"Embedding dimension must be positive, got {parsed}")
|
||||
return parsed
|
||||
|
||||
|
||||
class _BaseOpenAICompatibleEmbedder:
|
||||
"""
|
||||
Shared encode/similarity logic for embedders backed by the OpenAI Python
|
||||
SDK. Subclasses configure ``client``, ``model``, ``dimension``, and
|
||||
optionally ``_extra_headers`` (e.g. OpenRouter ranking headers).
|
||||
"""
|
||||
|
||||
# Subclasses may override; sent on every embeddings.create call when set.
|
||||
_extra_headers: Dict[str, str] = {}
|
||||
|
||||
# Set by subclasses
|
||||
client = None
|
||||
model: str = ""
|
||||
dimension: int = 0
|
||||
prompt_style: str = DEFAULT_PROMPT_STYLE
|
||||
|
||||
def encode_query(self, query: str, task: str = "search result") -> np.ndarray:
|
||||
"""
|
||||
Encode a search query. Prefix is selected by ``self.prompt_style``.
|
||||
|
||||
Args:
|
||||
query: The search query text
|
||||
task: Task hint used by the gemini-style prefix; ignored for
|
||||
e5/raw styles.
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (``self.dimension`` elements).
|
||||
"""
|
||||
text = _format_query(self.prompt_style, query, task)
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=text,
|
||||
encoding_format="float",
|
||||
extra_headers=self._extra_headers or None,
|
||||
)
|
||||
|
||||
embedding = np.array(response.data[0].embedding, dtype=np.float32)
|
||||
|
||||
# L2 normalize for cosine similarity
|
||||
norm = np.linalg.norm(embedding)
|
||||
if norm > 0:
|
||||
embedding = embedding / norm
|
||||
|
||||
logger.debug(f"Encoded query: {query[:50]}... -> shape: {embedding.shape}")
|
||||
return embedding
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode query: {e}")
|
||||
raise
|
||||
|
||||
def encode_documents(self, documents: List[str], titles: Optional[List[str]] = None) -> np.ndarray:
|
||||
"""
|
||||
Encode multiple documents with a batch API call.
|
||||
|
||||
Args:
|
||||
documents: List of document texts
|
||||
titles: Optional list of document titles
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (N x ``self.dimension``).
|
||||
"""
|
||||
if not documents:
|
||||
return np.array([])
|
||||
|
||||
texts = []
|
||||
for i, doc in enumerate(documents):
|
||||
title = titles[i] if titles and i < len(titles) else "none"
|
||||
texts.append(_format_document(self.prompt_style, doc, title))
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=texts,
|
||||
encoding_format="float",
|
||||
extra_headers=self._extra_headers or None,
|
||||
)
|
||||
|
||||
embeddings = np.array(
|
||||
[d.embedding for d in sorted(response.data, key=lambda x: x.index)],
|
||||
dtype=np.float32,
|
||||
)
|
||||
|
||||
# L2 normalize each embedding for cosine similarity
|
||||
norms = np.linalg.norm(embeddings, axis=1, keepdims=True)
|
||||
embeddings = embeddings / (norms + 1e-8)
|
||||
|
||||
logger.info(f"Encoded {len(documents)} documents -> shape: {embeddings.shape}")
|
||||
return embeddings
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode documents: {e}")
|
||||
raise
|
||||
|
||||
def compute_similarity(self, query_embedding: np.ndarray, document_embeddings: np.ndarray) -> np.ndarray:
|
||||
"""
|
||||
Compute cosine similarity between query and documents.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding (``self.dimension``,)
|
||||
document_embeddings: Document embeddings (N x ``self.dimension``)
|
||||
|
||||
Returns:
|
||||
Similarity scores (N,)
|
||||
"""
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Embeddings are already L2-normalized.
|
||||
similarities = np.dot(document_embeddings, query_embedding.T).squeeze()
|
||||
return similarities
|
||||
|
||||
|
||||
class OpenRouterEmbedder(_BaseOpenAICompatibleEmbedder):
|
||||
"""
|
||||
Embedder using OpenRouter's embedding API.
|
||||
|
||||
The model and dimension are configurable so users can pick any OpenRouter
|
||||
embedding model (e.g. when one becomes paid). Configuration precedence:
|
||||
explicit constructor args > environment variables > defaults.
|
||||
|
||||
Environment variables:
|
||||
OPENROUTER_API_KEY (required): OpenRouter credential
|
||||
OPENROUTER_EMBEDDING_MODEL (optional): override the embedding model id
|
||||
OPENROUTER_EMBEDDING_DIMENSION (optional): override the vector size
|
||||
|
||||
Defaults preserve backward compatibility: ``google/gemini-embedding-001``
|
||||
at 3072 dimensions.
|
||||
"""
|
||||
|
||||
_extra_headers = {
|
||||
"HTTP-Referer": "https://yargimcp.com",
|
||||
"X-Title": "Yargi MCP Server",
|
||||
}
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model: Optional[str] = None,
|
||||
dimension: Optional[int] = None,
|
||||
prompt_style: Optional[str] = None,
|
||||
):
|
||||
api_key = os.getenv("OPENROUTER_API_KEY")
|
||||
if not api_key:
|
||||
raise ValueError("OPENROUTER_API_KEY environment variable is not set")
|
||||
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError:
|
||||
raise ImportError("openai package is required. Install with: pip install openai")
|
||||
|
||||
self.client = OpenAI(
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
api_key=api_key,
|
||||
)
|
||||
self.model = model or os.getenv("OPENROUTER_EMBEDDING_MODEL") or DEFAULT_MODEL
|
||||
self.dimension = _coerce_dimension(
|
||||
dimension if dimension is not None else os.getenv("OPENROUTER_EMBEDDING_DIMENSION"),
|
||||
"OPENROUTER_EMBEDDING_DIMENSION",
|
||||
DEFAULT_DIMENSION,
|
||||
)
|
||||
# Default to gemini-style prefix for OpenRouter — matches the default
|
||||
# google/gemini-embedding-001 model. Override via constructor or
|
||||
# EMBEDDING_PROMPT_STYLE env var when picking a different model.
|
||||
self.prompt_style = _resolve_prompt_style(prompt_style, "gemini")
|
||||
|
||||
logger.info(
|
||||
f"OpenRouter Embedder initialized with model: {self.model} "
|
||||
f"(dimension={self.dimension}, prompt_style={self.prompt_style})"
|
||||
)
|
||||
|
||||
|
||||
class LocalEmbedder(_BaseOpenAICompatibleEmbedder):
|
||||
"""
|
||||
Embedder for a local OpenAI-compatible embedding server — Ollama,
|
||||
llama.cpp, vLLM, LM Studio, etc. Zero new Python dependencies; just
|
||||
point the existing OpenAI SDK at a local base URL.
|
||||
|
||||
Environment variables:
|
||||
EMBEDDING_PROVIDER=local (selects this provider)
|
||||
LOCAL_EMBEDDING_BASE_URL (default: http://localhost:11434/v1)
|
||||
LOCAL_EMBEDDING_MODEL (default: nomic-embed-text)
|
||||
LOCAL_EMBEDDING_DIMENSION (default: 768)
|
||||
LOCAL_EMBEDDING_API_KEY (optional; ignored by most local servers)
|
||||
|
||||
Setup (Ollama):
|
||||
$ ollama serve
|
||||
$ ollama pull nomic-embed-text # or bge-m3 for better Turkish
|
||||
|
||||
The dimension MUST match the model's actual output size (e.g. 768 for
|
||||
nomic-embed-text, 1024 for bge-m3, 1024 for mxbai-embed-large).
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_url: Optional[str] = None,
|
||||
model: Optional[str] = None,
|
||||
dimension: Optional[int] = None,
|
||||
api_key: Optional[str] = None,
|
||||
prompt_style: Optional[str] = None,
|
||||
):
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError:
|
||||
raise ImportError("openai package is required. Install with: pip install openai")
|
||||
|
||||
self.base_url = (
|
||||
base_url
|
||||
or os.getenv("LOCAL_EMBEDDING_BASE_URL")
|
||||
or LOCAL_DEFAULT_BASE_URL
|
||||
)
|
||||
# Most local servers don't validate the key — use a placeholder so
|
||||
# the OpenAI SDK doesn't error on the missing-key check.
|
||||
effective_key = (
|
||||
api_key
|
||||
or os.getenv("LOCAL_EMBEDDING_API_KEY")
|
||||
or "no-key-needed"
|
||||
)
|
||||
|
||||
self.client = OpenAI(base_url=self.base_url, api_key=effective_key)
|
||||
self.model = model or os.getenv("LOCAL_EMBEDDING_MODEL") or LOCAL_DEFAULT_MODEL
|
||||
self.dimension = _coerce_dimension(
|
||||
dimension if dimension is not None else os.getenv("LOCAL_EMBEDDING_DIMENSION"),
|
||||
"LOCAL_EMBEDDING_DIMENSION",
|
||||
LOCAL_DEFAULT_DIMENSION,
|
||||
)
|
||||
# Default to e5 prefix for local — the recommended Turkish setup
|
||||
# (multilingual-e5-large). Override via EMBEDDING_PROMPT_STYLE when
|
||||
# using a different model family (e.g. nomic, bge).
|
||||
self.prompt_style = _resolve_prompt_style(prompt_style, "e5")
|
||||
|
||||
logger.info(
|
||||
f"Local Embedder initialized: model={self.model} "
|
||||
f"base_url={self.base_url} dimension={self.dimension} "
|
||||
f"prompt_style={self.prompt_style}"
|
||||
)
|
||||
|
||||
|
||||
def get_embedder():
|
||||
"""
|
||||
Factory that picks the embedder based on EMBEDDING_PROVIDER.
|
||||
|
||||
- ``EMBEDDING_PROVIDER=local`` -> ``LocalEmbedder``
|
||||
- otherwise -> ``OpenRouterEmbedder`` (requires OPENROUTER_API_KEY)
|
||||
|
||||
Raises:
|
||||
ValueError: If no provider is configured (neither local nor OpenRouter).
|
||||
"""
|
||||
if is_local_embedding_configured():
|
||||
return LocalEmbedder()
|
||||
if is_openrouter_available():
|
||||
return OpenRouterEmbedder()
|
||||
raise ValueError(
|
||||
"No embedding provider configured. Set OPENROUTER_API_KEY for hosted "
|
||||
"embeddings, or EMBEDDING_PROVIDER=local (with LOCAL_EMBEDDING_* "
|
||||
"env vars) for a local OpenAI-compatible server like Ollama."
|
||||
)
|
||||
@@ -0,0 +1,305 @@
|
||||
# semantic_search/processor.py
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List, Dict, Any, Optional
|
||||
from dataclasses import dataclass
|
||||
import hashlib
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class DocumentChunk:
|
||||
"""Represents a chunk of a document."""
|
||||
chunk_id: str
|
||||
document_id: str
|
||||
text: str
|
||||
metadata: Dict[str, Any]
|
||||
chunk_index: int
|
||||
total_chunks: int
|
||||
|
||||
class DocumentProcessor:
|
||||
"""
|
||||
Processes legal documents for semantic search.
|
||||
Handles chunking, cleaning, and metadata extraction.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
chunk_size: int = 1000,
|
||||
chunk_overlap: int = 200,
|
||||
min_chunk_size: int = 100):
|
||||
"""
|
||||
Initialize document processor.
|
||||
|
||||
Args:
|
||||
chunk_size: Target size for each chunk in characters
|
||||
chunk_overlap: Number of overlapping characters between chunks
|
||||
min_chunk_size: Minimum chunk size to keep
|
||||
"""
|
||||
self.chunk_size = chunk_size
|
||||
self.chunk_overlap = chunk_overlap
|
||||
self.min_chunk_size = min_chunk_size
|
||||
|
||||
logger.info(f"Initialized DocumentProcessor (chunk_size={chunk_size}, overlap={chunk_overlap})")
|
||||
|
||||
def process_document(self,
|
||||
document_id: str,
|
||||
text: str,
|
||||
metadata: Optional[Dict[str, Any]] = None) -> List[DocumentChunk]:
|
||||
"""
|
||||
Process a single document into chunks.
|
||||
|
||||
Args:
|
||||
document_id: Unique document identifier
|
||||
text: Document text content
|
||||
metadata: Optional document metadata
|
||||
|
||||
Returns:
|
||||
List of document chunks
|
||||
"""
|
||||
if not text or len(text.strip()) < self.min_chunk_size:
|
||||
logger.warning(f"Document {document_id} too short to process")
|
||||
return []
|
||||
|
||||
# Clean text
|
||||
cleaned_text = self._clean_text(text)
|
||||
|
||||
# Extract metadata from text if not provided
|
||||
if metadata is None:
|
||||
metadata = {}
|
||||
|
||||
# Add extracted metadata
|
||||
extracted_metadata = self._extract_metadata(cleaned_text)
|
||||
metadata.update(extracted_metadata)
|
||||
|
||||
# Create chunks
|
||||
chunks = self._create_chunks(cleaned_text)
|
||||
|
||||
# Create DocumentChunk objects
|
||||
document_chunks = []
|
||||
for i, chunk_text in enumerate(chunks):
|
||||
chunk_id = self._generate_chunk_id(document_id, i)
|
||||
|
||||
chunk = DocumentChunk(
|
||||
chunk_id=chunk_id,
|
||||
document_id=document_id,
|
||||
text=chunk_text,
|
||||
metadata={
|
||||
**metadata,
|
||||
'chunk_index': i,
|
||||
'total_chunks': len(chunks)
|
||||
},
|
||||
chunk_index=i,
|
||||
total_chunks=len(chunks)
|
||||
)
|
||||
document_chunks.append(chunk)
|
||||
|
||||
logger.info(f"Processed document {document_id} into {len(chunks)} chunks")
|
||||
return document_chunks
|
||||
|
||||
def _clean_text(self, text: str) -> str:
|
||||
"""
|
||||
Clean and normalize text for processing.
|
||||
|
||||
Args:
|
||||
text: Raw text
|
||||
|
||||
Returns:
|
||||
Cleaned text
|
||||
"""
|
||||
# Remove excessive whitespace
|
||||
text = re.sub(r'\s+', ' ', text)
|
||||
|
||||
# Remove special characters but keep Turkish characters
|
||||
# Keep: letters, numbers, spaces, and common punctuation
|
||||
text = re.sub(r'[^\w\s\.\,\;\:\!\?\-\(\)\"\'ÇĞIİÖŞÜçğıiöşü]', ' ', text)
|
||||
|
||||
# Remove multiple spaces
|
||||
text = re.sub(r' +', ' ', text)
|
||||
|
||||
# Trim
|
||||
text = text.strip()
|
||||
|
||||
return text
|
||||
|
||||
def _extract_metadata(self, text: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Extract metadata from legal document text.
|
||||
|
||||
Args:
|
||||
text: Document text
|
||||
|
||||
Returns:
|
||||
Extracted metadata
|
||||
"""
|
||||
metadata = {}
|
||||
|
||||
# Extract case numbers (Esas/Karar)
|
||||
esas_pattern = r'E(?:sas)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
karar_pattern = r'K(?:arar)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
|
||||
esas_match = re.search(esas_pattern, text[:500]) # Look in first 500 chars
|
||||
if esas_match:
|
||||
metadata['esas_no'] = f"E.{esas_match.group(1)}/{esas_match.group(2)}"
|
||||
|
||||
karar_match = re.search(karar_pattern, text[:500])
|
||||
if karar_match:
|
||||
metadata['karar_no'] = f"K.{karar_match.group(1)}/{karar_match.group(2)}"
|
||||
|
||||
# Extract dates (DD.MM.YYYY or DD/MM/YYYY format)
|
||||
date_pattern = r'(\d{1,2})[\.\/](\d{1,2})[\.\/](\d{4})'
|
||||
dates = re.findall(date_pattern, text[:1000]) # Look in first 1000 chars
|
||||
if dates:
|
||||
# Take the first date as decision date
|
||||
day, month, year = dates[0]
|
||||
metadata['karar_tarihi'] = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
|
||||
# Extract court/chamber name
|
||||
chamber_patterns = [
|
||||
r'(\d+)\.\s*Hukuk\s+Dairesi',
|
||||
r'(\d+)\.\s*Ceza\s+Dairesi',
|
||||
r'Hukuk\s+Genel\s+Kurulu',
|
||||
r'Ceza\s+Genel\s+Kurulu',
|
||||
r'(\d+)\.\s*Daire'
|
||||
]
|
||||
|
||||
for pattern in chamber_patterns:
|
||||
match = re.search(pattern, text[:500], re.IGNORECASE)
|
||||
if match:
|
||||
metadata['chamber'] = match.group(0)
|
||||
break
|
||||
|
||||
return metadata
|
||||
|
||||
def _create_chunks(self, text: str) -> List[str]:
|
||||
"""
|
||||
Create overlapping chunks from text.
|
||||
|
||||
Args:
|
||||
text: Cleaned document text
|
||||
|
||||
Returns:
|
||||
List of text chunks
|
||||
"""
|
||||
chunks = []
|
||||
|
||||
# Split by sentences for better semantic coherence
|
||||
sentences = self._split_sentences(text)
|
||||
|
||||
current_chunk = []
|
||||
current_size = 0
|
||||
|
||||
for sentence in sentences:
|
||||
sentence_size = len(sentence)
|
||||
|
||||
# If adding this sentence exceeds chunk size
|
||||
if current_size + sentence_size > self.chunk_size and current_chunk:
|
||||
# Save current chunk
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
chunks.append(chunk_text)
|
||||
|
||||
# Create overlap for next chunk
|
||||
overlap_size = 0
|
||||
overlap_sentences = []
|
||||
|
||||
# Add sentences from the end until we reach overlap size
|
||||
for sent in reversed(current_chunk):
|
||||
overlap_size += len(sent)
|
||||
overlap_sentences.insert(0, sent)
|
||||
if overlap_size >= self.chunk_overlap:
|
||||
break
|
||||
|
||||
# Start new chunk with overlap
|
||||
current_chunk = overlap_sentences
|
||||
current_size = sum(len(s) for s in current_chunk)
|
||||
|
||||
# Add sentence to current chunk
|
||||
current_chunk.append(sentence)
|
||||
current_size += sentence_size
|
||||
|
||||
# Add final chunk if not empty
|
||||
if current_chunk:
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
if len(chunk_text) >= self.min_chunk_size:
|
||||
chunks.append(chunk_text)
|
||||
|
||||
return chunks
|
||||
|
||||
def _split_sentences(self, text: str) -> List[str]:
|
||||
"""
|
||||
Split text into sentences.
|
||||
|
||||
Args:
|
||||
text: Text to split
|
||||
|
||||
Returns:
|
||||
List of sentences
|
||||
"""
|
||||
# Simple sentence splitting for Turkish text
|
||||
# Split on period, question mark, exclamation, but not on abbreviations
|
||||
|
||||
# Common Turkish abbreviations to preserve
|
||||
abbreviations = ['Dr', 'Prof', 'Av', 'Md', 'Yrd', 'Doç', 'No', 'S', 'vs', 'vb', 'bkz']
|
||||
|
||||
# Replace abbreviations temporarily
|
||||
temp_text = text
|
||||
replacements = {}
|
||||
for i, abbr in enumerate(abbreviations):
|
||||
placeholder = f"__ABBR{i}__"
|
||||
temp_text = temp_text.replace(f"{abbr}.", placeholder)
|
||||
replacements[placeholder] = f"{abbr}."
|
||||
|
||||
# Split sentences
|
||||
sentence_endings = re.compile(r'[.!?]+')
|
||||
sentences = sentence_endings.split(temp_text)
|
||||
|
||||
# Restore abbreviations and clean
|
||||
cleaned_sentences = []
|
||||
for sentence in sentences:
|
||||
# Restore abbreviations
|
||||
for placeholder, original in replacements.items():
|
||||
sentence = sentence.replace(placeholder, original)
|
||||
|
||||
# Clean and add if not empty
|
||||
sentence = sentence.strip()
|
||||
if sentence and len(sentence) > 10: # Minimum sentence length
|
||||
cleaned_sentences.append(sentence)
|
||||
|
||||
return cleaned_sentences
|
||||
|
||||
def _generate_chunk_id(self, document_id: str, chunk_index: int) -> str:
|
||||
"""
|
||||
Generate unique chunk ID.
|
||||
|
||||
Args:
|
||||
document_id: Parent document ID
|
||||
chunk_index: Index of chunk in document
|
||||
|
||||
Returns:
|
||||
Unique chunk ID
|
||||
"""
|
||||
chunk_string = f"{document_id}_chunk_{chunk_index}"
|
||||
chunk_hash = hashlib.md5(chunk_string.encode()).hexdigest()[:8]
|
||||
return f"{document_id}_c{chunk_index}_{chunk_hash}"
|
||||
|
||||
def combine_chunks(self, chunks: List[DocumentChunk]) -> str:
|
||||
"""
|
||||
Combine chunks back into full document text.
|
||||
|
||||
Args:
|
||||
chunks: List of document chunks
|
||||
|
||||
Returns:
|
||||
Combined text
|
||||
"""
|
||||
if not chunks:
|
||||
return ""
|
||||
|
||||
# Sort by chunk index
|
||||
sorted_chunks = sorted(chunks, key=lambda x: x.chunk_index)
|
||||
|
||||
# For overlapping chunks, we need to be careful about duplication
|
||||
# Simple approach: just concatenate with space
|
||||
combined = " ".join([chunk.text for chunk in sorted_chunks])
|
||||
|
||||
return combined
|
||||
@@ -0,0 +1,235 @@
|
||||
# semantic_search/vector_store.py
|
||||
|
||||
import logging
|
||||
import numpy as np
|
||||
from typing import List, Dict, Any, Tuple, Optional
|
||||
from dataclasses import dataclass
|
||||
import json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class Document:
|
||||
"""Represents a document with its embedding and metadata."""
|
||||
id: str
|
||||
text: str
|
||||
embedding: np.ndarray
|
||||
metadata: Dict[str, Any]
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Convert to dictionary (excluding embedding for serialization)."""
|
||||
return {
|
||||
'id': self.id,
|
||||
'text': self.text,
|
||||
'metadata': self.metadata
|
||||
}
|
||||
|
||||
class VectorStore:
|
||||
"""
|
||||
In-memory vector storage with similarity search capabilities.
|
||||
Future versions can use Faiss, ChromaDB, or other vector databases.
|
||||
"""
|
||||
|
||||
def __init__(self, dimension: int = 768):
|
||||
"""
|
||||
Initialize vector store.
|
||||
|
||||
Args:
|
||||
dimension: Embedding dimension size
|
||||
"""
|
||||
self.dimension = dimension
|
||||
self.documents: List[Document] = []
|
||||
self.embeddings: Optional[np.ndarray] = None
|
||||
self.index_built = False
|
||||
|
||||
logger.info(f"Initialized VectorStore with dimension: {dimension}")
|
||||
|
||||
def add_documents(self,
|
||||
ids: List[str],
|
||||
texts: List[str],
|
||||
embeddings: np.ndarray,
|
||||
metadata: Optional[List[Dict[str, Any]]] = None) -> int:
|
||||
"""
|
||||
Add documents to the vector store.
|
||||
|
||||
Args:
|
||||
ids: Document IDs
|
||||
texts: Document texts
|
||||
embeddings: Document embeddings (N x dimension)
|
||||
metadata: Optional metadata for each document
|
||||
|
||||
Returns:
|
||||
Number of documents added
|
||||
"""
|
||||
if len(ids) != len(texts) or len(ids) != embeddings.shape[0]:
|
||||
raise ValueError("Mismatched lengths for ids, texts, and embeddings")
|
||||
|
||||
if metadata and len(metadata) != len(ids):
|
||||
raise ValueError("Metadata length doesn't match document count")
|
||||
|
||||
# Add documents
|
||||
for i in range(len(ids)):
|
||||
doc = Document(
|
||||
id=ids[i],
|
||||
text=texts[i],
|
||||
embedding=embeddings[i],
|
||||
metadata=metadata[i] if metadata else {}
|
||||
)
|
||||
self.documents.append(doc)
|
||||
|
||||
# Rebuild index
|
||||
self._build_index()
|
||||
|
||||
logger.info(f"Added {len(ids)} documents to vector store. Total: {len(self.documents)}")
|
||||
return len(ids)
|
||||
|
||||
def _build_index(self):
|
||||
"""Build or rebuild the embedding index."""
|
||||
if not self.documents:
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
return
|
||||
|
||||
# Stack all embeddings into a single array
|
||||
self.embeddings = np.vstack([doc.embedding for doc in self.documents])
|
||||
self.index_built = True
|
||||
|
||||
logger.debug(f"Built index with shape: {self.embeddings.shape}")
|
||||
|
||||
def search(self,
|
||||
query_embedding: np.ndarray,
|
||||
top_k: int = 10,
|
||||
threshold: Optional[float] = None) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Search for similar documents using cosine similarity.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
top_k: Number of results to return
|
||||
threshold: Optional similarity threshold (0-1)
|
||||
|
||||
Returns:
|
||||
List of (Document, similarity_score) tuples
|
||||
"""
|
||||
if not self.index_built or self.embeddings is None:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Ensure query is 2D
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarities (assuming normalized embeddings)
|
||||
similarities = np.dot(self.embeddings, query_embedding.T).squeeze()
|
||||
|
||||
# Apply threshold if specified
|
||||
if threshold is not None:
|
||||
valid_indices = np.where(similarities >= threshold)[0]
|
||||
if len(valid_indices) == 0:
|
||||
logger.info(f"No documents above threshold {threshold}")
|
||||
return []
|
||||
similarities = similarities[valid_indices]
|
||||
valid_docs = [self.documents[i] for i in valid_indices]
|
||||
else:
|
||||
valid_docs = self.documents
|
||||
|
||||
# Get top-k indices
|
||||
top_k = min(top_k, len(valid_docs))
|
||||
if top_k == 0:
|
||||
return []
|
||||
|
||||
# Use argpartition for efficiency with large arrays
|
||||
if len(similarities) > top_k:
|
||||
top_indices = np.argpartition(similarities, -top_k)[-top_k:]
|
||||
top_indices = top_indices[np.argsort(similarities[top_indices])[::-1]]
|
||||
else:
|
||||
top_indices = np.argsort(similarities)[::-1]
|
||||
|
||||
# Create results
|
||||
results = []
|
||||
for idx in top_indices:
|
||||
doc = valid_docs[idx] if threshold else self.documents[idx]
|
||||
score = float(similarities[idx])
|
||||
results.append((doc, score))
|
||||
|
||||
logger.info(f"Search returned {len(results)} results (top_k={top_k})")
|
||||
return results
|
||||
|
||||
def hybrid_search(self,
|
||||
query_embedding: np.ndarray,
|
||||
keyword_scores: Dict[str, float],
|
||||
top_k: int = 10,
|
||||
alpha: float = 0.5) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Hybrid search combining vector similarity and keyword scores.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
keyword_scores: Document ID to keyword relevance score mapping
|
||||
top_k: Number of results to return
|
||||
alpha: Weight for vector similarity (1-alpha for keyword score)
|
||||
|
||||
Returns:
|
||||
List of (Document, combined_score) tuples
|
||||
"""
|
||||
if not self.index_built:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Get vector similarities
|
||||
vector_results = self.search(query_embedding, top_k=len(self.documents))
|
||||
|
||||
# Combine scores
|
||||
combined_scores = []
|
||||
for doc, vector_score in vector_results:
|
||||
keyword_score = keyword_scores.get(doc.id, 0.0)
|
||||
# Normalize keyword score to 0-1 range if needed
|
||||
if keyword_score > 1.0:
|
||||
keyword_score = keyword_score / max(keyword_scores.values())
|
||||
|
||||
combined_score = alpha * vector_score + (1 - alpha) * keyword_score
|
||||
combined_scores.append((doc, combined_score))
|
||||
|
||||
# Sort by combined score and return top-k
|
||||
combined_scores.sort(key=lambda x: x[1], reverse=True)
|
||||
results = combined_scores[:top_k]
|
||||
|
||||
logger.info(f"Hybrid search returned {len(results)} results")
|
||||
return results
|
||||
|
||||
def clear(self):
|
||||
"""Clear all documents from the store."""
|
||||
self.documents = []
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
logger.info("Cleared vector store")
|
||||
|
||||
def size(self) -> int:
|
||||
"""Get number of documents in store."""
|
||||
return len(self.documents)
|
||||
|
||||
def get_by_id(self, doc_id: str) -> Optional[Document]:
|
||||
"""Get document by ID."""
|
||||
for doc in self.documents:
|
||||
if doc.id == doc_id:
|
||||
return doc
|
||||
return None
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""Get statistics about the vector store."""
|
||||
stats = {
|
||||
'num_documents': len(self.documents),
|
||||
'dimension': self.dimension,
|
||||
'index_built': self.index_built,
|
||||
'memory_usage_mb': 0
|
||||
}
|
||||
|
||||
if self.embeddings is not None:
|
||||
# Estimate memory usage
|
||||
memory_bytes = self.embeddings.nbytes
|
||||
for doc in self.documents:
|
||||
memory_bytes += len(doc.text.encode('utf-8'))
|
||||
memory_bytes += len(json.dumps(doc.metadata).encode('utf-8'))
|
||||
stats['memory_usage_mb'] = memory_bytes / (1024 * 1024)
|
||||
|
||||
return stats
|
||||
@@ -0,0 +1,21 @@
|
||||
# sigorta_tahkim_mcp_module/__init__.py
|
||||
|
||||
from .client import SigortaTahkimApiClient
|
||||
from .models import (
|
||||
SigortaTahkimSearchRequest,
|
||||
SigortaTahkimDecisionSummary,
|
||||
SigortaTahkimSearchResult,
|
||||
SigortaTahkimDocumentMarkdown,
|
||||
SigortaTahkimSearchWithinMatch,
|
||||
SigortaTahkimSearchWithinResult
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"SigortaTahkimApiClient",
|
||||
"SigortaTahkimSearchRequest",
|
||||
"SigortaTahkimDecisionSummary",
|
||||
"SigortaTahkimSearchResult",
|
||||
"SigortaTahkimDocumentMarkdown",
|
||||
"SigortaTahkimSearchWithinMatch",
|
||||
"SigortaTahkimSearchWithinResult"
|
||||
]
|
||||
@@ -0,0 +1,345 @@
|
||||
# sigorta_tahkim_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from typing import Optional
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
SigortaTahkimSearchRequest,
|
||||
SigortaTahkimDecisionSummary,
|
||||
SigortaTahkimSearchResult,
|
||||
SigortaTahkimDocumentMarkdown,
|
||||
SigortaTahkimSearchWithinMatch,
|
||||
SigortaTahkimSearchWithinResult
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
|
||||
# Turkish-specific lowercase: İ→i, I→ı (Python's str.lower() doesn't handle these)
|
||||
_TR_UPPER = str.maketrans("İIÇĞÖŞÜ", "iıçğöşü")
|
||||
|
||||
|
||||
def _turkish_lower(text: str) -> str:
|
||||
"""Lowercase with Turkish İ/I handling."""
|
||||
return text.translate(_TR_UPPER).lower()
|
||||
|
||||
|
||||
class SigortaTahkimApiClient:
|
||||
"""
|
||||
API client for searching and retrieving Sigorta Tahkim Komisyonu
|
||||
(Insurance Arbitration Commission) decisions using Tavily Search API
|
||||
for discovery and direct PDF download for content retrieval.
|
||||
|
||||
The commission publishes quarterly PDF journals ("Hakem Karar Dergisi")
|
||||
containing arbitration decisions. There are 64 issues spanning 2010-2025.
|
||||
"""
|
||||
|
||||
TAVILY_API_URL = "https://api.tavily.com/search"
|
||||
BASE_URL = "https://www.sigortatahkim.org"
|
||||
PDF_BASE_URL = "https://www.sigortatahkim.org/content/CmsFiles/"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
"""Initialize the Sigorta Tahkim API client."""
|
||||
self.tavily_api_key = os.getenv("TAVILY_API_KEY")
|
||||
if not self.tavily_api_key:
|
||||
self.tavily_api_key = "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"
|
||||
logger.info("Using fallback Tavily API token (development token)")
|
||||
else:
|
||||
logger.info("Using Tavily API key from environment variable")
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
headers={
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
|
||||
},
|
||||
timeout=httpx.Timeout(request_timeout)
|
||||
)
|
||||
self.markitdown = MarkItDown()
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close the HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("SigortaTahkimApiClient: HTTP client session closed.")
|
||||
|
||||
def _get_pdf_filename(self, issue_number: int) -> str:
|
||||
"""Get the PDF filename for a given journal issue number."""
|
||||
if issue_number == 4:
|
||||
return "karardergisisayi4.pdf"
|
||||
elif 57 <= issue_number <= 61:
|
||||
return f"revizekd{issue_number}.pdf"
|
||||
else:
|
||||
return f"karardrgs{issue_number}.pdf"
|
||||
|
||||
def _extract_issue_number(self, url: str) -> Optional[str]:
|
||||
"""Extract journal issue number from a sigortatahkim.org URL."""
|
||||
# Pattern: karardrgs{N}.pdf
|
||||
match = re.search(r'karardrgs(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: revizekd{N}.pdf
|
||||
match = re.search(r'revizekd(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: karardergisisayi{N}.pdf
|
||||
match = re.search(r'karardergisisayi(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: sayı or sayi in URL path with number
|
||||
match = re.search(r'say[ıi]\s*[-:]?\s*(\d+)', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
return None
|
||||
|
||||
async def search_decisions(
|
||||
self,
|
||||
request: SigortaTahkimSearchRequest
|
||||
) -> SigortaTahkimSearchResult:
|
||||
"""
|
||||
Search for Sigorta Tahkim Komisyonu decisions using Tavily API.
|
||||
|
||||
Args:
|
||||
request: Search request parameters
|
||||
|
||||
Returns:
|
||||
SigortaTahkimSearchResult with matching decisions
|
||||
"""
|
||||
try:
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.tavily_api_key}"
|
||||
}
|
||||
|
||||
payload = {
|
||||
"query": request.keywords,
|
||||
"country": "turkey",
|
||||
"include_domains": ["sigortatahkim.org"],
|
||||
"max_results": request.pageSize,
|
||||
"search_depth": "advanced"
|
||||
}
|
||||
|
||||
if request.page > 1:
|
||||
logger.warning(f"Tavily API doesn't support pagination. Page {request.page} requested.")
|
||||
|
||||
response = await self.http_client.post(
|
||||
self.TAVILY_API_URL,
|
||||
json=payload,
|
||||
headers=headers
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
logger.info(f"Tavily returned {len(data.get('results', []))} results for Sigorta Tahkim")
|
||||
|
||||
decisions = []
|
||||
for result in data.get("results", []):
|
||||
url = result.get("url", "")
|
||||
title = result.get("title", "").strip()
|
||||
content = result.get("content", "")[:500]
|
||||
|
||||
issue_num = self._extract_issue_number(url)
|
||||
doc_id = issue_num if issue_num else url
|
||||
|
||||
decision = SigortaTahkimDecisionSummary(
|
||||
title=title,
|
||||
document_id=doc_id,
|
||||
content=content,
|
||||
url=url
|
||||
)
|
||||
decisions.append(decision)
|
||||
|
||||
return SigortaTahkimSearchResult(
|
||||
decisions=decisions,
|
||||
total_results=len(data.get("results", [])),
|
||||
page=request.page,
|
||||
pageSize=request.pageSize
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error searching Sigorta Tahkim decisions: {e}")
|
||||
if e.response.status_code == 401:
|
||||
raise Exception("Tavily API authentication failed. Check API key.")
|
||||
raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error searching Sigorta Tahkim decisions: {e}")
|
||||
raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}")
|
||||
|
||||
# Regex pattern to split decisions within a journal issue
|
||||
DECISION_HEADER_PATTERN = re.compile(
|
||||
r'(\d{2}\.\d{2}\.\d{4}\s+Tarih\s+ve\s+K-\d{4}/\d+\s+Sayılı\s+Hakem\s+Kararı)'
|
||||
)
|
||||
# Minimum body length to distinguish real decisions from TOC entries
|
||||
MIN_DECISION_BODY_LENGTH = 1000
|
||||
|
||||
async def _download_and_convert_pdf(self, issue_number: str) -> tuple[str, str]:
|
||||
"""
|
||||
Download a journal issue PDF and convert to markdown.
|
||||
|
||||
Returns:
|
||||
Tuple of (markdown_content, pdf_url)
|
||||
"""
|
||||
issue_num = int(issue_number)
|
||||
filename = self._get_pdf_filename(issue_num)
|
||||
pdf_url = f"{self.PDF_BASE_URL}{filename}"
|
||||
|
||||
logger.info(f"Downloading Sigorta Tahkim PDF: {pdf_url}")
|
||||
|
||||
response = await self.http_client.get(pdf_url, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
|
||||
pdf_stream = io.BytesIO(response.content)
|
||||
# markitdown is sync; offload to thread so PDF parsing doesn't block
|
||||
# the event-loop / other in-flight MCP requests.
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, pdf_stream, file_extension=".pdf"
|
||||
)
|
||||
return result.text_content.strip(), pdf_url
|
||||
|
||||
def _split_into_decisions(self, markdown_content: str) -> list[tuple[str, str]]:
|
||||
"""
|
||||
Split markdown content into individual decisions.
|
||||
|
||||
Returns:
|
||||
List of (header, body) tuples for decisions with substantial content.
|
||||
"""
|
||||
parts = self.DECISION_HEADER_PATTERN.split(markdown_content)
|
||||
decisions = []
|
||||
for i in range(1, len(parts) - 1, 2):
|
||||
header = parts[i].strip()
|
||||
body = parts[i + 1].strip() if i + 1 < len(parts) else ""
|
||||
if len(body) >= self.MIN_DECISION_BODY_LENGTH:
|
||||
decisions.append((header, body))
|
||||
return decisions
|
||||
|
||||
async def get_document_markdown(
|
||||
self,
|
||||
issue_number: str,
|
||||
page_number: int = 1
|
||||
) -> SigortaTahkimDocumentMarkdown:
|
||||
"""
|
||||
Retrieve a Sigorta Tahkim journal issue PDF and convert to Markdown.
|
||||
|
||||
Args:
|
||||
issue_number: Journal issue number (e.g., '64')
|
||||
page_number: Page number for paginated content (1-indexed)
|
||||
|
||||
Returns:
|
||||
SigortaTahkimDocumentMarkdown with paginated content
|
||||
"""
|
||||
try:
|
||||
markdown_content, pdf_url = await self._download_and_convert_pdf(issue_number)
|
||||
|
||||
total_length = len(markdown_content)
|
||||
total_pages = max(1, math.ceil(total_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE))
|
||||
|
||||
start_idx = (page_number - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_idx = start_idx + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
page_content = markdown_content[start_idx:end_idx]
|
||||
|
||||
return SigortaTahkimDocumentMarkdown(
|
||||
document_id=issue_number,
|
||||
markdown_content=page_content,
|
||||
page_number=page_number,
|
||||
total_pages=total_pages,
|
||||
source_url=pdf_url
|
||||
)
|
||||
|
||||
except ValueError:
|
||||
raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').")
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error fetching Sigorta Tahkim issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error processing Sigorta Tahkim issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to process journal issue {issue_number}: {str(e)}")
|
||||
|
||||
async def search_within_issue(
|
||||
self,
|
||||
issue_number: str,
|
||||
keyword: str,
|
||||
max_results: int = 10
|
||||
) -> SigortaTahkimSearchWithinResult:
|
||||
"""
|
||||
Search for a keyword within a specific journal issue's decisions.
|
||||
|
||||
Downloads the PDF, splits into individual decisions, and returns
|
||||
matching decisions sorted by relevance (match count).
|
||||
|
||||
Args:
|
||||
issue_number: Journal issue number (e.g., '64')
|
||||
keyword: Search keyword or phrase in Turkish
|
||||
max_results: Maximum matching decisions to return
|
||||
|
||||
Returns:
|
||||
SigortaTahkimSearchWithinResult with matching decisions
|
||||
"""
|
||||
try:
|
||||
markdown_content, _ = await self._download_and_convert_pdf(issue_number)
|
||||
decisions = self._split_into_decisions(markdown_content)
|
||||
|
||||
logger.info(
|
||||
f"Searching '{keyword}' within issue {issue_number}: "
|
||||
f"{len(decisions)} decisions found"
|
||||
)
|
||||
|
||||
keyword_lower = _turkish_lower(keyword)
|
||||
matches = []
|
||||
|
||||
for header, body in decisions:
|
||||
body_lower = _turkish_lower(body)
|
||||
count = body_lower.count(keyword_lower)
|
||||
if count == 0:
|
||||
continue
|
||||
|
||||
# Extract excerpt around the first match
|
||||
first_pos = body_lower.find(keyword_lower)
|
||||
excerpt_start = max(0, first_pos - 200)
|
||||
excerpt_end = min(len(body), first_pos + len(keyword) + 200)
|
||||
excerpt = body[excerpt_start:excerpt_end].strip()
|
||||
if excerpt_start > 0:
|
||||
excerpt = "..." + excerpt
|
||||
if excerpt_end < len(body):
|
||||
excerpt = excerpt + "..."
|
||||
|
||||
matches.append(SigortaTahkimSearchWithinMatch(
|
||||
decision_header=header,
|
||||
relevance_score=count,
|
||||
excerpt=excerpt,
|
||||
body_length=len(body)
|
||||
))
|
||||
|
||||
# Sort by relevance (highest match count first)
|
||||
matches.sort(key=lambda m: m.relevance_score, reverse=True)
|
||||
matches = matches[:max_results]
|
||||
|
||||
return SigortaTahkimSearchWithinResult(
|
||||
issue_number=issue_number,
|
||||
keyword=keyword,
|
||||
total_decisions=len(decisions),
|
||||
matching_decisions=len(matches),
|
||||
matches=matches
|
||||
)
|
||||
|
||||
except ValueError:
|
||||
raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').")
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error in search_within issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error in search_within issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to search within issue {issue_number}: {str(e)}")
|
||||
@@ -0,0 +1,59 @@
|
||||
# sigorta_tahkim_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List
|
||||
|
||||
|
||||
class SigortaTahkimSearchRequest(BaseModel):
|
||||
"""Request model for searching Sigorta Tahkim Komisyonu decisions via Tavily API."""
|
||||
keywords: str = Field(..., description="Search keywords in Turkish")
|
||||
page: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)")
|
||||
|
||||
|
||||
class SigortaTahkimDecisionSummary(BaseModel):
|
||||
"""Summary of a Sigorta Tahkim decision from search results."""
|
||||
title: str = Field(..., description="Decision title or journal issue info")
|
||||
document_id: str = Field(..., description="Journal issue number (e.g., '64')")
|
||||
content: str = Field(..., description="Decision summary/excerpt")
|
||||
url: str = Field("", description="Source URL")
|
||||
|
||||
|
||||
class SigortaTahkimSearchResult(BaseModel):
|
||||
"""Response model for Sigorta Tahkim decision search results."""
|
||||
decisions: List[SigortaTahkimDecisionSummary] = Field(
|
||||
default_factory=list,
|
||||
description="List of matching decisions"
|
||||
)
|
||||
total_results: int = Field(0, description="Total number of results")
|
||||
page: int = Field(1, description="Current page number")
|
||||
pageSize: int = Field(10, description="Results per page")
|
||||
|
||||
|
||||
class SigortaTahkimDocumentMarkdown(BaseModel):
|
||||
"""Sigorta Tahkim journal issue converted to Markdown format."""
|
||||
document_id: str = Field(..., description="Journal issue number")
|
||||
markdown_content: str = Field("", description="Document content in Markdown")
|
||||
page_number: int = Field(1, description="Current page number")
|
||||
total_pages: int = Field(1, description="Total number of pages")
|
||||
source_url: str = Field("", description="PDF source URL")
|
||||
|
||||
|
||||
class SigortaTahkimSearchWithinMatch(BaseModel):
|
||||
"""A single matching decision from search within a journal issue."""
|
||||
decision_header: str = Field(..., description="Decision header (date and K-number)")
|
||||
relevance_score: int = Field(0, description="Number of keyword matches")
|
||||
excerpt: str = Field("", description="Matching excerpt with context")
|
||||
body_length: int = Field(0, description="Full decision body length in chars")
|
||||
|
||||
|
||||
class SigortaTahkimSearchWithinResult(BaseModel):
|
||||
"""Response model for search within a journal issue."""
|
||||
issue_number: str = Field(..., description="Journal issue number searched")
|
||||
keyword: str = Field("", description="Search keyword used")
|
||||
total_decisions: int = Field(0, description="Total decisions in issue")
|
||||
matching_decisions: int = Field(0, description="Number of matching decisions")
|
||||
matches: List[SigortaTahkimSearchWithinMatch] = Field(
|
||||
default_factory=list,
|
||||
description="List of matching decisions sorted by relevance"
|
||||
)
|
||||
@@ -1,159 +0,0 @@
|
||||
"""
|
||||
Starlette integration example for Yargı MCP Server
|
||||
|
||||
This module demonstrates how to integrate the Yargı MCP server
|
||||
with a Starlette application, including authentication middleware
|
||||
and custom routing.
|
||||
|
||||
Usage:
|
||||
uvicorn starlette_app:app --host 0.0.0.0 --port 8000
|
||||
"""
|
||||
|
||||
import os
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.middleware.authentication import AuthenticationMiddleware
|
||||
from starlette.authentication import (
|
||||
AuthenticationBackend, AuthCredentials, SimpleUser, AuthenticationError
|
||||
)
|
||||
|
||||
# Import the main MCP app
|
||||
from mcp_server_main import app as mcp_server
|
||||
|
||||
# Simple token authentication backend
|
||||
class TokenAuthBackend(AuthenticationBackend):
|
||||
async def authenticate(self, request):
|
||||
auth_header = request.headers.get("Authorization")
|
||||
expected_token = os.getenv("API_TOKEN")
|
||||
|
||||
# Skip auth for health check and public endpoints
|
||||
if request.url.path in ["/health", "/", "/login"]:
|
||||
return None
|
||||
|
||||
if not expected_token:
|
||||
# No token configured, allow all
|
||||
return AuthCredentials(["authenticated"]), SimpleUser("anonymous")
|
||||
|
||||
if not auth_header:
|
||||
raise AuthenticationError("Authorization header required")
|
||||
|
||||
try:
|
||||
scheme, token = auth_header.split()
|
||||
if scheme.lower() != "bearer":
|
||||
raise AuthenticationError("Invalid authentication scheme")
|
||||
|
||||
if token != expected_token:
|
||||
raise AuthenticationError("Invalid token")
|
||||
|
||||
return AuthCredentials(["authenticated"]), SimpleUser("user")
|
||||
except ValueError:
|
||||
raise AuthenticationError("Invalid authorization header format")
|
||||
|
||||
# Homepage
|
||||
async def homepage(request: Request):
|
||||
return JSONResponse({
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"endpoints": {
|
||||
"mcp": "/mcp-server/mcp/",
|
||||
"api": "/api/",
|
||||
"health": "/health"
|
||||
}
|
||||
})
|
||||
|
||||
# API info endpoint
|
||||
async def api_info(request: Request):
|
||||
if not request.user.is_authenticated:
|
||||
return JSONResponse({"error": "Authentication required"}, status_code=401)
|
||||
|
||||
return JSONResponse({
|
||||
"authenticated_as": request.user.display_name,
|
||||
"available_tools": len(mcp_server._tool_manager._tools),
|
||||
"databases": [
|
||||
"Yargıtay", "Danıştay", "Emsal", "Uyuşmazlık",
|
||||
"Anayasa", "KIK", "Rekabet", "Bedesten"
|
||||
]
|
||||
})
|
||||
|
||||
# Health check
|
||||
async def health_check(request: Request):
|
||||
return JSONResponse({
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server"
|
||||
})
|
||||
|
||||
# Login example (returns token for demo)
|
||||
async def login(request: Request):
|
||||
token = os.getenv("API_TOKEN", "demo-token")
|
||||
return JSONResponse({
|
||||
"message": "Use this token in Authorization header",
|
||||
"example": f"Authorization: Bearer {token}",
|
||||
"note": "Set API_TOKEN environment variable to change token"
|
||||
})
|
||||
|
||||
# Create MCP ASGI app
|
||||
mcp_app = mcp_server.http_app(path='/mcp')
|
||||
|
||||
# Configure middleware
|
||||
middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=os.getenv("ALLOWED_ORIGINS", "*").split(","),
|
||||
allow_credentials=True,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
),
|
||||
Middleware(AuthenticationMiddleware, backend=TokenAuthBackend()),
|
||||
]
|
||||
|
||||
# Create routes
|
||||
routes = [
|
||||
Route("/", homepage),
|
||||
Route("/health", health_check),
|
||||
Route("/login", login),
|
||||
Route("/api/info", api_info),
|
||||
Mount("/mcp-server", app=mcp_app),
|
||||
]
|
||||
|
||||
# Create Starlette app
|
||||
app = Starlette(
|
||||
routes=routes,
|
||||
middleware=middleware,
|
||||
lifespan=mcp_app.lifespan
|
||||
)
|
||||
|
||||
# Nested mount example
|
||||
def create_nested_app():
|
||||
"""Example of nested mounting for complex routing structures"""
|
||||
|
||||
# Create inner app with MCP
|
||||
inner_app = Starlette(
|
||||
routes=[Mount("/services", app=mcp_app)],
|
||||
middleware=middleware
|
||||
)
|
||||
|
||||
# Create outer app
|
||||
outer_app = Starlette(
|
||||
routes=[
|
||||
Route("/", homepage),
|
||||
Mount("/v1", app=inner_app),
|
||||
],
|
||||
lifespan=mcp_app.lifespan
|
||||
)
|
||||
|
||||
# MCP would be available at /v1/services/mcp/
|
||||
return outer_app
|
||||
|
||||
# Export both apps
|
||||
nested_app = create_nested_app()
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
print("Starting Starlette app with authentication...")
|
||||
print("Set API_TOKEN environment variable to enable authentication")
|
||||
print("Example: API_TOKEN=secret-token python starlette_app.py")
|
||||
uvicorn.run(app, host="0.0.0.0", port=8000)
|
||||
@@ -1,25 +0,0 @@
|
||||
import os, stripe
|
||||
from clerk_backend_api import Clerk # Clerk backend SDK
|
||||
from fastapi import APIRouter, Request, HTTPException
|
||||
|
||||
router = APIRouter()
|
||||
stripe.api_key = os.getenv("STRIPE_SECRET")
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
|
||||
@router.post("/stripe/webhook")
|
||||
async def stripe_hook(req: Request):
|
||||
payload, sig = await req.body(), req.headers["stripe-signature"]
|
||||
try:
|
||||
event = stripe.Webhook.construct_event( # Stripe-recommended verify
|
||||
payload, sig, os.getenv("STRIPE_WEBHOOK_SECRET"))
|
||||
except stripe.error.SignatureVerificationError:
|
||||
raise HTTPException(400, "Bad sig")
|
||||
|
||||
if event["type"] == "customer.subscription.updated":
|
||||
item = event["data"]["object"]["items"]["data"][0]
|
||||
plan = item["price"]["nickname"] # "Pro", "Enterprise"…
|
||||
userID = event["data"]["object"]["metadata"]["clerk_user_id"]
|
||||
clerk.users.update_user_metadata( # merge into unsafe_metadata
|
||||
userID, unsafe_metadata={"plan": plan})
|
||||
return {"ok": True}
|
||||
|
||||
+143
-208
@@ -1,244 +1,179 @@
|
||||
# uyusmazlik_mcp_module/client.py
|
||||
#
|
||||
# Client for the rebuilt Uyuşmazlık Mahkemesi search site
|
||||
# (https://kararlar.uyusmazlik.gov.tr). The site is an ASP.NET WebForms app:
|
||||
# searching is a form postback against "/" that returns an HTML page with a
|
||||
# GridView of results, and each decision is a PDF served from /Uploads/.
|
||||
#
|
||||
# The previous AJAX endpoint (/Arama/Search) was retired and now returns 404.
|
||||
|
||||
import asyncio
|
||||
import io
|
||||
import logging
|
||||
import re
|
||||
from typing import Dict, List, Optional
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import httpx
|
||||
import aiohttp
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Union, Tuple
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
from markitdown import MarkItDown
|
||||
from urllib.parse import urljoin, urlencode # urlencode for aiohttp form data
|
||||
|
||||
from .models import (
|
||||
UyusmazlikSearchRequest,
|
||||
UyusmazlikApiDecisionEntry,
|
||||
UyusmazlikSearchResponse,
|
||||
UyusmazlikDocumentMarkdown,
|
||||
UyusmazlikBolumEnum,
|
||||
UyusmazlikTuruEnum,
|
||||
UyusmazlikKararSonucuEnum
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s')
|
||||
|
||||
# --- Mappings from user-friendly Enum values to API IDs ---
|
||||
BOLUM_ENUM_TO_ID_MAP = {
|
||||
UyusmazlikBolumEnum.CEZA_BOLUMU: "f6b74320-f2d7-4209-ad6e-c6df180d4e7c",
|
||||
UyusmazlikBolumEnum.GENEL_KURUL_KARARLARI: "e4ca658d-a75a-4719-b866-b2d2f1c3b1d9",
|
||||
UyusmazlikBolumEnum.HUKUK_BOLUMU: "96b26fc4-ef8e-4a4f-a9cc-a3de89952aa1",
|
||||
UyusmazlikBolumEnum.TUMU: "", # Represents "...Seçiniz..." or all - empty string for API
|
||||
"ALL": "" # Also map the new "ALL" literal to empty string for backward compatibility
|
||||
}
|
||||
# ASP.NET hidden fields that must be round-tripped on every postback.
|
||||
_HIDDEN_FIELDS = ("__VIEWSTATE", "__VIEWSTATEGENERATOR", "__EVENTVALIDATION")
|
||||
|
||||
UYUSMAZLIK_TURU_ENUM_TO_ID_MAP = {
|
||||
UyusmazlikTuruEnum.GOREV_UYUSMAZLIGI: "7b1e2cd3-8f09-418a-921c-bbe501e1740c",
|
||||
UyusmazlikTuruEnum.HUKUM_UYUSMAZLIGI: "19b88402-172b-4c1d-8339-595c942a89f5",
|
||||
UyusmazlikTuruEnum.TUMU: "", # Represents "...Seçiniz..." or all - empty string for API
|
||||
"ALL": "" # Also map the new "ALL" literal to empty string for backward compatibility
|
||||
}
|
||||
|
||||
KARAR_SONUCU_ENUM_TO_ID_MAP = {
|
||||
# These IDs are from the form HTML provided by the user
|
||||
UyusmazlikKararSonucuEnum.HUKUM_UYUSMAZLIGI_OLMADIGINA_DAIR: "6f47d87f-dcb5-412e-9878-000385dba1d9",
|
||||
UyusmazlikKararSonucuEnum.HUKUM_UYUSMAZLIGI_OLDUGUNA_DAIR: "5a01742a-c440-4c4a-ba1f-da20837cffed",
|
||||
# Add all other 'Karar Sonucu' enum members and their corresponding GUIDs
|
||||
# by inspecting the 'KararSonucuList' checkboxes in the provided form HTML.
|
||||
}
|
||||
# --- End Mappings ---
|
||||
|
||||
class UyusmazlikApiClient:
|
||||
BASE_URL = "https://kararlar.uyusmazlik.gov.tr"
|
||||
SEARCH_ENDPOINT = "/Arama/Search"
|
||||
# Individual documents are fetched by their full URLs obtained from search results.
|
||||
SEARCH_PATH = "/"
|
||||
|
||||
def __init__(self, request_timeout: float = 30.0):
|
||||
self.request_timeout = request_timeout # Store timeout for aiohttp and httpx
|
||||
# Headers for aiohttp search. httpx for docs will create its own.
|
||||
self.default_aiohttp_search_headers = {
|
||||
"Accept": "*/*", # Mimicking browser headers provided by user
|
||||
"Accept-Encoding": "gzip, deflate, br, zstd",
|
||||
self.request_timeout = request_timeout
|
||||
# A persistent cookie-aware client so ASP.NET session/viewstate are kept.
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"X-Requested-With": "XMLHttpRequest",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": self.BASE_URL + "/",
|
||||
|
||||
}
|
||||
|
||||
|
||||
async def search_decisions(
|
||||
self,
|
||||
params: UyusmazlikSearchRequest
|
||||
) -> UyusmazlikSearchResponse:
|
||||
|
||||
bolum_id_for_api = BOLUM_ENUM_TO_ID_MAP.get(params.bolum, "")
|
||||
uyusmazlik_id_for_api = UYUSMAZLIK_TURU_ENUM_TO_ID_MAP.get(params.uyusmazlik_turu, "")
|
||||
|
||||
form_data_list: List[Tuple[str, str]] = []
|
||||
|
||||
def add_to_form_data(key: str, value: Optional[str]):
|
||||
# API expects empty strings for omitted optional fields based on user payload example
|
||||
form_data_list.append((key, value or ""))
|
||||
|
||||
add_to_form_data("BolumId", bolum_id_for_api)
|
||||
add_to_form_data("UyusmazlikId", uyusmazlik_id_for_api)
|
||||
|
||||
if params.karar_sonuclari:
|
||||
for enum_member in params.karar_sonuclari:
|
||||
api_id = KARAR_SONUCU_ENUM_TO_ID_MAP.get(enum_member)
|
||||
if api_id: # Only add if a valid ID is found
|
||||
form_data_list.append(('KararSonucuList', api_id))
|
||||
|
||||
add_to_form_data("EsasYil", params.esas_yil)
|
||||
add_to_form_data("EsasSayisi", params.esas_sayisi)
|
||||
add_to_form_data("KararYil", params.karar_yil)
|
||||
add_to_form_data("KararSayisi", params.karar_sayisi)
|
||||
add_to_form_data("KanunNo", params.kanun_no)
|
||||
add_to_form_data("KararDateBegin", params.karar_date_begin)
|
||||
add_to_form_data("KararDateEnd", params.karar_date_end)
|
||||
add_to_form_data("ResmiGazeteSayi", params.resmi_gazete_sayi)
|
||||
add_to_form_data("ResmiGazeteDate", params.resmi_gazete_date)
|
||||
add_to_form_data("Icerik", params.icerik)
|
||||
add_to_form_data("Tumce", params.tumce)
|
||||
add_to_form_data("WildCard", params.wild_card)
|
||||
add_to_form_data("Hepsi", params.hepsi)
|
||||
add_to_form_data("Herhangibirisi", params.herhangi_birisi)
|
||||
add_to_form_data("NotHepsi", params.not_hepsi)
|
||||
# X-Requested-With is handled by default_aiohttp_search_headers
|
||||
|
||||
search_url = urljoin(self.BASE_URL, self.SEARCH_ENDPOINT)
|
||||
# For aiohttp, data for application/x-www-form-urlencoded should be a dict or str.
|
||||
# Using urlencode for list of tuples.
|
||||
encoded_form_payload = urlencode(form_data_list, encoding='UTF-8')
|
||||
|
||||
logger.info(f"UyusmazlikApiClient (aiohttp): Performing search to {search_url} with form_data: {encoded_form_payload}")
|
||||
|
||||
html_content = ""
|
||||
aiohttp_headers = self.default_aiohttp_search_headers.copy()
|
||||
aiohttp_headers["Content-Type"] = "application/x-www-form-urlencoded; charset=UTF-8"
|
||||
|
||||
try:
|
||||
# Create a new session for each call for simplicity with aiohttp here
|
||||
async with aiohttp.ClientSession(headers=aiohttp_headers) as session:
|
||||
async with session.post(search_url, data=encoded_form_payload, timeout=self.request_timeout) as response:
|
||||
response.raise_for_status() # Raises ClientResponseError for 400-599
|
||||
html_content = await response.text(encoding='utf-8') # Ensure correct encoding
|
||||
logger.debug("UyusmazlikApiClient (aiohttp): Received HTML response for search.")
|
||||
|
||||
except aiohttp.ClientError as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): HTTP client error during search: {e}")
|
||||
raise # Re-raise to be handled by the MCP tool
|
||||
except Exception as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): Error processing search request: {e}")
|
||||
raise
|
||||
|
||||
# --- HTML Parsing (remains the same as previous version) ---
|
||||
soup = BeautifulSoup(html_content, 'html.parser')
|
||||
total_records_text_div = soup.find("div", class_="pull-right label label-important")
|
||||
total_records = None
|
||||
if total_records_text_div:
|
||||
match_records = re.search(r'(\d+)\s*adet kayıt bulundu', total_records_text_div.get_text(strip=True))
|
||||
if match_records:
|
||||
total_records = int(match_records.group(1))
|
||||
|
||||
result_table = soup.find("table", class_="table-hover")
|
||||
processed_decisions: List[UyusmazlikApiDecisionEntry] = []
|
||||
if result_table:
|
||||
rows = result_table.find_all("tr")
|
||||
if len(rows) > 1: # Skip header row
|
||||
for row in rows[1:]:
|
||||
cols = row.find_all('td')
|
||||
if len(cols) >= 5:
|
||||
try:
|
||||
popover_div = cols[0].find("div", attrs={"data-rel": "popover"})
|
||||
popover_content_raw = popover_div["data-content"] if popover_div and popover_div.has_attr("data-content") else None
|
||||
|
||||
link_tag = cols[0].find('a')
|
||||
doc_relative_url = link_tag['href'] if link_tag and link_tag.has_attr('href') else None
|
||||
|
||||
if not doc_relative_url: continue
|
||||
document_url_str = urljoin(self.BASE_URL, doc_relative_url)
|
||||
|
||||
pdf_link_tag = cols[5].find('a', href=re.compile(r'\.pdf$', re.IGNORECASE)) if len(cols) > 5 else None
|
||||
pdf_url_str = urljoin(self.BASE_URL, pdf_link_tag['href']) if pdf_link_tag and pdf_link_tag.has_attr('href') else None
|
||||
|
||||
decision_data_parsed = {
|
||||
"karar_sayisi": cols[0].get_text(strip=True),
|
||||
"esas_sayisi": cols[1].get_text(strip=True),
|
||||
"bolum": cols[2].get_text(strip=True),
|
||||
"uyusmazlik_konusu": cols[3].get_text(strip=True),
|
||||
"karar_sonucu": cols[4].get_text(strip=True),
|
||||
"popover_content": html.unescape(popover_content_raw) if popover_content_raw else None,
|
||||
"document_url": document_url_str,
|
||||
"pdf_url": pdf_url_str
|
||||
}
|
||||
decision_model = UyusmazlikApiDecisionEntry(**decision_data_parsed)
|
||||
processed_decisions.append(decision_model)
|
||||
except Exception as e:
|
||||
logger.warning(f"UyusmazlikApiClient: Could not parse decision row. Row content: {row.get_text(strip=True, separator=' | ')}, Error: {e}")
|
||||
|
||||
return UyusmazlikSearchResponse(
|
||||
decisions=processed_decisions,
|
||||
total_records_found=total_records
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=False,
|
||||
follow_redirects=True,
|
||||
)
|
||||
|
||||
def _convert_html_to_markdown_uyusmazlik(self, full_decision_html_content: str) -> Optional[str]:
|
||||
"""Converts direct HTML content (from an Uyuşmazlık decision page) to Markdown."""
|
||||
if not full_decision_html_content:
|
||||
@staticmethod
|
||||
def _extract_hidden_fields(html_content: str) -> Dict[str, str]:
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
fields: Dict[str, str] = {}
|
||||
for name in _HIDDEN_FIELDS:
|
||||
tag = soup.find("input", attrs={"name": name})
|
||||
fields[name] = tag["value"] if tag and tag.has_attr("value") else ""
|
||||
return fields
|
||||
|
||||
@staticmethod
|
||||
def _parse_results(html_content: str, base_url: str) -> UyusmazlikSearchResponse:
|
||||
soup = BeautifulSoup(html_content, "html.parser")
|
||||
|
||||
decisions: List[UyusmazlikApiDecisionEntry] = []
|
||||
grid = soup.find("table", id="GridView1")
|
||||
if grid:
|
||||
rows = grid.find_all("tr")
|
||||
for row in rows[1:]: # skip header row
|
||||
cells = row.find_all("td")
|
||||
if len(cells) < 4:
|
||||
continue
|
||||
# The İşlemler cell holds the PDF "Görüntüle" link. Pager rows also
|
||||
# contain <a> tags (javascript:__doPostBack ...), so require a real
|
||||
# document link and skip everything else.
|
||||
link_tag = cells[3].find(
|
||||
"a", href=lambda h: h and not h.strip().lower().startswith("javascript:")
|
||||
)
|
||||
if not link_tag:
|
||||
continue
|
||||
href = link_tag["href"].strip()
|
||||
if "uploads" not in href.lower() and not href.lower().endswith(".pdf"):
|
||||
continue
|
||||
document_url = urljoin(base_url + "/", href)
|
||||
decisions.append(UyusmazlikApiDecisionEntry(
|
||||
esas_sayisi=cells[0].get_text(strip=True) or None,
|
||||
karar_sayisi=cells[1].get_text(strip=True) or None,
|
||||
karar_tarihi=cells[2].get_text(strip=True) or None,
|
||||
document_url=document_url,
|
||||
))
|
||||
|
||||
# Try to read a "N kayıt/sonuç/karar bulundu" style count if present.
|
||||
total_records: Optional[int] = None
|
||||
count_match = re.search(r'(\d+)\s*(?:adet\s*)?(?:kayıt|sonuç|karar)\b', html_content, re.IGNORECASE)
|
||||
if count_match:
|
||||
total_records = int(count_match.group(1))
|
||||
|
||||
return UyusmazlikSearchResponse(decisions=decisions, total_records_found=total_records)
|
||||
|
||||
async def search_decisions(self, params: UyusmazlikSearchRequest) -> UyusmazlikSearchResponse:
|
||||
# 1. Load the landing page to obtain a fresh viewstate + session cookie.
|
||||
landing = await self.http_client.get(self.SEARCH_PATH)
|
||||
landing.raise_for_status()
|
||||
form_data = self._extract_hidden_fields(landing.text)
|
||||
|
||||
# 2. Submit the search form.
|
||||
form_data.update({
|
||||
"txtSearch": params.icerik or "",
|
||||
"rblSearchScope": params.search_scope,
|
||||
"btnSearch": "Ara",
|
||||
})
|
||||
if params.case_sensitive:
|
||||
form_data["chkCaseSensitive"] = "on"
|
||||
|
||||
logger.info("UyusmazlikApiClient: search icerik=%r scope=%s page=%s",
|
||||
params.icerik, params.search_scope, params.page_number)
|
||||
response = await self.http_client.post(
|
||||
self.SEARCH_PATH,
|
||||
data=form_data,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
||||
)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
|
||||
# 3. Navigate the GridView pager if a later page is requested.
|
||||
if params.page_number > 1:
|
||||
page_fields = self._extract_hidden_fields(html_content)
|
||||
page_fields.update({
|
||||
"txtSearch": params.icerik or "",
|
||||
"rblSearchScope": params.search_scope,
|
||||
"__EVENTTARGET": "GridView1",
|
||||
"__EVENTARGUMENT": f"Page${params.page_number}",
|
||||
})
|
||||
if params.case_sensitive:
|
||||
page_fields["chkCaseSensitive"] = "on"
|
||||
page_response = await self.http_client.post(
|
||||
self.SEARCH_PATH,
|
||||
data=page_fields,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
||||
)
|
||||
page_response.raise_for_status()
|
||||
html_content = page_response.text
|
||||
|
||||
return self._parse_results(html_content, self.BASE_URL)
|
||||
|
||||
def _convert_pdf_to_markdown(self, pdf_bytes: bytes) -> Optional[str]:
|
||||
try:
|
||||
pdf_stream = io.BytesIO(pdf_bytes)
|
||||
conversion_result = MarkItDown().convert(pdf_stream, file_extension=".pdf")
|
||||
return conversion_result.text_content
|
||||
except Exception as e:
|
||||
logger.error("UyusmazlikApiClient: PDF to Markdown conversion error: %s", e)
|
||||
return None
|
||||
|
||||
processed_html = html.unescape(full_decision_html_content)
|
||||
# As per user request, pass the full (unescaped) HTML to MarkItDown
|
||||
html_input_for_markdown = processed_html
|
||||
|
||||
markdown_text = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown()
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_file:
|
||||
tmp_file.write(html_input_for_markdown)
|
||||
temp_file_path = tmp_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
markdown_text = conversion_result.text_content
|
||||
logger.info("UyusmazlikApiClient: HTML to Markdown conversion successful.")
|
||||
except Exception as e:
|
||||
logger.error(f"UyusmazlikApiClient: Error during MarkItDown HTML to Markdown conversion: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path)
|
||||
return markdown_text
|
||||
|
||||
async def get_decision_document_as_markdown(self, document_url: str) -> UyusmazlikDocumentMarkdown:
|
||||
"""
|
||||
Retrieves a specific Uyuşmazlık decision from its full URL and returns content as Markdown.
|
||||
"""
|
||||
logger.info(f"UyusmazlikApiClient (httpx for docs): Fetching Uyuşmazlık document for Markdown from URL: {document_url}")
|
||||
"""Fetch an Uyuşmazlık decision PDF and return its content as Markdown."""
|
||||
logger.info("UyusmazlikApiClient: Fetching document PDF from %s", document_url)
|
||||
try:
|
||||
# Using a new httpx.AsyncClient instance for this GET request for simplicity
|
||||
async with httpx.AsyncClient(verify=False, timeout=self.request_timeout) as doc_fetch_client:
|
||||
|
||||
get_response = await doc_fetch_client.get(document_url, headers={"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"})
|
||||
get_response.raise_for_status()
|
||||
html_content_from_api = get_response.text
|
||||
|
||||
if not isinstance(html_content_from_api, str) or not html_content_from_api.strip():
|
||||
logger.warning(f"UyusmazlikApiClient: Received empty or non-string HTML from URL {document_url}.")
|
||||
return UyusmazlikDocumentMarkdown(source_url=document_url, markdown_content=None)
|
||||
|
||||
markdown_content = self._convert_html_to_markdown_uyusmazlik(html_content_from_api)
|
||||
response = await self.http_client.get(
|
||||
document_url,
|
||||
headers={"Accept": "application/pdf,*/*"},
|
||||
)
|
||||
response.raise_for_status()
|
||||
markdown_content = await asyncio.to_thread(self._convert_pdf_to_markdown, response.content)
|
||||
return UyusmazlikDocumentMarkdown(source_url=document_url, markdown_content=markdown_content)
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"UyusmazlikApiClient (httpx for docs): HTTP error fetching Uyuşmazlık document from {document_url}: {e}")
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error(f"UyusmazlikApiClient (httpx for docs): General error processing Uyuşmazlık document from {document_url}: {e}")
|
||||
except httpx.HTTPError as e:
|
||||
logger.error("UyusmazlikApiClient: HTTP error fetching document from %s: %s", document_url, e)
|
||||
raise
|
||||
|
||||
async def close_client_session(self):
|
||||
|
||||
logger.info("UyusmazlikApiClient: No persistent client session from __init__ to close.")
|
||||
if hasattr(self, "http_client") and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("UyusmazlikApiClient: HTTP client session closed.")
|
||||
|
||||
@@ -1,86 +1,41 @@
|
||||
# uyusmazlik_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field, HttpUrl
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
from typing import List, Optional, Literal
|
||||
|
||||
# Enum definitions for user-friendly input based on the provided HTML form
|
||||
class UyusmazlikBolumEnum(str, Enum):
|
||||
"""User-friendly names for 'BolumId'."""
|
||||
TUMU = "ALL" # Represents "...Seçiniz..." or all
|
||||
CEZA_BOLUMU = "Ceza Bölümü"
|
||||
GENEL_KURUL_KARARLARI = "Genel Kurul Kararları"
|
||||
HUKUK_BOLUMU = "Hukuk Bölümü"
|
||||
# The Uyuşmazlık Mahkemesi search site was rebuilt as an ASP.NET WebForms app.
|
||||
# It now offers only a single free-text search with a scope selector; the old
|
||||
# Bölüm / Uyuşmazlık Türü / Karar Sonucu / Esas-Karar year filters no longer exist.
|
||||
|
||||
class UyusmazlikTuruEnum(str, Enum):
|
||||
"""User-friendly names for 'UyusmazlikId'."""
|
||||
TUMU = "ALL" # Represents "...Seçiniz..." or all
|
||||
GOREV_UYUSMAZLIGI = "Görev Uyuşmazlığı"
|
||||
HUKUM_UYUSMAZLIGI = "Hüküm Uyuşmazlığı"
|
||||
UyusmazlikSearchScope = Literal["All", "EsasNo", "KararNo"]
|
||||
|
||||
class UyusmazlikKararSonucuEnum(str, Enum): # Based on checkbox text in the form
|
||||
"""User-friendly names for 'KararSonucuList' items."""
|
||||
HUKUM_UYUSMAZLIGI_OLMADIGINA_DAIR = "Hüküm Uyuşmazlığı Olmadığına Dair"
|
||||
HUKUM_UYUSMAZLIGI_OLDUGUNA_DAIR = "Hüküm Uyuşmazlığı Olduğuna Dair"
|
||||
# Add other "Karar Sonucu" options from the form's checkboxes as Enum members
|
||||
# Example: GOREVLI_YARGI_YERI_ADLI = "Görevli Yargı Yeri Belirlenmesine Dair (Adli Yargı)"
|
||||
# The client will map these enum values (which are strings) to their respective IDs.
|
||||
|
||||
class UyusmazlikSearchRequest(BaseModel): # This is the model the MCP tool will accept
|
||||
"""Model for Uyuşmazlık Mahkemesi search request using user-friendly terms."""
|
||||
icerik: Optional[str] = Field("", description="Keyword or content for main text search (Icerik).")
|
||||
|
||||
bolum: Optional[UyusmazlikBolumEnum] = Field(
|
||||
UyusmazlikBolumEnum.TUMU,
|
||||
description="Select the department (Bölüm)."
|
||||
)
|
||||
uyusmazlik_turu: Optional[UyusmazlikTuruEnum] = Field(
|
||||
UyusmazlikTuruEnum.TUMU,
|
||||
description="Select the type of dispute (Uyuşmazlık)."
|
||||
class UyusmazlikSearchRequest(BaseModel):
|
||||
"""Model for the Uyuşmazlık Mahkemesi search request."""
|
||||
icerik: str = Field("", description="Search text (txtSearch).")
|
||||
search_scope: UyusmazlikSearchScope = Field(
|
||||
"All",
|
||||
description="Search scope: 'All' (full text), 'EsasNo' (by case number), 'KararNo' (by decision number).",
|
||||
)
|
||||
case_sensitive: bool = Field(False, description="Whether the search is case sensitive (chkCaseSensitive).")
|
||||
page_number: int = Field(1, ge=1, description="Result page number (GridView pager).")
|
||||
|
||||
# User provides a list of user-friendly names for Karar Sonucu
|
||||
karar_sonuclari: Optional[List[UyusmazlikKararSonucuEnum]] = Field( # Changed to list of Enums
|
||||
default_factory=list,
|
||||
description="List of desired 'Karar Sonucu' types."
|
||||
)
|
||||
|
||||
esas_yil: Optional[str] = Field("", description="Case year ('Esas Yılı').")
|
||||
esas_sayisi: Optional[str] = Field("", description="Case number ('Esas Sayısı').")
|
||||
karar_yil: Optional[str] = Field("", description="Decision year ('Karar Yılı').")
|
||||
karar_sayisi: Optional[str] = Field("", description="Decision number ('Karar Sayısı').")
|
||||
kanun_no: Optional[str] = Field("", description="Relevant Law Number ('KanunNo').")
|
||||
|
||||
karar_date_begin: Optional[str] = Field("", description="Decision start date (DD.MM.YYYY) ('KararDateBegin').")
|
||||
karar_date_end: Optional[str] = Field("", description="Decision end date (DD.MM.YYYY) ('KararDateEnd').")
|
||||
|
||||
resmi_gazete_sayi: Optional[str] = Field("", description="Official Gazette number ('ResmiGazeteSayi').")
|
||||
resmi_gazete_date: Optional[str] = Field("", description="Official Gazette date (DD.MM.YYYY) ('ResmiGazeteDate').")
|
||||
|
||||
# Detailed text search fields from the "icerikDetail" section of the form
|
||||
tumce: Optional[str] = Field("", description="Exact phrase search ('Tumce').")
|
||||
wild_card: Optional[str] = Field("", description="Search for phrase and its inflections ('WildCard').") # Changed from WildCard for Pythonic name
|
||||
hepsi: Optional[str] = Field("", description="Search for texts containing all specified words ('Hepsi').")
|
||||
herhangi_birisi: Optional[str] = Field("", description="Search for texts containing any of the specified words ('Herhangibirisi').")
|
||||
not_hepsi: Optional[str] = Field("", description="Exclude texts containing these specified words ('NotHepsi').")
|
||||
|
||||
class UyusmazlikApiDecisionEntry(BaseModel):
|
||||
"""Model for an individual decision entry parsed from Uyuşmazlık API's HTML search response."""
|
||||
karar_sayisi: Optional[str] = Field(None)
|
||||
esas_sayisi: Optional[str] = Field(None)
|
||||
bolum: Optional[str] = Field(None)
|
||||
uyusmazlik_konusu: Optional[str] = Field(None)
|
||||
karar_sonucu: Optional[str] = Field(None)
|
||||
popover_content: Optional[str] = Field(None, description="Summary/description from popover.")
|
||||
document_url: HttpUrl # Full URL to the decision document HTML page
|
||||
pdf_url: Optional[HttpUrl] = Field(None, description="Direct URL to PDF if available.")
|
||||
"""A single decision row parsed from the Uyuşmazlık GridView results."""
|
||||
esas_sayisi: Optional[str] = Field(None, description="Case number (Esas No).")
|
||||
karar_sayisi: Optional[str] = Field(None, description="Decision number (Karar No).")
|
||||
karar_tarihi: Optional[str] = Field(None, description="Decision date (DD/MM/YYYY).")
|
||||
document_url: HttpUrl = Field(..., description="Full URL to the decision PDF document.")
|
||||
|
||||
class UyusmazlikSearchResponse(BaseModel): # This is what the MCP tool will return
|
||||
"""Response model for Uyuşmazlık Mahkemesi search results for the MCP tool."""
|
||||
|
||||
class UyusmazlikSearchResponse(BaseModel):
|
||||
"""Response model for Uyuşmazlık Mahkemesi search results."""
|
||||
decisions: List[UyusmazlikApiDecisionEntry]
|
||||
total_records_found: Optional[int] = Field(None, description="Total number of records found for the query, if available.")
|
||||
total_records_found: Optional[int] = Field(None, description="Total number of records found, if reported.")
|
||||
|
||||
|
||||
class UyusmazlikDocumentMarkdown(BaseModel):
|
||||
"""Model for an Uyuşmazlık decision document, containing only Markdown content."""
|
||||
source_url: HttpUrl # The URL from which the content was fetched
|
||||
markdown_content: Optional[str] = Field(None, description="The decision content converted to Markdown.")
|
||||
"""Model for an Uyuşmazlık decision document, containing Markdown content."""
|
||||
source_url: HttpUrl
|
||||
markdown_content: Optional[str] = Field(None, description="The decision PDF content converted to Markdown.")
|
||||
|
||||
@@ -1,13 +1,13 @@
|
||||
# yargitay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
|
||||
from typing import Dict, Any, List, Optional
|
||||
import logging
|
||||
import html
|
||||
import re
|
||||
import tempfile
|
||||
import os
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
@@ -66,6 +66,19 @@ class YargitayOfficialApiClient:
|
||||
response.raise_for_status() # Raise an exception for HTTP 4xx or 5xx status codes
|
||||
response_json_data = response.json()
|
||||
|
||||
logger.debug(f"YargitayOfficialApiClient: Raw API response: {response_json_data}")
|
||||
|
||||
# Handle None or empty data response from API
|
||||
if response_json_data is None:
|
||||
logger.warning("YargitayOfficialApiClient: API returned None response")
|
||||
response_json_data = {"data": {"data": [], "recordsTotal": 0, "recordsFiltered": 0}}
|
||||
elif not isinstance(response_json_data, dict):
|
||||
logger.warning(f"YargitayOfficialApiClient: API returned unexpected response type: {type(response_json_data)}")
|
||||
response_json_data = {"data": {"data": [], "recordsTotal": 0, "recordsFiltered": 0}}
|
||||
elif response_json_data.get("data") is None:
|
||||
logger.warning("YargitayOfficialApiClient: API response data field is None")
|
||||
response_json_data["data"] = {"data": [], "recordsTotal": 0, "recordsFiltered": 0}
|
||||
|
||||
# Validate and parse the response using Pydantic models
|
||||
api_response = YargitayApiSearchResponse(**response_json_data)
|
||||
|
||||
@@ -108,25 +121,20 @@ class YargitayOfficialApiClient:
|
||||
html_to_convert = processed_html
|
||||
|
||||
markdown_output = None
|
||||
temp_file_path = None
|
||||
try:
|
||||
md_converter = MarkItDown() # Plugins disabled as per basic usage
|
||||
# Convert HTML string to bytes and create BytesIO stream
|
||||
html_bytes = html_to_convert.encode('utf-8')
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
|
||||
# Write the HTML to a temporary file for MarkItDown to process
|
||||
with tempfile.NamedTemporaryFile(mode="w", delete=False, suffix=".html", encoding="utf-8") as tmp_html_file:
|
||||
tmp_html_file.write(html_to_convert)
|
||||
temp_file_path = tmp_html_file.name
|
||||
|
||||
conversion_result = md_converter.convert(temp_file_path)
|
||||
# Pass BytesIO stream to MarkItDown to avoid temp file creation
|
||||
md_converter = MarkItDown()
|
||||
conversion_result = md_converter.convert(html_stream)
|
||||
markdown_output = conversion_result.text_content
|
||||
|
||||
logger.info("Successfully converted HTML to Markdown.")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Error during MarkItDown HTML to Markdown conversion: {e}")
|
||||
finally:
|
||||
if temp_file_path and os.path.exists(temp_file_path):
|
||||
os.remove(temp_file_path) # Clean up the temporary file
|
||||
|
||||
return markdown_output
|
||||
|
||||
@@ -152,7 +160,7 @@ class YargitayOfficialApiClient:
|
||||
logger.error(f"YargitayOfficialApiClient: 'data' field in API response is not a string or not found (ID: {id}).")
|
||||
raise ValueError("Expected HTML content not found in API response's 'data' field.")
|
||||
|
||||
markdown_content = self._convert_html_to_markdown(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, html_content_from_api)
|
||||
|
||||
return YargitayDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
@@ -34,111 +34,70 @@ class YargitayDetailedSearchRequest(BaseModel):
|
||||
to Yargitay's detailed search endpoint (e.g., /aramadetaylist).
|
||||
Based on the payload provided by the user.
|
||||
"""
|
||||
arananKelime: Optional[str] = Field("", description="""Keyword to search for with advanced operators support:
|
||||
• Simple words: 'arsa payı' (OR logic - finds documents with ANY word)
|
||||
• Exact phrases: '"arsa payı"' (finds exact phrase)
|
||||
• AND logic: 'arsa+payı' (both words required)
|
||||
• Wildcards: 'bozma*' (matches bozma, bozması, bozmanın, etc.)
|
||||
• Multiple required: '+"arsa payı" +"bozma sebebi"'
|
||||
• Exclusion: '+"arsa payı" -"inşaat sözleşmesi"'
|
||||
Examples: arsa payı | "arsa payı" | +"mülkiyet hakkı" +"bozma sebebi" | hukuk*""")
|
||||
arananKelime: Optional[str] = Field("", description="Turkish keywords (supports +word -word \"phrase\" operators)")
|
||||
# Department/Board selection - Complete Court of Cassation chamber hierarchy
|
||||
birimYrgKurulDaire: Optional[str] = Field("ALL", description="""
|
||||
Court of Cassation (Yargıtay) chamber/board selection. Options include:
|
||||
- 'ALL' for all chambers
|
||||
- Civil: 'Civil General Assembly (Hukuk Genel Kurulu)', '1st Civil Chamber (1. Hukuk Dairesi)' through '23rd Civil Chamber (23. Hukuk Dairesi)', 'Civil Chambers Presidents Board (Hukuk Daireleri Başkanlar Kurulu)'
|
||||
- Criminal: 'Criminal General Assembly (Ceza Genel Kurulu)', '1st Criminal Chamber (1. Ceza Dairesi)' through '23rd Criminal Chamber (23. Ceza Dairesi)', 'Criminal Chambers Presidents Board (Ceza Daireleri Başkanlar Kurulu)'
|
||||
- General: 'Grand General Assembly (Büyük Genel Kurulu)'
|
||||
Total: 52 possible values (including 'ALL' for all chambers)
|
||||
""")
|
||||
birimYrgHukukDaire: Optional[str] = Field("", description="Legacy field - use birimYrgKurulDaire instead for chamber selection")
|
||||
birimYrgCezaDaire: Optional[str] = Field("", description="Legacy field - use birimYrgKurulDaire instead for chamber selection")
|
||||
birimYrgKurulDaire: Optional[str] = Field("ALL", description="Chamber (ALL or specific chamber name)")
|
||||
|
||||
esasYil: Optional[str] = Field("", description="""Case year for 'Esas No' filtering.
|
||||
Format: YYYY (e.g., '2024')
|
||||
Use with sequence numbers for precise case targeting""")
|
||||
esasIlkSiraNo: Optional[str] = Field("", description="""Starting sequence number for 'Esas No' range filtering.
|
||||
Format: numeric string (e.g., '1', '100')
|
||||
Use with esasSonSiraNo for range: cases 100-200 in specified year""")
|
||||
esasSonSiraNo: Optional[str] = Field("", description="""Ending sequence number for 'Esas No' range filtering.
|
||||
Format: numeric string (e.g., '500', '1000')
|
||||
Creates range from esasIlkSiraNo to this number""")
|
||||
esasYil: Optional[str] = Field("", description="Case year (YYYY)")
|
||||
esasIlkSiraNo: Optional[str] = Field("", description="Start case no")
|
||||
esasSonSiraNo: Optional[str] = Field("", description="End case no")
|
||||
|
||||
kararYil: Optional[str] = Field("", description="""Decision year for 'Karar No' filtering.
|
||||
Format: YYYY (e.g., '2024')
|
||||
Filters decisions by the year they were issued""")
|
||||
kararIlkSiraNo: Optional[str] = Field("", description="""Starting sequence number for 'Karar No' range filtering.
|
||||
Format: numeric string (e.g., '1', '50')
|
||||
Use with kararSonSiraNo for decision number ranges""")
|
||||
kararSonSiraNo: Optional[str] = Field("", description="""Ending sequence number for 'Karar No' range filtering.
|
||||
Format: numeric string (e.g., '100', '500')
|
||||
Creates range from kararIlkSiraNo to this number""")
|
||||
kararYil: Optional[str] = Field("", description="Decision year (YYYY)")
|
||||
kararIlkSiraNo: Optional[str] = Field("", description="Start decision no")
|
||||
kararSonSiraNo: Optional[str] = Field("", description="End decision no")
|
||||
|
||||
baslangicTarihi: Optional[str] = Field("", description="""Start date for decision search.
|
||||
Format: DD.MM.YYYY (e.g., '01.01.2024')
|
||||
Use with bitisTarihi for date range filtering
|
||||
Examples: '01.01.2024', '15.06.2023'""")
|
||||
bitisTarihi: Optional[str] = Field("", description="""End date for decision search.
|
||||
Format: DD.MM.YYYY (e.g., '31.12.2024')
|
||||
Creates date range from baslangicTarihi to this date
|
||||
Examples: '31.12.2024', '30.06.2023'""")
|
||||
baslangicTarihi: Optional[str] = Field("", description="Start date (DD.MM.YYYY)")
|
||||
bitisTarihi: Optional[str] = Field("", description="End date (DD.MM.YYYY)")
|
||||
|
||||
siralama: Optional[str] = Field("3", description="""Sorting criteria for search results:
|
||||
• '1': Esas No (Case Number) - sorts by case registration order
|
||||
• '2': Karar No (Decision Number) - sorts by decision issuance order
|
||||
• '3': Karar Tarihi (Decision Date) - sorts by chronological order [DEFAULT]
|
||||
Recommended: Use '3' for most recent decisions first""")
|
||||
siralamaDirection: Optional[str] = Field("desc", description="""Sorting direction for results:
|
||||
• 'desc': Descending order (newest/highest first) [DEFAULT]
|
||||
• 'asc': Ascending order (oldest/lowest first)
|
||||
Most common: 'desc' for latest decisions first""")
|
||||
|
||||
pageSize: int = Field(10, ge=1, le=100, description="""Number of results per page.
|
||||
Range: 1-100 results per page
|
||||
Recommended: 10-50 for balanced performance and coverage
|
||||
Large values (50-100) for comprehensive analysis""")
|
||||
pageNumber: int = Field(1, ge=1, description="""Page number to retrieve (1-indexed).
|
||||
Start with 1 for first page
|
||||
Use with pageSize to navigate through large result sets
|
||||
Example: pageSize=50, pageNumber=3 gets results 101-150""")
|
||||
pageSize: int = Field(10, ge=1, le=10, description="Results per page (1-100)")
|
||||
pageNumber: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
|
||||
class YargitayApiDecisionEntry(BaseModel):
|
||||
"""Model for an individual decision entry from the Yargitay API search response."""
|
||||
id: str # Unique system ID of the decision
|
||||
daire: Optional[str] = Field(None, description="The chamber (Daire) that made the decision.")
|
||||
esasNo: Optional[str] = Field(None, alias="esasNo", description="Case registry number (Esas No).")
|
||||
kararNo: Optional[str] = Field(None, alias="kararNo", description="Decision number (Karar No).")
|
||||
kararTarihi: Optional[str] = Field(None, alias="kararTarihi", description="Date of the decision (Karar Tarihi).")
|
||||
arananKelime: Optional[str] = Field(None, alias="arananKelime", description="Matched keyword (Aranan Kelime) in the search result item.")
|
||||
daire: Optional[str] = Field(None, description="Chamber")
|
||||
esasNo: Optional[str] = Field(None, alias="esasNo", description="Case no")
|
||||
kararNo: Optional[str] = Field(None, alias="kararNo", description="Decision no")
|
||||
kararTarihi: Optional[str] = Field(None, alias="kararTarihi", description="Date")
|
||||
# 'index' and 'siraNo' from API response are not critical for MCP tool, so omitted for brevity
|
||||
|
||||
# This field will be populated by the client after fetching the search list
|
||||
document_url: Optional[HttpUrl] = Field(None, description="Direct URL (Belge URL) to the decision document.")
|
||||
document_url: Optional[HttpUrl] = Field(None, description="Document URL")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True) # To allow populating by alias from API response
|
||||
|
||||
|
||||
class YargitayApiResponseInnerData(BaseModel):
|
||||
"""Model for the inner 'data' object in the Yargitay API search response."""
|
||||
data: List[YargitayApiDecisionEntry]
|
||||
data: List[YargitayApiDecisionEntry] = Field(default_factory=list)
|
||||
# draw: Optional[int] = None # Typically used by DataTables, not essential for MCP
|
||||
recordsTotal: int # Total number of records matching the query
|
||||
recordsFiltered: int # Total number of records after filtering (usually same as recordsTotal)
|
||||
recordsTotal: int = Field(default=0) # Total number of records matching the query
|
||||
recordsFiltered: int = Field(default=0) # Total number of records after filtering (usually same as recordsTotal)
|
||||
|
||||
class YargitayApiSearchResponse(BaseModel):
|
||||
"""Model for the complete search response from the Yargitay API."""
|
||||
data: YargitayApiResponseInnerData
|
||||
data: Optional[YargitayApiResponseInnerData] = Field(default_factory=lambda: YargitayApiResponseInnerData())
|
||||
# metadata: Optional[Dict[str, Any]] = None # Optional metadata from API
|
||||
|
||||
class YargitayDocumentMarkdown(BaseModel):
|
||||
"""Model for a Yargitay decision document, containing only Markdown content."""
|
||||
id: str = Field(..., description="The unique ID (Belge Kimliği) of the document.")
|
||||
markdown_content: Optional[str] = Field(None, description="The decision content (Karar İçeriği) converted to Markdown.")
|
||||
source_url: HttpUrl = Field(..., description="The source URL (Kaynak URL) of the original document.")
|
||||
id: str = Field(..., description="Document ID")
|
||||
markdown_content: Optional[str] = Field(None, description="Content")
|
||||
source_url: HttpUrl = Field(..., description="Source URL")
|
||||
|
||||
class CleanYargitayDecisionEntry(BaseModel):
|
||||
"""Clean decision entry without arananKelime field to reduce token usage."""
|
||||
id: str
|
||||
daire: Optional[str] = Field(None, description="Chamber")
|
||||
esasNo: Optional[str] = Field(None, description="Case no")
|
||||
kararNo: Optional[str] = Field(None, description="Decision no")
|
||||
kararTarihi: Optional[str] = Field(None, description="Date")
|
||||
document_url: Optional[HttpUrl] = Field(None, description="Document URL")
|
||||
|
||||
class CompactYargitaySearchResult(BaseModel):
|
||||
"""A more compact search result model for the MCP tool to return."""
|
||||
decisions: List[YargitayApiDecisionEntry]
|
||||
decisions: List[CleanYargitayDecisionEntry]
|
||||
total_records: int
|
||||
requested_page: int
|
||||
page_size: int
|
||||
Reference in New Issue
Block a user