Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a062237474 | ||
|
|
b69eda77af | ||
|
|
3927dcee8f | ||
|
|
3768104679 | ||
|
|
aa580ffafc | ||
|
|
931eb3ca8f | ||
|
|
d258ad2375 | ||
|
|
061887f870 | ||
|
|
ac611f840c | ||
|
|
c938f10ba2 | ||
|
|
1356c4d020 | ||
|
|
5392435c7a | ||
|
|
96a5a538b2 | ||
|
|
26aa3dacc6 | ||
|
|
58457b076f | ||
|
|
4521e1de85 | ||
|
|
8f04010c57 | ||
|
|
7ed9c25687 | ||
|
|
4fdc7a3689 | ||
|
|
1538a4c145 | ||
|
|
a24def2e66 | ||
|
|
6b781b61d2 | ||
|
|
fb29146755 | ||
|
|
42731a2c03 | ||
|
|
ae5d590cca | ||
|
|
ee544dc603 | ||
|
|
355f505da9 | ||
|
|
4c06a5926b | ||
|
|
2cec4dccd6 | ||
|
|
5c2e9cc92b | ||
|
|
a66a3f2053 | ||
|
|
a4d9e2e53d | ||
|
|
036e49a928 | ||
|
|
7f78f87508 | ||
|
|
d8805cb93b | ||
|
|
28ff2e39a5 | ||
|
|
fd08637ca2 | ||
|
|
a5e6baeec8 | ||
|
|
8818a7809a | ||
|
|
12d51e3735 | ||
|
|
efe962abf1 | ||
|
|
1d73265f10 | ||
|
|
f1d3b60efb | ||
|
|
93e64bc1fc | ||
|
|
77e2748ade | ||
|
|
da146cf3ec | ||
|
|
e771c5b3c5 | ||
|
|
1223b37adb | ||
|
|
e26f09aced | ||
|
|
ae5bae2f4a | ||
|
|
a2b50951e9 | ||
|
|
91ad04cf09 | ||
|
|
5cec0df785 | ||
|
|
d51f11c7ba | ||
|
|
b207b16ef7 | ||
|
|
f47147ba44 | ||
|
|
4d57a3939f | ||
|
|
50c6963eee | ||
|
|
def7e7d65e | ||
|
|
82a0d13d25 | ||
|
|
18b552ca2f | ||
|
|
1e96b1888e | ||
|
|
815786a09d | ||
|
|
260adb3ac9 | ||
|
|
7164205425 | ||
|
|
3961a23d3a | ||
|
|
25723f070f | ||
|
|
6376037ccf | ||
|
|
d1728ce114 | ||
|
|
69b5da5cef | ||
|
|
91564bf0a1 | ||
|
|
6f94eca33c | ||
|
|
4122790821 | ||
|
|
0f5bae8bb1 | ||
|
|
4d7da0d3ba | ||
|
|
4f48681b09 | ||
|
|
b401bad890 | ||
|
|
0a80bc535b | ||
|
|
b1da034ea9 | ||
|
|
e900bc03dd | ||
|
|
54f81e18f0 | ||
|
|
4e18e792c5 | ||
|
|
4a3edef287 | ||
|
|
e2ca844ab9 | ||
|
|
a49d0859ea | ||
|
|
a7877f34f4 | ||
|
|
364f3761d7 | ||
|
|
673f996f5f | ||
|
|
90a7a23064 | ||
|
|
443657f9e2 |
@@ -181,6 +181,3 @@ site
|
||||
|
||||
# Production logs
|
||||
**/logs/*.log.*
|
||||
**/Dockerfile
|
||||
**/Dockerfile
|
||||
fly.toml
|
||||
|
||||
@@ -70,6 +70,50 @@ JWT_SECRET_KEY=your_jwt_secret_key_here
|
||||
# MAX_REQUESTS_PER_MINUTE=60
|
||||
# BURST_CAPACITY=20
|
||||
|
||||
# =============================================================================
|
||||
# SEMANTIC SEARCH SETTINGS (Optional)
|
||||
# =============================================================================
|
||||
|
||||
# Embedding provider for the semantic_search tool.
|
||||
# Pick exactly one of: OpenRouter (hosted) or Local (your own server).
|
||||
|
||||
# --- Option A: OpenRouter (hosted, default) -----------------------------------
|
||||
# Get your API key from: https://openrouter.ai/keys
|
||||
# If neither this nor EMBEDDING_PROVIDER=local is set, semantic search is off.
|
||||
OPENROUTER_API_KEY=sk-or-v1-your_openrouter_api_key_here
|
||||
|
||||
# Optional: override the OpenRouter embedding model and dimension.
|
||||
# Defaults: google/gemini-embedding-001 at 3072 dims (paid on OpenRouter).
|
||||
# Pick any model from https://openrouter.ai/models?modality=embedding
|
||||
# and set the dimension to that model's output size — they must match.
|
||||
# OPENROUTER_EMBEDDING_MODEL=google/gemini-embedding-001
|
||||
# OPENROUTER_EMBEDDING_DIMENSION=3072
|
||||
|
||||
# --- Option B: Local OpenAI-compatible server (no API key required) ----------
|
||||
# Recommended for Turkish: intfloat/multilingual-e5-large served by HuggingFace
|
||||
# Text Embeddings Inference (TEI). One-line setup:
|
||||
#
|
||||
# docker run -p 8080:80 ghcr.io/huggingface/text-embeddings-inference:latest \
|
||||
# --model-id intfloat/multilingual-e5-large
|
||||
#
|
||||
# Then uncomment the block below. Other model families work too — set
|
||||
# EMBEDDING_PROMPT_STYLE to match: e5 / gemini / raw.
|
||||
#
|
||||
# EMBEDDING_PROVIDER=local
|
||||
# LOCAL_EMBEDDING_BASE_URL=http://localhost:8080/v1
|
||||
# LOCAL_EMBEDDING_MODEL=intfloat/multilingual-e5-large
|
||||
# LOCAL_EMBEDDING_DIMENSION=1024
|
||||
# EMBEDDING_PROMPT_STYLE=e5
|
||||
# LOCAL_EMBEDDING_API_KEY= # most local servers ignore this
|
||||
#
|
||||
# Ollama fallback (if you prefer Ollama and don't need top Turkish quality):
|
||||
# ollama serve && ollama pull nomic-embed-text
|
||||
# EMBEDDING_PROVIDER=local
|
||||
# LOCAL_EMBEDDING_BASE_URL=http://localhost:11434/v1
|
||||
# LOCAL_EMBEDDING_MODEL=nomic-embed-text
|
||||
# LOCAL_EMBEDDING_DIMENSION=768
|
||||
# EMBEDDING_PROMPT_STYLE=raw # nomic uses its own search_query/search_document
|
||||
|
||||
# =============================================================================
|
||||
# USAGE INSTRUCTIONS
|
||||
# =============================================================================
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
# Serena
|
||||
.serena/
|
||||
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
@@ -213,3 +216,4 @@ measure_mcp_directly.py
|
||||
playwright_mcp_overhead.json
|
||||
simple_test.py
|
||||
analyze_anayasa_html.py
|
||||
CLAUDE.md
|
||||
|
||||
@@ -72,7 +72,10 @@ uv run simple_test.py # Simple individual tool test
|
||||
uv run measure_mcp_directly.py # Measure MCP token overhead
|
||||
|
||||
# Test KVKK module (uses fallback token if BRAVE_API_TOKEN not set)
|
||||
python test_kvkk_module.py # Test KVKK search and document retrieval
|
||||
python test_kvkv_module.py # Test KVKK search and document retrieval
|
||||
|
||||
# Test KİK v2 comprehensive functionality
|
||||
uv run test_kik_v2_comprehensive.py # Test all three KİK v2 decision types (uyusmazlik, duzenleyici, mahkeme)
|
||||
|
||||
# MCP Optimization Testing
|
||||
uv run test_core_tools_quick.py # Verify all tools work after optimization
|
||||
@@ -182,7 +185,7 @@ This MCP server has undergone comprehensive optimization to minimize token overh
|
||||
4. **emsal_mcp_module**: Emsal (UYAP precedent) decisions
|
||||
5. **uyusmazlik_mcp_module**: Uyuşmazlık Mahkemesi (Jurisdictional Disputes Court)
|
||||
6. **anayasa_mcp_module**: Constitutional Court (both norm control and individual applications)
|
||||
7. **kik_mcp_module**: KİK (Public Procurement Authority) decisions
|
||||
7. **kik_mcp_module**: KİK (Public Procurement Authority) decisions with v2 API support for three decision types (uyusmazlik, duzenleyici, mahkeme)
|
||||
8. **rekabet_mcp_module**: Competition Authority decisions
|
||||
9. **kvkk_mcp_module**: KVKK (Personal Data Protection Authority) decisions - Brave API integration
|
||||
10. **bddk_mcp_module**: BDDK (Banking Regulation and Supervision Agency) decisions - Tavily API integration
|
||||
@@ -466,6 +469,28 @@ doc8 = await get_kvkk_document_markdown(decision_url="https://www.kvkk.gov.tr/Ic
|
||||
- **Fallback Token**: If not set, uses a limited free token automatically
|
||||
- KVKK search tools will work without configuration (with rate limits)
|
||||
|
||||
### Rate Limits
|
||||
|
||||
| API | Rate Limit | Notes |
|
||||
|-----|------------|-------|
|
||||
| Bedesten Unified | ~10 req / 30s window per source IP (measured 2026-05-08); 11th req → HTTP 429 with `Retry-After: 30`. Client uses an internal token bucket (default 1 token, refill 1/3.5s) plus 429 back-pressure (whole bucket pauses for the Retry-After window). Override via `BEDESTEN_RATE_CAPACITY` / `BEDESTEN_RATE_REFILL_S`. |
|
||||
| Yargıtay Primary | Unknown | Official government API |
|
||||
| Danıştay | Unknown | Official government API |
|
||||
| Anayasa Mahkemesi | Unknown | Constitutional Court API |
|
||||
| KİK v2 | Unknown | Public Procurement Authority API |
|
||||
| Rekabet Kurumu | Unknown | Competition Authority API |
|
||||
| Sayıştay | Unknown | Court of Accounts API |
|
||||
| Uyuşmazlık | Unknown | Jurisdictional Disputes Court API |
|
||||
| Emsal | Unknown | UYAP Precedent Database API |
|
||||
| KVKK (Brave) | 1,000/month | Brave Search API free tier limit |
|
||||
| BDDK | Unknown | Banking Regulation API |
|
||||
|
||||
**Recommendations:**
|
||||
- Implement client-side caching for repeated queries
|
||||
- Use pagination parameters to limit result sizes
|
||||
- Space out requests during bulk operations
|
||||
- Consider implementing retry logic with exponential backoff
|
||||
|
||||
### OAuth Authentication Configuration
|
||||
|
||||
The server uses **Clerk JWT tokens** for all authentication. **Cross-origin authentication** is implemented using Bearer JWT tokens as per Clerk's best practices.
|
||||
@@ -1077,15 +1102,21 @@ yargi-mcp
|
||||
|
||||
|
||||
### Date Filtering Format (Bedesten API)
|
||||
All Bedesten API tools support consistent date filtering:
|
||||
- **Format**: ISO 8601 with Z timezone: `YYYY-MM-DDTHH:MM:SS.000Z`
|
||||
All Bedesten API tools support consistent date filtering with **automatic format conversion**:
|
||||
- **Accepted Formats**:
|
||||
- Simple format: `YYYY-MM-DD` (e.g., `2020-01-01`) - **automatically converted**
|
||||
- Full ISO 8601: `YYYY-MM-DDTHH:MM:SS.000Z` (e.g., `2020-01-01T00:00:00.000Z`)
|
||||
- **Automatic Conversion**: ✅ **NEW FEATURE** - Simple dates are automatically converted to ISO 8601 format
|
||||
- Start dates: `2020-01-01` → `2020-01-01T00:00:00.000Z`
|
||||
- End dates: `2020-01-01` → `2020-01-01T23:59:59.999Z`
|
||||
- **Parameters**: `kararTarihiStart` (start date) and `kararTarihiEnd` (end date)
|
||||
- **Usage**: Both parameters are optional, use together for date ranges or single parameter for one-sided filtering
|
||||
- **Examples**:
|
||||
- Single date: `kararTarihiStart="2024-06-25T00:00:00.000Z", kararTarihiEnd="2024-06-25T23:59:59.999Z"`
|
||||
- Year range: `kararTarihiStart="2024-01-01T00:00:00.000Z", kararTarihiEnd="2024-12-31T23:59:59.999Z"`
|
||||
- From date: `kararTarihiStart="2024-01-01T00:00:00.000Z"` (no end date)
|
||||
- Until date: `kararTarihiEnd="2024-12-31T23:59:59.999Z"` (no start date)
|
||||
- **Simple format**: `kararTarihiStart="2024-06-25", kararTarihiEnd="2024-06-25"` ✅ **Works automatically**
|
||||
- **Year range**: `kararTarihiStart="2024-01-01", kararTarihiEnd="2024-12-31"` ✅ **Auto-converted**
|
||||
- **Full ISO format**: `kararTarihiStart="2024-01-01T00:00:00.000Z", kararTarihiEnd="2024-12-31T23:59:59.999Z"`
|
||||
- **From date**: `kararTarihiStart="2024-01-01"` (no end date)
|
||||
- **Until date**: `kararTarihiEnd="2024-12-31"` (no start date)
|
||||
|
||||
### Exact Phrase Search Format (Bedesten API)
|
||||
All Bedesten API tools support two types of phrase searching:
|
||||
@@ -1542,13 +1573,25 @@ This ASGI support transforms the Yargı MCP server into a versatile web service
|
||||
**Authentication**: ✅ **OAuth 2.0 + Bearer JWT** - Cross-origin authentication working
|
||||
**Last Updated**: 2025-01-21 - All critical issues resolved
|
||||
|
||||
**Free Deployment**: `https://yargi-mcp-free.fly.dev` - No authentication required
|
||||
**Status**: ✅ **OPERATIONAL** - Authorization disabled for open access
|
||||
**Use Case**: Development, testing, and open-source usage without OAuth setup
|
||||
|
||||
#### Live Production Endpoints
|
||||
|
||||
**Authenticated Deployment (api.yargimcp.com)**:
|
||||
- **Health Check**: https://api.yargimcp.com/health
|
||||
- **OAuth Login**: https://api.yargimcp.com/auth/login
|
||||
- **MCP Endpoint (HTTP)**: https://api.yargimcp.com/mcp/
|
||||
- **MCP Endpoint (SSE)**: https://api.yargimcp.com/sse/
|
||||
- **OAuth Discovery**: https://api.yargimcp.com/.well-known/oauth-authorization-server
|
||||
|
||||
**Free Deployment (yargi-mcp-free.fly.dev)**:
|
||||
- **Health Check**: https://yargi-mcp-free.fly.dev/health
|
||||
- **MCP Endpoint (HTTP)**: https://yargi-mcp-free.fly.dev/mcp/
|
||||
- **MCP Endpoint (SSE)**: https://yargi-mcp-free.fly.dev/sse/
|
||||
- **Direct Access**: No authentication required - immediate usage
|
||||
|
||||
### Redis Configuration (Fly.io Native Upstash) ✅
|
||||
|
||||
The server uses Fly.io's native Upstash Redis integration for OAuth session storage:
|
||||
@@ -1631,6 +1674,8 @@ ENABLE_AUTH=true
|
||||
```
|
||||
|
||||
#### MCP Connection Details for Claude AI
|
||||
|
||||
**Authenticated Production (api.yargimcp.com)**:
|
||||
```
|
||||
MCP Server URL (HTTP): https://api.yargimcp.com/mcp/
|
||||
MCP Server URL (SSE): https://api.yargimcp.com/sse/
|
||||
@@ -1640,6 +1685,15 @@ Authentication: OAuth 2.0 with PKCE + JWT tokens + Bearer JWT (optional)
|
||||
Transports: HTTP (Streamable) + SSE (Server-Sent Events)
|
||||
```
|
||||
|
||||
**Free Open Access (yargi-mcp-free.fly.dev)**:
|
||||
```
|
||||
MCP Server URL (HTTP): https://yargi-mcp-free.fly.dev/mcp/
|
||||
MCP Server URL (SSE): https://yargi-mcp-free.fly.dev/sse/
|
||||
Authentication: None - Direct access
|
||||
Transports: HTTP (Streamable) + SSE (Server-Sent Events)
|
||||
Use Case: Development, testing, immediate usage without OAuth setup
|
||||
```
|
||||
|
||||
#### SSE Transport Implementation ✅
|
||||
|
||||
The server now supports **Server-Sent Events (SSE)** transport alongside HTTP:
|
||||
@@ -2031,7 +2085,7 @@ build-backend = "setuptools.build_meta"
|
||||
4. **Emsal**: 2 tools (search + document)
|
||||
5. **Uyuşmazlık**: 2 tools (search + document)
|
||||
6. **Constitutional Court**: ✅ 2 tools (unified norm control + individual applications) - **NEWLY UNIFIED**
|
||||
7. **KİK**: 2 tools (search + document)
|
||||
7. **KİK**: 2 tools (search + document) - **v2 API with three decision types: uyusmazlik, duzenleyici, mahkeme** ✅
|
||||
8. **Competition Authority**: 2 tools (search + document)
|
||||
9. **KVKK**: 2 tools (search + document)
|
||||
10. **Sayıştay**: 4 tools (3 search types + document)
|
||||
@@ -2069,6 +2123,52 @@ build-backend = "setuptools.build_meta"
|
||||
- ✅ Session management and tool discovery
|
||||
- **Claude AI**: Successfully connects and uses all Turkish legal database tools
|
||||
|
||||
#### ✅ Bedesten Tools Null Safety Fixes (Completed - Jul 23, 2025)
|
||||
- **Issue**: TypeError "cannot convert undefined or null to object" in Bedesten search and document tools
|
||||
- **Root Cause**: API responses containing null/undefined fields without proper validation
|
||||
- **Fixes Applied**:
|
||||
- ✅ **Search Function**: Added null safety checks for `response.data.emsalKararList` and `response.data.total`
|
||||
- ✅ **Document Function**: Added comprehensive validation for `doc_response.data`, `content`, and `mimeType` fields
|
||||
- ✅ **Error Handling**: Added descriptive error messages and graceful fallbacks
|
||||
- ✅ **Base64 Decoding**: Protected base64 operations with try-catch blocks
|
||||
- **Result**: Bedesten tools now handle API edge cases gracefully without crashing
|
||||
- **Production Status**: Deployed and operational on api.yargimcp.com
|
||||
|
||||
#### ✅ Automatic Date Format Conversion (Completed - Sep 2, 2025)
|
||||
- **Issue**: Bedesten API requires ISO 8601 format with timezone, but users were providing simple dates
|
||||
- **Problem**: Queries like `kararTarihiStart="2020-01-01"` returned "No data returned from Bedesten API"
|
||||
- **Root Cause**: Simple date format `YYYY-MM-DD` not converted to required `YYYY-MM-DDTHH:MM:SS.000Z` format
|
||||
- **Solution Applied**: ✅ **Automatic Date Format Conversion** in `search_bedesten_unified`
|
||||
- **Start dates**: `2020-01-01` → `2020-01-01T00:00:00.000Z` (beginning of day)
|
||||
- **End dates**: `2020-01-01` → `2020-01-01T23:59:59.999Z` (end of day)
|
||||
- **Backwards compatible**: Full ISO 8601 dates still work unchanged
|
||||
- **Smart detection**: Only converts if date doesn't already end with 'Z'
|
||||
- **Benefits**:
|
||||
- ✅ **User-friendly**: Simple date input now works seamlessly
|
||||
- ✅ **Inclusive ranges**: End dates include the entire specified day
|
||||
- ✅ **No breaking changes**: Existing ISO 8601 usage unaffected
|
||||
- **Production Status**: Deployed on api.yargimcp.com - **Version 353**
|
||||
|
||||
#### ✅ KİK v2 MCP Implementation Testing (Completed - Sep 2, 2025)
|
||||
- **Issue**: Test KİK v2 MCP implementation with all three decision types
|
||||
- **Request**: "üç karar türü ile de mcpyi test et" (test the MCP with all three decision types)
|
||||
- **Decision Types Tested**:
|
||||
- ✅ **uyusmazlik** (dispute) - 500 decisions found and searchable
|
||||
- ✅ **duzenleyici** (regulatory) - 8 decisions found and searchable
|
||||
- ✅ **mahkeme** (court) - 318 decisions found and searchable
|
||||
- **Total Coverage**: 826 decisions across all three decision types
|
||||
- **SSL Issues**: ✅ Resolved with legacy server connect configuration
|
||||
- **API Endpoints**: All three endpoints working correctly
|
||||
- `/api/KurulKararlari/GetKurulKararlari` (uyusmazlik)
|
||||
- `/api/KurulKararlari/GetKurulKararlariDk` (duzenleyici)
|
||||
- `/api/KurulKararlari/GetKurulKararlariMk` (mahkeme)
|
||||
- **Hash Analysis**: Comprehensive testing performed to understand document ID encryption
|
||||
- Tested various hash generation patterns (SHA256, HMAC, composite hashes)
|
||||
- Angular/cryptoService.encrypt() style approaches tested
|
||||
- Hash eşleşmesi bulunamadı - client-side session data veya farklı algoritma kullanılıyor olabilir
|
||||
- **Result**: ✅ KİK v2 MCP implementation fully operational for all three decision types
|
||||
- **Production Status**: All tests passing, search functionality working, ready for production deployment
|
||||
|
||||
### Key Features
|
||||
- **FastMCP Framework**: Modern MCP server implementation
|
||||
- **Unified APIs**: Single interface for multiple court systems
|
||||
|
||||
+43
-19
@@ -1,29 +1,53 @@
|
||||
# -------- BASE IMAGE (includes Chromium & deps) ----------------------------
|
||||
FROM mcr.microsoft.com/playwright/python:v1.53.0-noble
|
||||
# Use Python 3.12 slim image
|
||||
FROM python:3.12-slim
|
||||
|
||||
# -------- Runtime setup ----------------------------------------------------
|
||||
# Set working directory
|
||||
WORKDIR /app
|
||||
|
||||
# Copy dependency manifests first for layer-cache
|
||||
COPY pyproject.toml poetry.lock* requirements*.txt* ./
|
||||
# Install system dependencies (gcc/g++ kept in case any wheel falls back to source build)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc \
|
||||
g++ \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Fast, deterministic install with `uv`
|
||||
RUN pip install --no-cache-dir uv && \
|
||||
uv pip install --system --no-cache-dir .[asgi,saas]
|
||||
# Copy project metadata first for better Docker layer caching
|
||||
COPY pyproject.toml ./
|
||||
COPY README.md ./
|
||||
|
||||
# Copy application source
|
||||
COPY . .
|
||||
# Copy entry points
|
||||
COPY app.py ./
|
||||
COPY asgi_app.py ./
|
||||
COPY mcp_server_main.py ./
|
||||
|
||||
# -------- Environment ------------------------------------------------------
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
ENV ENABLE_AUTH=true
|
||||
ENV PORT=8000
|
||||
# Copy MCP modules and shared packages
|
||||
COPY anayasa_mcp_module ./anayasa_mcp_module
|
||||
COPY bddk_mcp_module ./bddk_mcp_module
|
||||
COPY bedesten_mcp_module ./bedesten_mcp_module
|
||||
COPY danistay_mcp_module ./danistay_mcp_module
|
||||
COPY emsal_mcp_module ./emsal_mcp_module
|
||||
COPY gib_mcp_module ./gib_mcp_module
|
||||
COPY kik_mcp_module ./kik_mcp_module
|
||||
COPY kvkk_mcp_module ./kvkk_mcp_module
|
||||
COPY rekabet_mcp_module ./rekabet_mcp_module
|
||||
COPY sayistay_mcp_module ./sayistay_mcp_module
|
||||
COPY sigorta_tahkim_mcp_module ./sigorta_tahkim_mcp_module
|
||||
COPY uyusmazlik_mcp_module ./uyusmazlik_mcp_module
|
||||
COPY yargitay_mcp_module ./yargitay_mcp_module
|
||||
COPY semantic_search ./semantic_search
|
||||
|
||||
# -------- Health check -----------------------------------------------------
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=10s --retries=3 \
|
||||
CMD python -c "import httpx, os, sys; r=httpx.get(f'http://localhost:{os.getenv(\"PORT\",\"8000\")}/health'); sys.exit(0 if r.status_code==200 else 1)"
|
||||
# Install the package with ASGI extras (uvicorn + starlette)
|
||||
RUN pip install --no-cache-dir -e ".[asgi]"
|
||||
|
||||
# Expose port
|
||||
EXPOSE 8000
|
||||
|
||||
# -------- Entrypoint -------------------------------------------------------
|
||||
CMD ["uvicorn", "asgi_app:app", "--host", "0.0.0.0", "--port", "8000", "--proxy-headers"]
|
||||
# Set environment variables
|
||||
ENV PORT=8000
|
||||
ENV PYTHONUNBUFFERED=1
|
||||
|
||||
# Health check
|
||||
HEALTHCHECK --interval=30s --timeout=10s --start-period=40s --retries=3 \
|
||||
CMD python -c "import httpx; httpx.get('http://localhost:8000/health', timeout=5)" || exit 1
|
||||
|
||||
# Run the ASGI application
|
||||
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8000"]
|
||||
|
||||
@@ -1,8 +1,103 @@
|
||||
# Yargı MCP: Türk Hukuk Kaynakları için MCP Sunucusu
|
||||
|
||||
> ## ✨ Profesyonel Sürüm Hazır: Yargı MCP Pro
|
||||
>
|
||||
> **Mevzuat ve içtihatı tek bir MCP sunucusunda birleştiren** profesyonel sürüm yayında:
|
||||
>
|
||||
> 👉 **https://yargi.betaspacestudio.com**
|
||||
|
||||
> ## 🚨 SUNUCU YENİ ADRESE TAŞINDI
|
||||
>
|
||||
> **Yeni Remote MCP adresi:** `https://yargimcp.surucu.dev/mcp`
|
||||
>
|
||||
> **Eski adres** (`https://yargimcp.fastmcp.app/mcp`) **artık kullanım dışıdır** — yalnızca taşındığını bildiren bir uyarı tool'u döner.
|
||||
>
|
||||
> **Yapmanız gereken:** MCP istemcinizdeki (Claude Desktop, 5ire, Google Antigravity, ChatGPT vb.) sunucu URL'sini yukarıdaki yeni adresle güncelleyin.
|
||||
|
||||
## Word'den UDF'ye profesyonel dönüşüm için yeni uygulamam [udfcevir.com](https://udfcevir.com) adresinde!
|
||||
|
||||
[](https://www.star-history.com/#saidsurucu/yargi-mcp&Date)
|
||||
|
||||
Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kararlar, Uyuşmazlık Mahkemesi, Anayasa Mahkemesi - Norm Denetimi ile Bireysel Başvuru Kararları, Kamu İhale Kurulu Kararları, Rekabet Kurumu Kararları, Sayıştay Kararları, KVKK Kararları ve BDDK Kararları) erişimi kolaylaştıran bir [FastMCP](https://gofastmcp.com/) sunucusu oluşturur. Bu sayede, bu kaynaklardan veri arama ve belge getirme işlemleri, Model Context Protocol (MCP) destekleyen LLM (Büyük Dil Modeli) uygulamaları (örneğin Claude Desktop veya [5ire](https://5ire.app)) ve diğer istemciler tarafından araç (tool) olarak kullanılabilir hale gelir.
|
||||
Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kararlar, Uyuşmazlık Mahkemesi, Anayasa Mahkemesi - Norm Denetimi ile Bireysel Başvuru Kararları, Kamu İhale Kurulu Kararları, Rekabet Kurumu Kararları, Sayıştay Kararları, KVKK Kararları, BDDK Kararları, GİB Özelgeleri ve Sigorta Tahkim Komisyonu Kararları) erişimi kolaylaştıran bir [FastMCP](https://gofastmcp.com/) sunucusu oluşturur. Bu sayede, bu kaynaklardan veri arama ve belge getirme işlemleri, Model Context Protocol (MCP) destekleyen LLM (Büyük Dil Modeli) uygulamaları (örneğin Claude Desktop veya [5ire](https://5ire.app)) ve diğer istemciler tarafından araç (tool) olarak kullanılabilir hale gelir.
|
||||
|
||||
---
|
||||
|
||||
## 🚀 5 Dakikada Başla (Remote MCP)
|
||||
|
||||
### ✅ Kurulum Gerektirmez! Hemen Kullan!
|
||||
|
||||
🔗 **Remote MCP Adresi:** `https://yargimcp.surucu.dev/mcp`
|
||||
|
||||
> ⚠️ **Eski adres** `https://yargimcp.fastmcp.app/mcp` **artık kullanım dışıdır** — yalnızca taşındığını bildiren bir uyarı tool'u döner. Lütfen yukarıdaki yeni adresi kullanın.
|
||||
|
||||
### Claude Desktop ile Kullanım (Ücretli abonelik gerekir)
|
||||
|
||||
1. **Claude Desktop'ı açın**
|
||||
2. **Settings → Connectors → Add Custom Connector**
|
||||
3. **Bilgileri girin:**
|
||||
- **Name:** `Yargı MCP`
|
||||
- **URL:** `https://yargimcp.surucu.dev/mcp`
|
||||
4. **Add** butonuna tıklayın
|
||||
5. **Hemen kullanmaya başlayın!** 🎉
|
||||
|
||||
### Google Antigravity ile Kullanım (Lokal `uv` Kurulumu — Kopyala-Yapıştır)
|
||||
|
||||
> **Ön Gereksinimler:** Bilgisayarınızda **Python**, **`uv`** ([kurulum](https://docs.astral.sh/uv/getting-started/installation/)) ve **Node.js** ([indir](https://nodejs.org/en/download)) kurulu olmalı. (Node.js yalnızca aşağıdaki kurulum komutunu çalıştırmak için gerekir; MCP'yi `uvx` çalıştırır.)
|
||||
|
||||
Aşağıdaki **bloğun tamamını** terminale yapıştırın. Komut, Antigravity'nin okuduğu `~/.gemini/config/mcp_config.json` dosyasını sizin yerinize oluşturur/günceller (varsa diğer sunucularınız korunur):
|
||||
|
||||
**macOS / Linux** (Terminal):
|
||||
|
||||
```bash
|
||||
node - <<'YARGI'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=path.join(os.homedir(),".gemini","config"),file=path.join(dir,"mcp_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
YARGI
|
||||
```
|
||||
|
||||
**Windows** (PowerShell):
|
||||
|
||||
```powershell
|
||||
@'
|
||||
const fs=require("fs"),os=require("os"),path=require("path");
|
||||
const dir=path.join(os.homedir(),".gemini","config"),file=path.join(dir,"mcp_config.json");
|
||||
fs.mkdirSync(dir,{recursive:true});
|
||||
let cfg={};try{cfg=JSON.parse(fs.readFileSync(file,"utf8"))}catch{}
|
||||
if(typeof cfg!=="object"||cfg===null||Array.isArray(cfg))cfg={};
|
||||
if(typeof cfg.mcpServers!=="object"||cfg.mcpServers===null)cfg.mcpServers={};
|
||||
cfg.mcpServers["yargi-mcp"]={command:"uvx",args:["yargi-mcp"]};
|
||||
fs.writeFileSync(file,JSON.stringify(cfg,null,2)+"\n");
|
||||
console.log("yargi-mcp eklendi -> "+file);
|
||||
'@ | node -
|
||||
```
|
||||
|
||||
Komut `yargi-mcp eklendi -> ...` çıktısını verdiğinde kurulum tamamlanmıştır. Antigravity'yi (açıksa kapatıp) yeniden başlatın; `yargi-mcp` araçları otomatik yüklenir.
|
||||
|
||||
> 💡 **İpucu:** Lokal kurulumda hukuk kaynaklarına erişim doğrudan bilgisayarınızda `uvx yargi-mcp` ile çalışır; uzaktan sunucuya ihtiyaç duymaz.
|
||||
|
||||
### Remote MCP Sorun Giderme
|
||||
|
||||
`https://yargimcp.surucu.dev/mcp` bir web sayfası değil, Streamable HTTP MCP uç noktasıdır. Tarayıcıda açınca veya düz `curl` ile GET isteği atınca `406 Not Acceptable` ve `Client must accept text/event-stream` benzeri bir yanıt görmek normaldir; bu, sunucunun kapalı olduğu anlamına gelmez. MCP istemcisi `Accept: application/json, text/event-stream` başlığıyla JSON-RPC isteği göndermelidir.
|
||||
|
||||
Hızlı sağlık kontrolü için tarayıcıda şu adresleri açabilirsiniz:
|
||||
|
||||
- `https://yargimcp.surucu.dev/health` — servis sağlık durumu
|
||||
|
||||
Claude.ai veya başka bir istemci "araç yok" gibi davranırsa:
|
||||
|
||||
1. Connector'ı kaldırıp yeniden ekleyin.
|
||||
2. URL olarak önce `https://yargimcp.surucu.dev/mcp` deneyin; istemciniz yönlendirmeleri takip etmiyorsa `https://yargimcp.surucu.dev/mcp/` deneyin.
|
||||
3. Eski `https://yargimcp.fastmcp.app/mcp` adresinin istemci ayarlarında veya önbellekte kalmadığından emin olun.
|
||||
4. İstemcinin remote/Streamable HTTP MCP desteklediğini ve `text/event-stream` kabul ettiğini kontrol edin.
|
||||
|
||||
---
|
||||
|
||||

|
||||
|
||||
@@ -30,6 +125,8 @@ Bu proje, çeşitli Türk hukuk kaynaklarına (Yargıtay, Danıştay, Emsal Kara
|
||||
* **Sayıştay:** 3 karar türü ile kapsamlı denetim kararlarına erişim + **8 Daire Filtreleme** + **Tarih Aralığı & İçerik Arama** (Genel Kurul yorumlayıcı kararları, Temyiz Kurulu itiraz kararları, Daire ilk derece denetim kararları)
|
||||
* **KVKK (Kişisel Verilerin Korunması Kurulu):** Brave Search API ile veri koruma kararlarını arama; uzun karar metinlerini (5.000 karakterlik) sayfalanmış Markdown formatında getirme + **Türkçe Arama** + **Site Hedeflemeli Arama** (kvkk.gov.tr kararları)
|
||||
* **BDDK (Bankacılık Düzenleme ve Denetleme Kurumu):** Bankacılık düzenleme kararlarını arama; karar metinlerini Markdown formatında getirme + **Optimized Search** + **"Karar Sayısı" Targeting** + **Spesifik URL Filtreleme** (bddk.org.tr/Mevzuat/DokumanGetir)
|
||||
* **GİB (Gelir İdaresi Başkanlığı) Özelgeleri:** Resmi vergi özelgelerini arama (18.000+ özelge: KDV, Kurumlar, Gelir, ÖTV, Damga vb.); tam metni sayfalanmış Markdown formatında getirme + **Keyword + Özelge No + Kanun No + Tarih Aralığı** + **Otomatik ISO 8601 Dönüşümü** + **Metadata Başlık Bloğu**
|
||||
* **Sigorta Tahkim Komisyonu:** Hakem Karar Dergisi (64 sayı, 2010-2025) içindeki sigorta tahkim kararlarını arama; dergi PDF'lerini Markdown formatında getirme + **Sayı İçi Karar Arama** + **Türkçe Büyük/Küçük Harf Desteği** + **Relevance Scoring**
|
||||
|
||||
* Karar metinlerinin daha kolay işlenebilmesi için Markdown formatına çevrilmesi.
|
||||
* Claude Desktop uygulaması ile `fastmcp install` komutu kullanılarak kolay entegrasyon.
|
||||
@@ -132,10 +229,110 @@ Yargı MCP'yi Gemini CLI ile kullanmak için:
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
<details>
|
||||
<summary>🧠 <strong>Semantik Arama (Opsiyonel)</strong></summary>
|
||||
|
||||
Yargı MCP, **semantik arama** özelliği ile kararları anlamsal olarak sıralayabilir. Opsiyoneldir; iki yoldan biri yapılandırıldığında otomatik etkinleşir:
|
||||
|
||||
- **Yerel** (önerilen, ücretsiz): kendi makinenizdeki OpenAI-uyumlu embedding sunucusu (HuggingFace TEI, llama.cpp, Ollama, vLLM, LM Studio…)
|
||||
- **Hosted**: OpenRouter API anahtarı
|
||||
|
||||
### Semantik Arama Nasıl Çalışır?
|
||||
1. `initial_keyword` ile Bedesten API'den 100 karar çekilir
|
||||
2. `query` ile bu kararlar embedding modeli kullanılarak anlamsal olarak sıralanır
|
||||
3. En alakalı kararlar döndürülür
|
||||
|
||||
### Önerilen Türkçe Kurulumu (Yerel — `multilingual-e5-large`)
|
||||
|
||||
`intfloat/multilingual-e5-large` Türkçe için kıyas ettiğimiz açık kaynak modeller arasında en iyilerinden. HuggingFace'in **Text Embeddings Inference (TEI)** sunucusuyla tek komutta ayağa kalkar ve OpenAI-uyumlu API sunar:
|
||||
|
||||
```bash
|
||||
docker run -p 8080:80 ghcr.io/huggingface/text-embeddings-inference:latest \
|
||||
--model-id intfloat/multilingual-e5-large
|
||||
```
|
||||
|
||||
Sonra Yargı MCP'ye şu env vars'ları geçirin:
|
||||
|
||||
```bash
|
||||
EMBEDDING_PROVIDER=local
|
||||
LOCAL_EMBEDDING_BASE_URL=http://localhost:8080/v1
|
||||
LOCAL_EMBEDDING_MODEL=intfloat/multilingual-e5-large
|
||||
LOCAL_EMBEDDING_DIMENSION=1024
|
||||
EMBEDDING_PROMPT_STYLE=e5
|
||||
```
|
||||
|
||||
> ⚠️ **Önemli:** `EMBEDDING_PROMPT_STYLE=e5` şart — e5 modelleri `query:` / `passage:` öneki bekleyecek şekilde eğitilmiştir; yanlış önek sessizce kaliteyi düşürür.
|
||||
|
||||
#### Claude Desktop örneği (yerel TEI)
|
||||
```json
|
||||
{
|
||||
"mcpServers": {
|
||||
"Yargı MCP": {
|
||||
"command": "uvx",
|
||||
"args": ["yargi-mcp"],
|
||||
"env": {
|
||||
"EMBEDDING_PROVIDER": "local",
|
||||
"LOCAL_EMBEDDING_BASE_URL": "http://localhost:8080/v1",
|
||||
"LOCAL_EMBEDDING_MODEL": "intfloat/multilingual-e5-large",
|
||||
"LOCAL_EMBEDDING_DIMENSION": "1024",
|
||||
"EMBEDDING_PROMPT_STYLE": "e5"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Alternatif 1: Ollama (yerel, daha hafif kurulum)
|
||||
|
||||
```bash
|
||||
ollama serve
|
||||
ollama pull nomic-embed-text # 768 dim, İngilizce ağırlıklı
|
||||
```
|
||||
|
||||
```bash
|
||||
EMBEDDING_PROVIDER=local
|
||||
LOCAL_EMBEDDING_BASE_URL=http://localhost:11434/v1
|
||||
LOCAL_EMBEDDING_MODEL=nomic-embed-text
|
||||
LOCAL_EMBEDDING_DIMENSION=768
|
||||
EMBEDDING_PROMPT_STYLE=raw
|
||||
```
|
||||
|
||||
> Ollama kütüphanesinde `multilingual-e5-large` doğrudan yok; Türkçe için TEI yolu daha doğru sonuç verir.
|
||||
|
||||
### Alternatif 2: OpenRouter (hosted)
|
||||
|
||||
```bash
|
||||
OPENROUTER_API_KEY=sk-or-v1-xxx...
|
||||
# İsteğe bağlı — varsayılan google/gemini-embedding-001 (3072 dim, ÜCRETLİ)
|
||||
# OPENROUTER_EMBEDDING_MODEL=...
|
||||
# OPENROUTER_EMBEDDING_DIMENSION=...
|
||||
# EMBEDDING_PROMPT_STYLE=gemini # varsayılan
|
||||
```
|
||||
|
||||
API anahtarınızı [openrouter.ai/keys](https://openrouter.ai/keys) adresinden alın. Varsayılan model `google/gemini-embedding-001` artık ücretli — ücretsiz bir model seçerseniz `OPENROUTER_EMBEDDING_MODEL`, `OPENROUTER_EMBEDDING_DIMENSION` ve uygun `EMBEDDING_PROMPT_STYLE` değerlerini birlikte ayarlayın.
|
||||
|
||||
### Yapılandırma Referansı
|
||||
|
||||
| Env Var | Açıklama | Örnek |
|
||||
|---|---|---|
|
||||
| `EMBEDDING_PROVIDER` | `local` ise yerel sunucu, boş ise OpenRouter | `local` |
|
||||
| `EMBEDDING_PROMPT_STYLE` | `gemini` / `e5` / `raw` — modelin beklediği önek | `e5` |
|
||||
| `LOCAL_EMBEDDING_BASE_URL` | Yerel sunucunun OpenAI-uyumlu URL'i | `http://localhost:8080/v1` |
|
||||
| `LOCAL_EMBEDDING_MODEL` | Model adı | `intfloat/multilingual-e5-large` |
|
||||
| `LOCAL_EMBEDDING_DIMENSION` | Modelin çıktı boyutu (mutlaka eşleşmeli) | `1024` |
|
||||
| `OPENROUTER_API_KEY` | OpenRouter anahtarı (sadece hosted için) | `sk-or-v1-…` |
|
||||
| `OPENROUTER_EMBEDDING_MODEL` | OpenRouter model id'si | `google/gemini-embedding-001` |
|
||||
| `OPENROUTER_EMBEDDING_DIMENSION` | OpenRouter modelinin çıktı boyutu | `3072` |
|
||||
|
||||
> 💡 **Not:** Hiçbir embedding sağlayıcı yapılandırılmazsa semantik arama aracı görünmez, diğer 24 araç normal şekilde çalışır.
|
||||
|
||||
</details>
|
||||
|
||||
<details>
|
||||
<summary>🛠️ <strong>Kullanılabilir Araçlar (MCP Tools)</strong></summary>
|
||||
|
||||
Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliği için optimize edilmiş):
|
||||
Bu FastMCP sunucusu **26 aktif MCP aracı** + **1 opsiyonel semantik arama aracı** sunar (token verimliliği için optimize edilmiş):
|
||||
|
||||
### **Yargıtay Araçları (Birleşik Bedesten API - Token Optimized)**
|
||||
*Not: Yargıtay araçları token verimliliği için birleşik Bedesten API'ye entegre edilmiştir*
|
||||
@@ -160,8 +357,8 @@ Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliğ
|
||||
8. `get_anayasa_document_unified(document_url, page_number)`: AYM kararlarını birleşik belge getirme - **sayfalanmış Markdown** içeriği
|
||||
|
||||
### **KİK (Kamu İhale Kurulu) Araçları**
|
||||
9. `search_kik_decisions(karar_tipi, ...)`: KİK (Kamu İhale Kurulu) kararlarını arar.
|
||||
10. `get_kik_document_markdown(karar_id, page_number)`: Belirli bir KİK kararını, Base64 ile encode edilmiş `karar_id`'sini kullanarak alır ve **sayfalanmış Markdown** içeriğini getirir.
|
||||
9. `search_kik_v2_decisions(decision_type, karar_metni, karar_no, basvuran, idare_adi, baslangic_tarihi, bitis_tarihi)`: KİK v2 API ile uyuşmazlık, düzenleyici ve mahkeme kararlarını arar.
|
||||
10. `get_kik_v2_document_markdown(gundemMaddesiId)`: Arama sonucundaki `gundemMaddesiId` ile KİK karar metnini Markdown formatında getirir.
|
||||
### **Rekabet Kurumu Araçları**
|
||||
* `search_rekabet_kurumu_decisions(KararTuru: Literal[...], ...) -> RekabetSearchResult`: Rekabet Kurumu kararlarını arar. `KararTuru` için kullanıcı dostu isimler kullanılır (örn: "Birleşme ve Devralma").
|
||||
* `get_rekabet_kurumu_document(karar_id: str, page_number: Optional[int] = 1) -> RekabetDocument`: Belirli bir Rekabet Kurumu kararını `karar_id` ile alır. Kararın PDF formatındaki orijinalinden istenen sayfayı ayıklar ve Markdown formatında döndürür.
|
||||
@@ -169,22 +366,32 @@ Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliğ
|
||||
|
||||
---
|
||||
|
||||
* **Sayıştay Araçları (3 Karar Türü + 8 Daire Filtreleme):**
|
||||
* `search_sayistay_genel_kurul(karar_no, karar_tarih_baslangic, karar_tamami, ...)`: Sayıştay Genel Kurul (yorumlayıcı) kararlarını arar. **Tarih aralığı** (2006-2024) + **İçerik arama** (400 karakter)
|
||||
* `search_sayistay_temyiz_kurulu(ilam_dairesi, kamu_idaresi_turu, temyiz_karar, ...)`: Temyiz Kurulu (itiraz) kararlarını arar. **8 Daire filtreleme** + **Kurum türü** + **Konu sınıflandırması**
|
||||
* `search_sayistay_daire(yargilama_dairesi, web_karar_metni, hesap_yili, ...)`: Daire (ilk derece denetim) kararlarını arar. **8 Daire filtreleme** + **Hesap yılı** + **İçerik arama**
|
||||
* `get_sayistay_genel_kurul_document_markdown(decision_id: str)`: Genel Kurul kararının tam metnini Markdown formatında getirir
|
||||
* `get_sayistay_temyiz_kurulu_document_markdown(decision_id: str)`: Temyiz Kurulu kararının tam metnini Markdown formatında getirir
|
||||
* `get_sayistay_daire_document_markdown(decision_id: str)`: Daire kararının tam metnini Markdown formatında getirir
|
||||
* **Sayıştay Araçları (Birleşik API, 3 Karar Türü + 8 Daire Filtreleme):**
|
||||
* `search_sayistay_unified(decision_type, start, length, ...)`: `genel_kurul`, `temyiz_kurulu` veya `daire` kararlarını tek araçla arar. `length` 1-100 aralığındadır.
|
||||
* `get_sayistay_document_unified(decision_id, decision_type)`: Birleşik arama sonucundaki karar ID'si ve karar türüyle tam metni Markdown formatında getirir.
|
||||
|
||||
* **KVKK Araçları (Brave Search API + Türkçe Arama):**
|
||||
* `search_kvkk_decisions(keywords, page, pageSize, ...)`: KVKK (Kişisel Verilerin Korunması Kurulu) kararlarını Brave Search API ile arar. **Türkçe arama** + **Site hedeflemeli** (`site:kvkk.gov.tr "karar özeti"`) + **Sayfalama desteği**
|
||||
* `search_kvkk_decisions(keywords, page)`: KVKK (Kişisel Verilerin Korunması Kurulu) kararlarını Brave Search API ile arar. **Türkçe arama** + **Site hedeflemeli** (`site:kvkk.gov.tr "karar özeti"`) + **Sayfalama desteği**. Sonuç sayısı sunucuda 10 olarak sabitlenmiştir.
|
||||
* `get_kvkk_document_markdown(decision_url: str, page_number: Optional[int] = 1)`: KVKK kararının tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa)
|
||||
|
||||
### BDDK Araçları
|
||||
* `search_bddk_decisions(keywords, page)`: BDDK (Bankacılık Düzenleme ve Denetleme Kurumu) kararlarını arar. **"Karar Sayısı" targeting** + **Spesifik URL filtreleme** (`bddk.org.tr/Mevzuat/DokumanGetir`) + **Optimized search**
|
||||
* `get_bddk_document_markdown(document_id: str, page_number: Optional[int] = 1)`: BDDK kararının tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa)
|
||||
|
||||
### GİB (Gelir İdaresi Başkanlığı) Özelge Araçları (Resmi GİB JSON API)
|
||||
* `search_gib_ozelge(keywords, ozelgeNo, kanunNo, ozelgeStartDate, ozelgeEndDate, page, pageSize)`: GİB özelgelerini (Türk Gelir İdaresi Başkanlığı vergi özelgeleri) arar — **18.000+ özelge** (KDV, Kurumlar, Gelir, ÖTV, Damga, VUK vb.). **Keyword + Özelge No + Kanun No + Tarih Aralığı** + **Otomatik ISO 8601 Dönüşümü** (`YYYY-MM-DD` girdileri otomatik olarak full ISO 8601'e çevrilir)
|
||||
* `get_gib_ozelge_document_markdown(ozelge_id: int, page_number: int = 1)`: Belirli bir özelgenin tam metnini **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa) + **Metadata başlık bloğu** (Başlık, Sayı, Tarih, Kanun, Kaynak URL)
|
||||
|
||||
### Sigorta Tahkim Komisyonu Araçları (Tavily Search API + PDF)
|
||||
* `search_sigorta_tahkim_decisions(keywords, page)`: Sigorta Tahkim Komisyonu kararlarını Tavily Search API ile arar. **Site hedeflemeli** (`sigortatahkim.org`) + **Sayfalama desteği**. Sonuç sayısı sunucuda 10 olarak sabitlenmiştir.
|
||||
* `get_sigorta_tahkim_document_markdown(issue_number: str, page_number: int)`: Hakem Karar Dergisi sayısının PDF'ini indirip **sayfalanmış Markdown** formatında getirir (5.000 karakterlik sayfa). 64 sayı (2010-2025)
|
||||
* `search_within_sigorta_tahkim_issue(issue_number: str, keyword: str, max_results: int)`: Belirli bir dergi sayısı içindeki kararları anahtar kelime ile arar. **Türkçe İ/I desteği** + **Relevance scoring** + **Excerpt** ile sonuç
|
||||
|
||||
### Yardımcı ve Uyumluluk Araçları
|
||||
* `check_government_servers_health()`: Yargı kaynaklarının erişilebilirliğini kontrol eder.
|
||||
* `search(query)`: ChatGPT Deep Research uyumluluğu için Bedesten destekli kaynaklarda arama yapar.
|
||||
* `fetch(id)`: ChatGPT Deep Research uyumluluğu için tek bir Bedesten belge ID'sinin tam metnini getirir.
|
||||
|
||||
</details>
|
||||
|
||||
---
|
||||
@@ -199,8 +406,8 @@ Bu FastMCP sunucusu **19 optimize edilmiş MCP aracı** sunar (token verimliliğ
|
||||
- **Korunan İşlevsellik:** %100 özellik desteği devam ediyor
|
||||
|
||||
**GENEL İSTATİSTİKLER:**
|
||||
- **Toplam Mahkeme/Kurum:** 13 farklı hukuki kurum (KVKK dahil)
|
||||
- **Toplam MCP Tool:** 19 optimize edilmiş arama ve belge getirme aracı
|
||||
- **Toplam Mahkeme/Kurum:** 15 farklı hukuki kurum (GİB Özelgeleri ve Sigorta Tahkim Komisyonu dahil)
|
||||
- **Toplam MCP Tool:** 26 aktif araç + 1 opsiyonel semantik arama aracı
|
||||
- **Daire/Kurul Filtreleme:** 87 farklı seçenek (52 Yargıtay + 27 Danıştay + 8 Sayıştay)
|
||||
- **Tarih Filtreleme:** Birleşik Bedesten API aracında ISO 8601 formatında tam tarih aralığı desteği
|
||||
- **Kesin Cümle Arama:** Birleşik Bedesten API aracında çift tırnak ile tam cümle arama (`"\"mülkiyet kararı\""` formatı)
|
||||
|
||||
@@ -0,0 +1,239 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
"""
|
||||
Analyze KİK v2 hash generation by examining JavaScript code patterns
|
||||
and trying to reverse engineer the hash generation logic.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import hashlib
|
||||
import hmac
|
||||
import base64
|
||||
from fastmcp import Client
|
||||
from mcp_server_main import app
|
||||
|
||||
def analyze_webpack_hash_patterns():
|
||||
"""
|
||||
Analyze the webpack JavaScript code you provided to find hash generation patterns
|
||||
"""
|
||||
print("🔍 Analyzing webpack hash generation patterns...")
|
||||
|
||||
# From the JavaScript code, I can see several hash/ID generation patterns:
|
||||
hash_patterns = {
|
||||
# Webpack chunk system hashes (from the JS code)
|
||||
"webpack_chunks": {
|
||||
315: "d9a9486a4f5ba326",
|
||||
531: "cd8fb385c88033ae",
|
||||
671: "04c48b287646627a",
|
||||
856: "682c9a7b87351f90",
|
||||
1017: "9de022378fc275f6",
|
||||
# ... many more from the __webpack_require__.u function
|
||||
},
|
||||
|
||||
# Symbol generation from Zone.js
|
||||
"zone_symbols": [
|
||||
"__zone_symbol__",
|
||||
"__Zone_symbol_prefix",
|
||||
"Zone.__symbol__"
|
||||
],
|
||||
|
||||
# Angular module federation patterns
|
||||
"module_federation": [
|
||||
"__webpack_modules__",
|
||||
"__webpack_module_cache__",
|
||||
"__webpack_require__"
|
||||
]
|
||||
}
|
||||
|
||||
# The target hash format
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"🎯 Target hash: {target_hash}")
|
||||
print(f" Length: {len(target_hash)} characters")
|
||||
print(f" Format: {'SHA256' if len(target_hash) == 64 else 'Other'} (64 chars = SHA256)")
|
||||
|
||||
return hash_patterns
|
||||
|
||||
def test_webpack_style_hashing(data_dict):
|
||||
"""Test webpack-style hash generation methods"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try various webpack-style hash methods
|
||||
hashes[f"webpack_md5_{key}"] = hashlib.md5(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha1_{key}"] = hashlib.sha1(test_string.encode()).hexdigest()
|
||||
hashes[f"webpack_sha256_{key}"] = hashlib.sha256(test_string.encode()).hexdigest()
|
||||
|
||||
# Try with various prefixes/suffixes (common in webpack)
|
||||
prefixed = f"__webpack__{test_string}"
|
||||
hashes[f"webpack_prefixed_sha256_{key}"] = hashlib.sha256(prefixed.encode()).hexdigest()
|
||||
|
||||
# Try with module federation style
|
||||
module_style = f"shell:{test_string}"
|
||||
hashes[f"module_fed_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
# Try JSON stringified
|
||||
json_style = json.dumps({"id": value, "type": "decision"}, separators=(',', ':'))
|
||||
hashes[f"json_sha256_{key}"] = hashlib.sha256(json_style.encode()).hexdigest()
|
||||
|
||||
# Try with timestamp or sequence
|
||||
with_seq = f"{test_string}_0"
|
||||
hashes[f"seq_sha256_{key}"] = hashlib.sha256(with_seq.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_angular_routing_hashes(data_dict):
|
||||
"""Test Angular routing/state management hash generation"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
# Angular often uses route parameters for hash generation
|
||||
route_style = f"/kurul-kararlari/{value}"
|
||||
hashes[f"route_sha256_{key}"] = hashlib.sha256(route_style.encode()).hexdigest()
|
||||
|
||||
# Component state style
|
||||
state_style = f"KurulKararGoster_{value}"
|
||||
hashes[f"state_sha256_{key}"] = hashlib.sha256(state_style.encode()).hexdigest()
|
||||
|
||||
# Angular module style
|
||||
module_style = f"kik.kurul.karar.{value}"
|
||||
hashes[f"module_sha256_{key}"] = hashlib.sha256(module_style.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
def test_base64_encoding_variants(data_dict):
|
||||
"""Test various base64 and encoding variants"""
|
||||
hashes = {}
|
||||
|
||||
for key, value in data_dict.items():
|
||||
test_string = str(value)
|
||||
|
||||
# Try base64 encoding then hashing
|
||||
b64_encoded = base64.b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64_sha256_{key}"] = hashlib.sha256(b64_encoded.encode()).hexdigest()
|
||||
|
||||
# Try URL-safe base64
|
||||
b64_url = base64.urlsafe_b64encode(test_string.encode()).decode()
|
||||
hashes[f"b64url_sha256_{key}"] = hashlib.sha256(b64_url.encode()).hexdigest()
|
||||
|
||||
# Try hex encoding
|
||||
hex_encoded = test_string.encode().hex()
|
||||
hashes[f"hex_sha256_{key}"] = hashlib.sha256(hex_encoded.encode()).hexdigest()
|
||||
|
||||
return hashes
|
||||
|
||||
async def test_hash_generation_comprehensive():
|
||||
print("🔐 Comprehensive KİK document hash generation analysis...")
|
||||
print("=" * 70)
|
||||
|
||||
# First analyze the webpack patterns
|
||||
webpack_patterns = analyze_webpack_hash_patterns()
|
||||
|
||||
client = Client(app)
|
||||
|
||||
async with client:
|
||||
print("✅ MCP client connected")
|
||||
|
||||
# Get sample decisions
|
||||
print("\n📊 Getting sample decisions for hash analysis...")
|
||||
search_result = await client.call_tool("search_kik_v2_decisions", {
|
||||
"decision_type": "uyusmazlik",
|
||||
"karar_metni": "2024"
|
||||
})
|
||||
|
||||
if hasattr(search_result, 'content') and search_result.content:
|
||||
search_data = json.loads(search_result.content[0].text)
|
||||
decisions = search_data.get('decisions', [])
|
||||
|
||||
if decisions:
|
||||
print(f"✅ Found {len(decisions)} decisions")
|
||||
|
||||
# Test with first decision
|
||||
sample_decision = decisions[0]
|
||||
print(f"\n📋 Sample decision for hash analysis:")
|
||||
for key, value in sample_decision.items():
|
||||
print(f" {key}: {value}")
|
||||
|
||||
target_hash = "42f9bcd59e0dfbca36dec9accf5686c7a92aa97724cd8fc3550beb84b80409da"
|
||||
print(f"\n🎯 Target hash to match: {target_hash}")
|
||||
|
||||
all_hashes = {}
|
||||
|
||||
# Test different hash generation methods
|
||||
print(f"\n🔨 Testing webpack-style hashing...")
|
||||
webpack_hashes = test_webpack_style_hashing(sample_decision)
|
||||
all_hashes.update(webpack_hashes)
|
||||
|
||||
print(f"🔨 Testing Angular routing hashes...")
|
||||
angular_hashes = test_angular_routing_hashes(sample_decision)
|
||||
all_hashes.update(angular_hashes)
|
||||
|
||||
print(f"🔨 Testing base64 encoding variants...")
|
||||
b64_hashes = test_base64_encoding_variants(sample_decision)
|
||||
all_hashes.update(b64_hashes)
|
||||
|
||||
# Check for matches
|
||||
print(f"\n🎯 Checking for hash matches...")
|
||||
matches_found = []
|
||||
partial_matches = []
|
||||
|
||||
for hash_name, hash_value in all_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
matches_found.append((hash_name, hash_value))
|
||||
print(f" 🎉 EXACT MATCH FOUND: {hash_name}")
|
||||
elif hash_value[:8] == target_hash[:8]: # First 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (first 8): {hash_name} -> {hash_value[:16]}...")
|
||||
elif hash_value[-8:] == target_hash[-8:]: # Last 8 chars match
|
||||
partial_matches.append((hash_name, hash_value))
|
||||
print(f" 🔍 Partial match (last 8): {hash_name} -> ...{hash_value[-16:]}")
|
||||
|
||||
if not matches_found and not partial_matches:
|
||||
print(f" ❌ No matches found")
|
||||
print(f"\n📝 Sample generated hashes (first 10):")
|
||||
for i, (hash_name, hash_value) in enumerate(list(all_hashes.items())[:10]):
|
||||
print(f" {hash_name}: {hash_value}")
|
||||
|
||||
# Try combinations with other decisions
|
||||
print(f"\n🔄 Testing hash combinations with multiple decisions...")
|
||||
if len(decisions) > 1:
|
||||
for i, decision in enumerate(decisions[1:3]): # Test 2 more
|
||||
print(f"\n Testing decision {i+2}: {decision.get('kararNo')}")
|
||||
decision_hashes = test_webpack_style_hashing(decision)
|
||||
|
||||
for hash_name, hash_value in decision_hashes.items():
|
||||
if hash_value == target_hash:
|
||||
print(f" 🎉 MATCH FOUND in decision {i+2}: {hash_name}")
|
||||
matches_found.append((f"decision_{i+2}_{hash_name}", hash_value))
|
||||
|
||||
# Try composite hashes (combining multiple fields)
|
||||
print(f"\n🔗 Testing composite hash generation...")
|
||||
composite_tests = [
|
||||
f"{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararNo')}",
|
||||
f"{sample_decision.get('kararNo')}_{sample_decision.get('kararTarihi')}",
|
||||
f"uyusmazlik_{sample_decision.get('gundemMaddesiId')}_{sample_decision.get('kararTarihi')}",
|
||||
json.dumps(sample_decision, separators=(',', ':'), sort_keys=True),
|
||||
f"{sample_decision.get('basvuran')}_{sample_decision.get('gundemMaddesiId')}",
|
||||
]
|
||||
|
||||
for i, composite_str in enumerate(composite_tests):
|
||||
composite_hash = hashlib.sha256(composite_str.encode()).hexdigest()
|
||||
if composite_hash == target_hash:
|
||||
print(f" 🎉 COMPOSITE MATCH FOUND: test_{i} -> {composite_str[:50]}...")
|
||||
matches_found.append((f"composite_{i}", composite_hash))
|
||||
|
||||
print(f"\n🎯 Hash analysis completed!")
|
||||
print(f" Total matches found: {len(matches_found)}")
|
||||
print(f" Partial matches: {len(partial_matches)}")
|
||||
|
||||
else:
|
||||
print("❌ No decisions found")
|
||||
else:
|
||||
print("❌ Search failed")
|
||||
|
||||
print("=" * 70)
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(test_hash_generation_comprehensive())
|
||||
@@ -1,6 +1,7 @@
|
||||
# anayasa_mcp_module/bireysel_client.py
|
||||
# This client is for Bireysel Başvuru: https://kararlarbilgibankasi.anayasa.gov.tr
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup, Tag
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
@@ -302,7 +303,7 @@ class AnayasaBireyselBasvuruApiClient:
|
||||
elif "Karar Tarihi" in key and not karar_tarihi_from_page: karar_tarihi_from_page = value
|
||||
elif "Resmi Gazete Tarih / Sayı" in key: resmi_gazete_info_from_page = value
|
||||
|
||||
full_markdown_content = self._convert_html_to_markdown_bireysel(html_content_from_api)
|
||||
full_markdown_content = await asyncio.to_thread(self._convert_html_to_markdown_bireysel, html_content_from_api)
|
||||
|
||||
if not full_markdown_content:
|
||||
return AnayasaBireyselBasvuruDocumentMarkdown(
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
# anayasa_mcp_module/client.py
|
||||
# This client is for Norm Denetimi: https://normkararlarbilgibankasi.anayasa.gov.tr
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Tuple
|
||||
@@ -309,7 +310,7 @@ class AnayasaMahkemesiApiClient:
|
||||
official_gazette_from_page = rg_text_content.replace("Resmî Gazete tarih ve sayısı:", "").replace("Resmi Gazete tarih/sayı:", "").strip()
|
||||
|
||||
|
||||
full_markdown_content = self._convert_html_to_markdown_norm_denetimi(html_content_from_api)
|
||||
full_markdown_content = await asyncio.to_thread(self._convert_html_to_markdown_norm_denetimi, html_content_from_api)
|
||||
|
||||
if not full_markdown_content:
|
||||
return AnayasaDocumentMarkdown(
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
# Unified client for both Norm Denetimi and Bireysel Başvuru
|
||||
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlparse
|
||||
from typing import Optional, Tuple
|
||||
from urllib.parse import urlparse, urlunparse
|
||||
|
||||
from .models import (
|
||||
AnayasaUnifiedSearchRequest,
|
||||
@@ -18,6 +18,51 @@ from .bireysel_client import AnayasaBireyselBasvuruApiClient
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Canonical hosts per decision type. Norm Denetimi (/ND/) documents live on the
|
||||
# "norm" subdomain; Bireysel Başvuru (/BB/) documents on the plain subdomain.
|
||||
# Callers (or upstream search links) sometimes supply the wrong host for a given
|
||||
# path, which makes the AYM server return 404. We re-key the host off the path.
|
||||
_NORM_HOST = "normkararlarbilgibankasi.anayasa.gov.tr"
|
||||
_BIREYSEL_HOST = "kararlarbilgibankasi.anayasa.gov.tr"
|
||||
|
||||
|
||||
def normalize_anayasa_document_url(document_url: str) -> Tuple[Optional[str], str]:
|
||||
"""Detect the AYM decision type from the URL path and force the correct host.
|
||||
|
||||
Detection is path-based (``/ND/`` vs ``/BB/``) because the path is
|
||||
unambiguous, whereas the supplied host may be wrong. Query params and
|
||||
fragment are preserved (they are harmless for document fetches).
|
||||
|
||||
Returns ``(decision_type, normalized_url)`` where ``decision_type`` is
|
||||
``"norm_denetimi"``, ``"bireysel_basvuru"``, or ``None`` if it cannot be
|
||||
determined (URL returned unchanged in that case).
|
||||
"""
|
||||
parsed = urlparse(document_url)
|
||||
path = parsed.path or ""
|
||||
|
||||
if "/ND/" in path:
|
||||
decision_type, host = "norm_denetimi", _NORM_HOST
|
||||
elif "/BB/" in path:
|
||||
decision_type, host = "bireysel_basvuru", _BIREYSEL_HOST
|
||||
else:
|
||||
# Fall back to host-based detection when the path is uninformative.
|
||||
if "normkararlarbilgibankasi" in parsed.netloc:
|
||||
return "norm_denetimi", document_url
|
||||
if "kararlarbilgibankasi" in parsed.netloc:
|
||||
return "bireysel_basvuru", document_url
|
||||
return None, document_url
|
||||
|
||||
normalized = urlunparse((
|
||||
parsed.scheme or "https",
|
||||
host,
|
||||
parsed.path,
|
||||
parsed.params,
|
||||
parsed.query,
|
||||
parsed.fragment,
|
||||
))
|
||||
return decision_type, normalized
|
||||
|
||||
|
||||
class AnayasaUnifiedClient:
|
||||
"""Unified client that handles both Norm Denetimi and Bireysel Başvuru searches."""
|
||||
|
||||
@@ -80,12 +125,18 @@ class AnayasaUnifiedClient:
|
||||
async def get_document_unified(self, document_url: str, page_number: int = 1) -> AnayasaUnifiedDocumentMarkdown:
|
||||
"""Unified document retrieval that auto-detects the appropriate client."""
|
||||
|
||||
# Auto-detect decision type based on URL
|
||||
parsed_url = urlparse(document_url)
|
||||
# Auto-detect decision type from the path and force the correct host.
|
||||
# This repairs malformed URLs (e.g. a /ND/ path on the bireysel host),
|
||||
# which otherwise 404 against the AYM server.
|
||||
decision_type, normalized_url = normalize_anayasa_document_url(document_url)
|
||||
if normalized_url != document_url:
|
||||
logger.info(
|
||||
f"AnayasaUnifiedClient: Normalized document URL "
|
||||
f"'{document_url}' -> '{normalized_url}'"
|
||||
)
|
||||
|
||||
if "normkararlarbilgibankasi" in parsed_url.netloc or "/ND/" in document_url:
|
||||
# Norm Denetimi document
|
||||
result = await self.norm_client.get_decision_document_as_markdown(document_url, page_number)
|
||||
if decision_type == "norm_denetimi":
|
||||
result = await self.norm_client.get_decision_document_as_markdown(normalized_url, page_number)
|
||||
|
||||
return AnayasaUnifiedDocumentMarkdown(
|
||||
decision_type="norm_denetimi",
|
||||
@@ -97,9 +148,8 @@ class AnayasaUnifiedClient:
|
||||
is_paginated=result.is_paginated
|
||||
)
|
||||
|
||||
elif "kararlarbilgibankasi" in parsed_url.netloc or "/BB/" in document_url:
|
||||
# Bireysel Başvuru document
|
||||
result = await self.bireysel_client.get_decision_document_as_markdown(document_url, page_number)
|
||||
elif decision_type == "bireysel_basvuru":
|
||||
result = await self.bireysel_client.get_decision_document_as_markdown(normalized_url, page_number)
|
||||
|
||||
return AnayasaUnifiedDocumentMarkdown(
|
||||
decision_type="bireysel_basvuru",
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
"""
|
||||
ASGI application for Yargı MCP Server (simple deployment variant).
|
||||
|
||||
This is a minimal ASGI application that can be run with:
|
||||
uvicorn app:app --host 0.0.0.0 --port 8000
|
||||
|
||||
The MCP server will be available at:
|
||||
http://localhost:8000/mcp/
|
||||
|
||||
For the FastAPI-wrapped variant with CORS and extra metadata routes,
|
||||
see asgi_app.py instead.
|
||||
"""
|
||||
|
||||
from starlette.responses import JSONResponse
|
||||
from mcp_server_main import create_app
|
||||
|
||||
mcp = create_app()
|
||||
|
||||
|
||||
@mcp.custom_route("/health", methods=["GET"])
|
||||
async def health_check(request):
|
||||
"""Health check endpoint for monitoring services (Fly.io, Render, etc.)."""
|
||||
return JSONResponse({
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.2.1",
|
||||
})
|
||||
|
||||
|
||||
# Create ASGI app directly from FastMCP server
|
||||
app = mcp.http_app()
|
||||
|
||||
# Endpoints:
|
||||
# - /mcp/ - MCP server (Streamable HTTP transport, default FastMCP path)
|
||||
# - /health - Health check for monitoring
|
||||
Regular → Executable
+37
-457
@@ -2,109 +2,36 @@
|
||||
ASGI application for Yargı MCP Server
|
||||
|
||||
This module provides ASGI/HTTP access to the Yargı MCP server,
|
||||
allowing it to be deployed as a web service with FastAPI wrapper
|
||||
for Stripe webhook integration.
|
||||
allowing it to be deployed as a web service with FastAPI wrapper.
|
||||
|
||||
Usage:
|
||||
uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
"""
|
||||
|
||||
import os
|
||||
import time
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime, timedelta
|
||||
from fastapi import FastAPI, Request, HTTPException, Query
|
||||
from fastapi.responses import JSONResponse, HTMLResponse
|
||||
from fastapi.exception_handlers import http_exception_handler
|
||||
from fastapi import FastAPI, Request
|
||||
from fastapi.responses import JSONResponse
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.responses import Response
|
||||
from starlette.requests import Request as StarletteRequest
|
||||
|
||||
# Import the MCP app creator function
|
||||
from mcp_server_main import create_app
|
||||
|
||||
# Import Stripe webhook router
|
||||
from stripe_webhook import router as stripe_router
|
||||
|
||||
# Import simplified MCP Auth HTTP adapter
|
||||
from mcp_auth_http_simple import router as mcp_auth_router
|
||||
|
||||
# OAuth configuration from environment variables
|
||||
CLERK_ISSUER = os.getenv("CLERK_ISSUER", "https://accounts.yargimcp.com")
|
||||
BASE_URL = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
# Setup logging
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Configure CORS and Auth middleware
|
||||
# Configure CORS
|
||||
cors_origins = os.getenv("ALLOWED_ORIGINS", "*").split(",")
|
||||
|
||||
# Import FastMCP Bearer Auth Provider
|
||||
from fastmcp.server.auth import BearerAuthProvider
|
||||
from fastmcp.server.auth.providers.bearer import RSAKeyPair
|
||||
# Create MCP app
|
||||
mcp_server = create_app()
|
||||
|
||||
# Clerk JWT configuration for Bearer token validation
|
||||
CLERK_SECRET_KEY = os.getenv("CLERK_SECRET_KEY")
|
||||
CLERK_ISSUER = os.getenv("CLERK_ISSUER", "https://accounts.yargimcp.com")
|
||||
CLERK_PUBLISHABLE_KEY = os.getenv("CLERK_PUBLISHABLE_KEY")
|
||||
|
||||
# Configure Bearer token authentication
|
||||
bearer_auth = None
|
||||
if CLERK_SECRET_KEY and CLERK_ISSUER:
|
||||
# Production: Use Clerk JWKS endpoint for token validation
|
||||
bearer_auth = BearerAuthProvider(
|
||||
jwks_uri=f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
issuer=CLERK_ISSUER,
|
||||
algorithm="RS256",
|
||||
audience=None, # Disable audience validation - Clerk uses different audience format
|
||||
required_scopes=[] # Disable scope validation - Clerk JWT has ['read', 'search']
|
||||
)
|
||||
logger.info(f"Bearer auth configured with Clerk JWKS: {CLERK_ISSUER}/.well-known/jwks.json")
|
||||
else:
|
||||
# Development: Generate RSA key pair for testing
|
||||
logger.warning("No Clerk credentials found - using development RSA key pair")
|
||||
dev_key_pair = RSAKeyPair.generate()
|
||||
bearer_auth = BearerAuthProvider(
|
||||
public_key=dev_key_pair.public_key,
|
||||
issuer="https://dev.yargimcp.com",
|
||||
audience="dev-mcp-server",
|
||||
required_scopes=["yargi.read"]
|
||||
)
|
||||
|
||||
# Generate a test token for development
|
||||
dev_token = dev_key_pair.create_token(
|
||||
subject="dev-user",
|
||||
issuer="https://dev.yargimcp.com",
|
||||
audience="dev-mcp-server",
|
||||
scopes=["yargi.read", "yargi.search"],
|
||||
expires_in_seconds=3600 * 24 # 24 hours for development
|
||||
)
|
||||
logger.info(f"Development Bearer token: {dev_token}")
|
||||
|
||||
custom_middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=cors_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST", "OPTIONS", "DELETE"],
|
||||
allow_headers=["Content-Type", "Authorization", "X-Request-ID", "X-Session-ID"],
|
||||
),
|
||||
]
|
||||
|
||||
# Create MCP app with Bearer authentication
|
||||
mcp_server = create_app(auth=bearer_auth)
|
||||
|
||||
# Add Starlette middleware to FastAPI (not MCP)
|
||||
# MCP already has Bearer auth, no need for additional middleware on MCP level
|
||||
|
||||
# Create MCP Starlette sub-application with root path - mount will add /mcp prefix
|
||||
# Create MCP Starlette sub-application
|
||||
mcp_app = mcp_server.http_app(path="/")
|
||||
|
||||
# Configure JSON encoder for proper Turkish character support
|
||||
import json
|
||||
from fastapi.responses import JSONResponse
|
||||
|
||||
# Configure JSON encoder for proper Turkish character support
|
||||
class UTF8JSONResponse(JSONResponse):
|
||||
def __init__(self, content=None, status_code=200, headers=None, **kwargs):
|
||||
if headers is None:
|
||||
@@ -121,82 +48,55 @@ class UTF8JSONResponse(JSONResponse):
|
||||
separators=(",", ":"),
|
||||
).encode("utf-8")
|
||||
|
||||
custom_middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=cors_origins,
|
||||
allow_credentials=True,
|
||||
allow_methods=["GET", "POST", "OPTIONS"],
|
||||
allow_headers=["Content-Type", "X-Request-ID", "X-Session-ID"],
|
||||
),
|
||||
]
|
||||
|
||||
# Create FastAPI wrapper application
|
||||
app = FastAPI(
|
||||
title="Yargı MCP Server",
|
||||
description="MCP server for Turkish legal databases with OAuth authentication",
|
||||
description="MCP server for Turkish legal databases",
|
||||
version="0.1.0",
|
||||
middleware=custom_middleware,
|
||||
default_response_class=UTF8JSONResponse # Use UTF-8 JSON encoder
|
||||
default_response_class=UTF8JSONResponse,
|
||||
redirect_slashes=False,
|
||||
)
|
||||
|
||||
# Add Stripe webhook router to FastAPI
|
||||
app.include_router(stripe_router, prefix="/api")
|
||||
|
||||
# Add MCP Auth HTTP adapter to FastAPI (handles OAuth endpoints)
|
||||
app.include_router(mcp_auth_router)
|
||||
|
||||
# Custom 401 exception handler for MCP spec compliance
|
||||
@app.exception_handler(401)
|
||||
async def custom_401_handler(request: Request, exc: HTTPException):
|
||||
"""Custom 401 handler that adds WWW-Authenticate header as required by MCP spec"""
|
||||
response = await http_exception_handler(request, exc)
|
||||
|
||||
# Add WWW-Authenticate header pointing to protected resource metadata
|
||||
# as required by RFC 9728 Section 5.1 and MCP Authorization spec
|
||||
response.headers["WWW-Authenticate"] = (
|
||||
'Bearer '
|
||||
'error="invalid_token", '
|
||||
'error_description="The access token is missing or invalid", '
|
||||
f'resource="{BASE_URL}/.well-known/oauth-protected-resource"'
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
# FastAPI health check endpoint - BEFORE mounting MCP app
|
||||
@app.get("/health")
|
||||
async def health_check():
|
||||
"""Health check endpoint for monitoring"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"auth_enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true"
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
# Add explicit redirect for /mcp to /mcp/ with method preservation
|
||||
@app.api_route("/mcp", methods=["GET", "POST", "HEAD", "OPTIONS"])
|
||||
async def redirect_to_slash(request: Request):
|
||||
"""Redirect /mcp to /mcp/ preserving HTTP method with 308"""
|
||||
from fastapi.responses import RedirectResponse
|
||||
return RedirectResponse(url="/mcp/", status_code=308)
|
||||
|
||||
# Mount MCP app at /mcp/ with trailing slash
|
||||
app.mount("/mcp/", mcp_app)
|
||||
|
||||
# Set the lifespan context after mounting
|
||||
app.router.lifespan_context = mcp_app.lifespan
|
||||
|
||||
|
||||
# SSE transport deprecated - removed
|
||||
|
||||
# FastAPI root endpoint
|
||||
@app.get("/")
|
||||
async def root():
|
||||
"""Root endpoint with service information"""
|
||||
return JSONResponse({
|
||||
return {
|
||||
"service": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases with OAuth authentication",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"endpoints": {
|
||||
"mcp": "/mcp",
|
||||
"health": "/health",
|
||||
"status": "/status",
|
||||
"stripe_webhook": "/api/stripe/webhook",
|
||||
"oauth_login": "/auth/login",
|
||||
"oauth_callback": "/auth/callback",
|
||||
"oauth_google": "/auth/google/login",
|
||||
"user_info": "/auth/user"
|
||||
},
|
||||
"transports": {
|
||||
"http": "/mcp"
|
||||
@@ -210,166 +110,14 @@ async def root():
|
||||
"Kamu İhale Kurulu (Public Procurement Authority)",
|
||||
"Rekabet Kurumu (Competition Authority)",
|
||||
"Sayıştay (Court of Accounts)",
|
||||
"Bedesten API (Multiple courts)"
|
||||
"KVKK (Personal Data Protection Authority)",
|
||||
"BDDK (Banking Regulation and Supervision Agency)",
|
||||
"Bedesten API (Multiple courts)",
|
||||
"Sigorta Tahkim Komisyonu (Insurance Arbitration Commission)",
|
||||
],
|
||||
"authentication": {
|
||||
"enabled": os.getenv("ENABLE_AUTH", "false").lower() == "true",
|
||||
"type": "OAuth 2.0 via Clerk",
|
||||
"issuer": os.getenv("CLERK_ISSUER", "https://clerk.accounts.dev"),
|
||||
"providers": ["google"],
|
||||
"flow": "authorization_code"
|
||||
}
|
||||
})
|
||||
|
||||
# OAuth 2.0 Authorization Server Metadata proxy (for MCP clients that can't reach Clerk directly)
|
||||
# MCP Auth Toolkit expects this to be under /mcp/.well-known/oauth-authorization-server
|
||||
@app.get("/mcp/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server():
|
||||
"""OAuth 2.0 Authorization Server Metadata proxy to Clerk - MCP Auth Toolkit standard location"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"token_endpoint_auth_methods_supported": ["client_secret_basic", "none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"subject_types_supported": ["public"],
|
||||
"id_token_signing_alg_values_supported": ["RS256"],
|
||||
"claims_supported": ["sub", "iss", "aud", "exp", "iat", "email", "name"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
|
||||
# Claude AI MCP specific endpoint format
|
||||
@app.get("/.well-known/oauth-authorization-server/mcp")
|
||||
async def oauth_authorization_server_mcp_suffix():
|
||||
"""OAuth 2.0 Authorization Server Metadata - Claude AI MCP specific format"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"token_endpoint_auth_methods_supported": ["client_secret_basic", "none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"subject_types_supported": ["public"],
|
||||
"id_token_signing_alg_values_supported": ["RS256"],
|
||||
"claims_supported": ["sub", "iss", "aud", "exp", "iat", "email", "name"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
|
||||
@app.get("/.well-known/oauth-protected-resource/mcp")
|
||||
async def oauth_protected_resource_mcp_suffix():
|
||||
"""OAuth 2.0 Protected Resource Metadata - Claude AI MCP specific format"""
|
||||
return JSONResponse({
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [
|
||||
BASE_URL
|
||||
],
|
||||
"scopes_supported": ["read", "search"],
|
||||
"bearer_methods_supported": ["header"],
|
||||
"resource_documentation": f"{BASE_URL}/mcp",
|
||||
"resource_policy_uri": f"{BASE_URL}/privacy"
|
||||
})
|
||||
|
||||
# Keep root level for compatibility with some MCP clients
|
||||
@app.get("/.well-known/oauth-authorization-server")
|
||||
async def oauth_authorization_server_root():
|
||||
"""OAuth 2.0 Authorization Server Metadata proxy to Clerk - root level for compatibility"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"jwks_uri": f"{CLERK_ISSUER}/.well-known/jwks.json",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"token_endpoint_auth_methods_supported": ["client_secret_basic", "none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"subject_types_supported": ["public"],
|
||||
"id_token_signing_alg_values_supported": ["RS256"],
|
||||
"claims_supported": ["sub", "iss", "aud", "exp", "iat", "email", "name"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"service_documentation": f"{BASE_URL}/mcp",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"resource_documentation": f"{BASE_URL}/mcp"
|
||||
})
|
||||
|
||||
# Note: GET /mcp is handled by the mounted MCP app itself
|
||||
# This prevents 405 Method Not Allowed errors on POST requests
|
||||
|
||||
# OAuth 2.0 Protected Resource Metadata (RFC 9728) - MCP Spec Required
|
||||
@app.get("/.well-known/oauth-protected-resource")
|
||||
async def oauth_protected_resource():
|
||||
"""OAuth 2.0 Protected Resource Metadata as required by MCP spec"""
|
||||
return JSONResponse({
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [
|
||||
BASE_URL
|
||||
],
|
||||
"scopes_supported": ["read", "search"],
|
||||
"bearer_methods_supported": ["header"],
|
||||
"resource_documentation": f"{BASE_URL}/mcp",
|
||||
"resource_policy_uri": f"{BASE_URL}/privacy"
|
||||
})
|
||||
|
||||
# Standard well-known discovery endpoint
|
||||
@app.get("/.well-known/mcp")
|
||||
async def well_known_mcp():
|
||||
"""Standard MCP discovery endpoint"""
|
||||
return JSONResponse({
|
||||
"mcp_server": {
|
||||
"name": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"endpoint": f"{BASE_URL}/mcp",
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": f"{BASE_URL}/auth/login",
|
||||
"scopes": ["read", "search"]
|
||||
},
|
||||
"capabilities": ["tools", "resources"],
|
||||
"tools_count": len(mcp_server._tool_manager._tools)
|
||||
}
|
||||
})
|
||||
|
||||
# MCP Discovery endpoint for ChatGPT integration
|
||||
@app.get("/mcp/discovery")
|
||||
async def mcp_discovery():
|
||||
"""MCP Discovery endpoint for ChatGPT and other MCP clients"""
|
||||
return JSONResponse({
|
||||
"name": "Yargı MCP Server",
|
||||
"description": "MCP server for Turkish legal databases",
|
||||
"version": "0.1.0",
|
||||
"protocol": "mcp",
|
||||
"transport": "http",
|
||||
"endpoint": "/mcp",
|
||||
"authentication": {
|
||||
"type": "oauth2",
|
||||
"authorization_url": "/auth/login",
|
||||
"token_url": "/auth/callback",
|
||||
"scopes": ["read", "search"],
|
||||
"provider": "clerk"
|
||||
},
|
||||
"capabilities": {
|
||||
"tools": True,
|
||||
"resources": True,
|
||||
"prompts": False
|
||||
},
|
||||
"tools_count": len(mcp_server._tool_manager._tools),
|
||||
"contact": {
|
||||
"url": BASE_URL,
|
||||
"email": "support@yargi-mcp.dev"
|
||||
}
|
||||
})
|
||||
|
||||
# FastAPI status endpoint
|
||||
@app.get("/status")
|
||||
async def status():
|
||||
"""Status endpoint with detailed information"""
|
||||
@@ -380,187 +128,19 @@ async def status():
|
||||
"description": tool.description[:100] + "..." if len(tool.description) > 100 else tool.description
|
||||
})
|
||||
|
||||
return JSONResponse({
|
||||
return {
|
||||
"status": "operational",
|
||||
"tools": tools,
|
||||
"total_tools": len(tools),
|
||||
"transport": "streamable_http",
|
||||
"architecture": "FastAPI wrapper + MCP Starlette sub-app",
|
||||
"auth_status": "enabled" if os.getenv("ENABLE_AUTH", "false").lower() == "true" else "disabled"
|
||||
})
|
||||
}
|
||||
|
||||
# Note: JWT token validation is now handled entirely by Clerk
|
||||
# All authentication flows use Clerk JWT tokens directly
|
||||
|
||||
async def validate_clerk_session(request: Request, clerk_token: str = None) -> str:
|
||||
"""Validate Clerk session from cookies or JWT token and return user_id"""
|
||||
logger.info(f"Validating Clerk session - token provided: {bool(clerk_token)}")
|
||||
# Mount MCP app at /mcp/
|
||||
app.mount("/mcp/", mcp_app)
|
||||
|
||||
try:
|
||||
# Try to import Clerk SDK
|
||||
from clerk_backend_api import Clerk
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
|
||||
# Try JWT token first (from URL parameter)
|
||||
if clerk_token:
|
||||
logger.info("Validating Clerk JWT token from URL parameter")
|
||||
try:
|
||||
# Extract session_id from JWT token and verify with Clerk
|
||||
import jwt
|
||||
decoded_token = jwt.decode(clerk_token, options={"verify_signature": False})
|
||||
session_id = decoded_token.get("sid") # Use standard JWT 'sid' claim
|
||||
|
||||
if session_id:
|
||||
# Verify with Clerk using session_id
|
||||
session = clerk.sessions.verify(session_id=session_id, token=clerk_token)
|
||||
user_id = session.user_id if session else None
|
||||
|
||||
if user_id:
|
||||
logger.info(f"JWT token validation successful - user_id: {user_id}")
|
||||
return user_id
|
||||
else:
|
||||
logger.error("JWT token validation failed - no user_id in session")
|
||||
else:
|
||||
logger.error("No session_id found in JWT token")
|
||||
except Exception as e:
|
||||
logger.error(f"JWT token validation failed: {str(e)}")
|
||||
# Fall through to cookie validation
|
||||
|
||||
# Fallback to cookie validation
|
||||
logger.info("Attempting cookie-based session validation")
|
||||
clerk_session = request.cookies.get("__session")
|
||||
if not clerk_session:
|
||||
logger.error("No Clerk session cookie found")
|
||||
raise HTTPException(status_code=401, detail="No Clerk session found")
|
||||
|
||||
# Validate session with Clerk
|
||||
session = clerk.sessions.verify_session(clerk_session)
|
||||
logger.info(f"Cookie session validation successful - user_id: {session.user_id}")
|
||||
return session.user_id
|
||||
|
||||
except ImportError:
|
||||
# Fallback for development without Clerk SDK
|
||||
logger.warning("Clerk SDK not available - using development fallback")
|
||||
return "dev_user_123"
|
||||
except Exception as e:
|
||||
logger.error(f"Session validation failed: {str(e)}")
|
||||
raise HTTPException(status_code=401, detail=f"Session validation failed: {str(e)}")
|
||||
|
||||
# MCP OAuth Callback Endpoint
|
||||
@app.get("/auth/mcp-callback")
|
||||
async def mcp_oauth_callback(request: Request, clerk_token: str = Query(None)):
|
||||
"""Handle OAuth callback for MCP token generation"""
|
||||
logger.info(f"MCP OAuth callback - clerk_token provided: {bool(clerk_token)}")
|
||||
|
||||
try:
|
||||
# Validate Clerk session with JWT token support
|
||||
user_id = await validate_clerk_session(request, clerk_token)
|
||||
logger.info(f"User authenticated successfully - user_id: {user_id}")
|
||||
|
||||
# Use the Clerk JWT token directly (no need to generate custom token)
|
||||
logger.info("User authenticated successfully via Clerk")
|
||||
|
||||
# Return success response
|
||||
return HTMLResponse(f"""
|
||||
<html>
|
||||
<head>
|
||||
<title>MCP Connection Successful</title>
|
||||
<style>
|
||||
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
|
||||
.success {{ color: #28a745; }}
|
||||
.token {{ background: #f8f9fa; padding: 15px; border-radius: 5px; margin: 20px 0; word-break: break-all; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<h1 class="success">✅ MCP Connection Successful!</h1>
|
||||
<p>Your Yargı MCP integration is now active.</p>
|
||||
<div class="token">
|
||||
<strong>Authentication:</strong><br>
|
||||
<code>Use your Clerk JWT token directly with Bearer authentication</code>
|
||||
</div>
|
||||
<p>You can now close this window and return to your MCP client.</p>
|
||||
<script>
|
||||
// Try to close the popup if opened as such
|
||||
if (window.opener) {{
|
||||
window.opener.postMessage({{
|
||||
type: 'MCP_AUTH_SUCCESS',
|
||||
token: 'use_clerk_jwt_token'
|
||||
}}, '*');
|
||||
setTimeout(() => window.close(), 3000);
|
||||
}}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
""")
|
||||
|
||||
except HTTPException as e:
|
||||
logger.error(f"MCP OAuth callback failed: {e.detail}")
|
||||
return HTMLResponse(f"""
|
||||
<html>
|
||||
<head>
|
||||
<title>MCP Connection Failed</title>
|
||||
<style>
|
||||
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
|
||||
.error {{ color: #dc3545; }}
|
||||
.debug {{ background: #f8f9fa; padding: 10px; margin: 20px 0; border-radius: 5px; font-family: monospace; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<h1 class="error">❌ MCP Connection Failed</h1>
|
||||
<p>{e.detail}</p>
|
||||
<div class="debug">
|
||||
<strong>Debug Info:</strong><br>
|
||||
Clerk Token: {'✅ Provided' if clerk_token else '❌ Missing'}<br>
|
||||
Error: {e.detail}<br>
|
||||
Status: {e.status_code}
|
||||
</div>
|
||||
<p>Please try again or contact support.</p>
|
||||
<a href="https://yargimcp.com/sign-in">Return to Sign In</a>
|
||||
</body>
|
||||
</html>
|
||||
""", status_code=e.status_code)
|
||||
except Exception as e:
|
||||
logger.error(f"Unexpected error in MCP OAuth callback: {str(e)}")
|
||||
return HTMLResponse(f"""
|
||||
<html>
|
||||
<head>
|
||||
<title>MCP Connection Error</title>
|
||||
<style>
|
||||
body {{ font-family: Arial, sans-serif; text-align: center; padding: 50px; }}
|
||||
.error {{ color: #dc3545; }}
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<h1 class="error">❌ Unexpected Error</h1>
|
||||
<p>An unexpected error occurred during authentication.</p>
|
||||
<p>Error: {str(e)}</p>
|
||||
<a href="https://yargimcp.com/sign-in">Return to Sign In</a>
|
||||
</body>
|
||||
</html>
|
||||
""", status_code=500)
|
||||
|
||||
# OAuth2 Token Endpoint - Now uses Clerk JWT tokens directly
|
||||
@app.post("/auth/mcp-token")
|
||||
async def mcp_token_endpoint(request: Request):
|
||||
"""OAuth2 token endpoint for MCP clients - returns Clerk JWT token info"""
|
||||
try:
|
||||
# Validate Clerk session
|
||||
user_id = await validate_clerk_session(request)
|
||||
|
||||
return JSONResponse({
|
||||
"message": "Use your Clerk JWT token directly with Bearer authentication",
|
||||
"token_type": "Bearer",
|
||||
"scope": "yargi.read",
|
||||
"user_id": user_id,
|
||||
"instructions": "Include 'Authorization: Bearer YOUR_CLERK_JWT_TOKEN' in your requests"
|
||||
})
|
||||
except HTTPException as e:
|
||||
return JSONResponse(
|
||||
status_code=e.status_code,
|
||||
content={"error": "invalid_request", "error_description": e.detail}
|
||||
)
|
||||
|
||||
# Note: Only HTTP transport supported - SSE transport deprecated
|
||||
# Set the lifespan context after mounting
|
||||
app.router.lifespan_context = mcp_app.lifespan
|
||||
|
||||
# Export for uvicorn
|
||||
__all__ = ["app"]
|
||||
@@ -1,5 +1,6 @@
|
||||
# bddk_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from typing import List, Optional, Dict, Any
|
||||
import logging
|
||||
@@ -210,14 +211,19 @@ class BddkApiClient:
|
||||
|
||||
# Convert to Markdown based on content type
|
||||
if "pdf" in content_type:
|
||||
# Handle PDF documents
|
||||
# Handle PDF documents. markitdown is sync; offload to thread
|
||||
# so PDF parsing doesn't block the event-loop / other requests.
|
||||
pdf_stream = io.BytesIO(response.content)
|
||||
result = self.markitdown.convert_stream(pdf_stream, file_extension=".pdf")
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, pdf_stream, file_extension=".pdf"
|
||||
)
|
||||
markdown_content = result.text_content
|
||||
else:
|
||||
# Handle HTML documents
|
||||
# Handle HTML documents (sync conversion offloaded to thread)
|
||||
html_stream = io.BytesIO(response.content)
|
||||
result = self.markitdown.convert_stream(html_stream, file_extension=".html")
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, html_stream, file_extension=".html"
|
||||
)
|
||||
markdown_content = result.text_content
|
||||
|
||||
# Clean up the markdown content
|
||||
|
||||
@@ -1,11 +1,15 @@
|
||||
# bedesten_mcp_module/client.py
|
||||
|
||||
import httpx
|
||||
import asyncio
|
||||
import base64
|
||||
from typing import Optional
|
||||
import logging
|
||||
from markitdown import MarkItDown
|
||||
import io
|
||||
import logging
|
||||
import os
|
||||
import time
|
||||
from typing import Optional
|
||||
|
||||
import httpx
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
BedestenSearchRequest, BedestenSearchResponse,
|
||||
@@ -16,6 +20,72 @@ from .enums import get_full_birim_adi
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class BedestenRateLimited(Exception):
|
||||
"""Raised when the local rate-limit bucket would block longer than allowed.
|
||||
|
||||
Carries the suggested retry-after (seconds) so callers can surface a
|
||||
structured 429-style response to the MCP client instead of silently
|
||||
blocking the event-loop slot for the full bucket-pause window.
|
||||
"""
|
||||
|
||||
def __init__(self, retry_after: float) -> None:
|
||||
self.retry_after = retry_after
|
||||
super().__init__(f"local bucket would block {retry_after:.1f}s")
|
||||
|
||||
|
||||
class _TokenBucket:
|
||||
"""Asyncio token bucket with explicit back-pressure.
|
||||
|
||||
Measured Bedesten limit (per source IP, 2026-05-08): 10 requests per
|
||||
rolling 30s window with full refill — equivalent to capacity=10,
|
||||
refill_rate=1 token / 3s. Even with margin, 429s still leak through
|
||||
when other clients share the egress IP, so we also expose
|
||||
``penalize_until`` so callers can freeze the bucket when the server
|
||||
actually returns 429 (Retry-After).
|
||||
"""
|
||||
|
||||
def __init__(self, capacity: int, refill_per_s: float) -> None:
|
||||
self.capacity = float(capacity)
|
||||
self.refill_per_s = float(refill_per_s)
|
||||
self._tokens = float(capacity)
|
||||
self._last = time.monotonic()
|
||||
self._not_before = 0.0
|
||||
self._lock = asyncio.Lock()
|
||||
|
||||
async def acquire(self, max_wait: Optional[float] = None) -> None:
|
||||
"""Acquire one token. If ``max_wait`` is set and the next wait would
|
||||
exceed it, raise :class:`BedestenRateLimited` immediately instead of
|
||||
sleeping — keeps a single rate-limited request from holding the
|
||||
worker-slot for the full bucket-pause window (up to ~30s on 429)."""
|
||||
deadline = (time.monotonic() + max_wait) if max_wait is not None else None
|
||||
while True:
|
||||
async with self._lock:
|
||||
now = time.monotonic()
|
||||
if now < self._not_before:
|
||||
wait_s = self._not_before - now
|
||||
else:
|
||||
self._tokens = min(
|
||||
self.capacity,
|
||||
self._tokens + (now - self._last) * self.refill_per_s,
|
||||
)
|
||||
self._last = now
|
||||
if self._tokens >= 1.0:
|
||||
self._tokens -= 1.0
|
||||
return
|
||||
wait_s = (1.0 - self._tokens) / self.refill_per_s
|
||||
if deadline is not None:
|
||||
remaining = deadline - time.monotonic()
|
||||
if wait_s > remaining:
|
||||
raise BedestenRateLimited(retry_after=wait_s)
|
||||
await asyncio.sleep(wait_s)
|
||||
|
||||
def penalize_until(self, monotonic_deadline: float) -> None:
|
||||
"""Pause the bucket until ``monotonic_deadline`` (drains tokens)."""
|
||||
self._not_before = max(self._not_before, monotonic_deadline)
|
||||
self._tokens = 0.0
|
||||
self._last = time.monotonic()
|
||||
|
||||
class BedestenApiClient:
|
||||
"""
|
||||
API Client for Bedesten (bedesten.adalet.gov.tr) - Alternative legal decision search system.
|
||||
@@ -25,6 +95,17 @@ class BedestenApiClient:
|
||||
SEARCH_ENDPOINT = "/emsal-karar/searchDocuments"
|
||||
DOCUMENT_ENDPOINT = "/emsal-karar/getDocumentContent"
|
||||
|
||||
# Measured limit (per source IP): 10 requests per 30s window with full
|
||||
# refill (≈ 1 token / 3s steady). We default to 1-token capacity and
|
||||
# 3.5s spacing (no burst, ~14% safety margin). Override via env:
|
||||
# BEDESTEN_RATE_CAPACITY (default 1)
|
||||
# BEDESTEN_RATE_REFILL_S (default 3.5; seconds per token)
|
||||
# BEDESTEN_RATE_MAX_WAIT_S (default 8.0; max seconds to wait in the
|
||||
# local bucket before returning a structured 429 to the caller)
|
||||
_DEFAULT_CAPACITY = int(os.getenv("BEDESTEN_RATE_CAPACITY", "1"))
|
||||
_DEFAULT_REFILL_S = float(os.getenv("BEDESTEN_RATE_REFILL_S", "3.5"))
|
||||
_DEFAULT_MAX_WAIT_S = float(os.getenv("BEDESTEN_RATE_MAX_WAIT_S", "8.0"))
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
@@ -42,6 +123,24 @@ class BedestenApiClient:
|
||||
},
|
||||
timeout=request_timeout
|
||||
)
|
||||
self._bucket = _TokenBucket(
|
||||
capacity=self._DEFAULT_CAPACITY,
|
||||
refill_per_s=1.0 / self._DEFAULT_REFILL_S,
|
||||
)
|
||||
|
||||
def _handle_429(self, response: httpx.Response, op: str) -> None:
|
||||
"""Apply back-pressure to the shared bucket based on Retry-After."""
|
||||
retry_after_raw = response.headers.get("Retry-After", "")
|
||||
try:
|
||||
retry_after = float(retry_after_raw)
|
||||
except (TypeError, ValueError):
|
||||
retry_after = 30.0
|
||||
# Cap penalty so a hostile/buggy server can't freeze us indefinitely.
|
||||
retry_after = max(1.0, min(retry_after, 60.0))
|
||||
self._bucket.penalize_until(time.monotonic() + retry_after + 0.5)
|
||||
logger.warning(
|
||||
f"BedestenApiClient: 429 on {op}; bucket paused {retry_after + 0.5:.1f}s"
|
||||
)
|
||||
|
||||
async def search_documents(self, search_request: BedestenSearchRequest) -> BedestenSearchResponse:
|
||||
"""
|
||||
@@ -63,10 +162,13 @@ class BedestenApiClient:
|
||||
if not request_dict["data"]["birimAdi"]: # Remove if empty string
|
||||
del request_dict["data"]["birimAdi"]
|
||||
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.post(
|
||||
self.SEARCH_ENDPOINT,
|
||||
json=request_dict
|
||||
)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, "search")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -94,26 +196,51 @@ class BedestenApiClient:
|
||||
)
|
||||
|
||||
# Get document
|
||||
await self._bucket.acquire(max_wait=self._DEFAULT_MAX_WAIT_S)
|
||||
response = await self.http_client.post(
|
||||
self.DOCUMENT_ENDPOINT,
|
||||
json=doc_request.model_dump()
|
||||
)
|
||||
if response.status_code == 429:
|
||||
self._handle_429(response, f"document {document_id}")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
doc_response = BedestenDocumentResponse(**response_json)
|
||||
|
||||
# Decode base64 content
|
||||
# Add null safety checks for document data
|
||||
if not hasattr(doc_response, 'data') or doc_response.data is None:
|
||||
raise ValueError("Document response does not contain data")
|
||||
|
||||
if not hasattr(doc_response.data, 'content') or doc_response.data.content is None:
|
||||
raise ValueError("Document data does not contain content")
|
||||
|
||||
if not hasattr(doc_response.data, 'mimeType') or doc_response.data.mimeType is None:
|
||||
raise ValueError("Document data does not contain mimeType")
|
||||
|
||||
# Decode base64 content with error handling
|
||||
try:
|
||||
content_bytes = base64.b64decode(doc_response.data.content)
|
||||
except Exception as e:
|
||||
raise ValueError(f"Failed to decode base64 content: {str(e)}")
|
||||
|
||||
mime_type = doc_response.data.mimeType
|
||||
|
||||
logger.info(f"BedestenApiClient: Document mime type: {mime_type}")
|
||||
|
||||
# Convert to markdown based on mime type
|
||||
# Convert to markdown based on mime type. markitdown is sync and
|
||||
# PDF parsing in particular can block the event-loop for seconds,
|
||||
# which on a single-worker uvicorn deployment stalls every other
|
||||
# in-flight MCP request and new TLS handshakes. Offload to a
|
||||
# thread so the event-loop stays responsive.
|
||||
if mime_type == "text/html":
|
||||
html_content = content_bytes.decode('utf-8')
|
||||
markdown_content = self._convert_html_to_markdown(html_content)
|
||||
markdown_content = await asyncio.to_thread(
|
||||
self._convert_html_to_markdown, html_content
|
||||
)
|
||||
elif mime_type == "application/pdf":
|
||||
markdown_content = self._convert_pdf_to_markdown(content_bytes)
|
||||
markdown_content = await asyncio.to_thread(
|
||||
self._convert_pdf_to_markdown, content_bytes
|
||||
)
|
||||
else:
|
||||
logger.warning(f"Unsupported mime type: {mime_type}")
|
||||
markdown_content = f"Unsupported content type: {mime_type}. Unable to convert to markdown."
|
||||
@@ -121,7 +248,7 @@ class BedestenApiClient:
|
||||
return BedestenDocumentMarkdown(
|
||||
documentId=document_id,
|
||||
markdown_content=markdown_content,
|
||||
source_url=f"{self.BASE_URL}/document/{document_id}",
|
||||
source_url=f"https://mevzuat.adalet.gov.tr/ictihat/{document_id}",
|
||||
mime_type=mime_type
|
||||
)
|
||||
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
# danistay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional
|
||||
@@ -170,7 +171,7 @@ class DanistayApiClient:
|
||||
source_url=source_url
|
||||
)
|
||||
|
||||
markdown_content = self._convert_html_to_markdown_danistay(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown_danistay, html_content_from_api)
|
||||
|
||||
return DanistayDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
@@ -1,66 +0,0 @@
|
||||
version: '3.8'
|
||||
|
||||
services:
|
||||
yargi-mcp:
|
||||
build: .
|
||||
image: yargi-mcp:latest
|
||||
container_name: yargi-mcp-server
|
||||
ports:
|
||||
- "${PORT:-8000}:8000"
|
||||
environment:
|
||||
- HOST=0.0.0.0
|
||||
- PORT=8000
|
||||
- LOG_LEVEL=${LOG_LEVEL:-info}
|
||||
- ALLOWED_ORIGINS=${ALLOWED_ORIGINS:-*}
|
||||
- API_TOKEN=${API_TOKEN:-}
|
||||
- PYTHONUNBUFFERED=1
|
||||
volumes:
|
||||
# Mount logs directory
|
||||
- ./logs:/app/logs
|
||||
# Mount .env file if it exists
|
||||
- ./.env:/app/.env:ro
|
||||
restart: unless-stopped
|
||||
healthcheck:
|
||||
test: ["CMD", "python", "-c", "import httpx; httpx.get('http://localhost:8000/health').raise_for_status()"]
|
||||
interval: 30s
|
||||
timeout: 10s
|
||||
retries: 3
|
||||
start_period: 10s
|
||||
networks:
|
||||
- yargi-network
|
||||
|
||||
# Optional: Nginx reverse proxy
|
||||
nginx:
|
||||
image: nginx:alpine
|
||||
container_name: yargi-nginx
|
||||
ports:
|
||||
- "80:80"
|
||||
- "443:443"
|
||||
volumes:
|
||||
- ./nginx.conf:/etc/nginx/nginx.conf:ro
|
||||
- ./ssl:/etc/nginx/ssl:ro
|
||||
depends_on:
|
||||
- yargi-mcp
|
||||
networks:
|
||||
- yargi-network
|
||||
profiles:
|
||||
- production
|
||||
|
||||
# Optional: Redis for caching (future enhancement)
|
||||
redis:
|
||||
image: redis:alpine
|
||||
container_name: yargi-redis
|
||||
command: redis-server --appendonly yes
|
||||
volumes:
|
||||
- redis-data:/data
|
||||
networks:
|
||||
- yargi-network
|
||||
profiles:
|
||||
- with-cache
|
||||
|
||||
networks:
|
||||
yargi-network:
|
||||
driver: bridge
|
||||
|
||||
volumes:
|
||||
redis-data:
|
||||
@@ -1,428 +0,0 @@
|
||||
# Yargı MCP Server Dağıtım Rehberi
|
||||
|
||||
Bu rehber, Yargı MCP Server'ın ASGI web servisi olarak çeşitli dağıtım seçeneklerini kapsar.
|
||||
|
||||
## İçindekiler
|
||||
|
||||
- [Hızlı Başlangıç](#hızlı-başlangıç)
|
||||
- [Yerel Geliştirme](#yerel-geliştirme)
|
||||
- [Production Dağıtımı](#production-dağıtımı)
|
||||
- [Cloud Dağıtımı](#cloud-dağıtımı)
|
||||
- [Docker Dağıtımı](#docker-dağıtımı)
|
||||
- [Güvenlik Hususları](#güvenlik-hususları)
|
||||
- [İzleme](#izleme)
|
||||
|
||||
## Hızlı Başlangıç
|
||||
|
||||
### 1. Bağımlılıkları Yükleyin
|
||||
|
||||
```bash
|
||||
# ASGI sunucusu için uvicorn yükleyin
|
||||
pip install uvicorn
|
||||
|
||||
# Veya tüm bağımlılıklarla birlikte yükleyin
|
||||
pip install -e .
|
||||
pip install uvicorn
|
||||
```
|
||||
|
||||
### 2. Sunucuyu Çalıştırın
|
||||
|
||||
```bash
|
||||
# Temel başlatma
|
||||
python run_asgi.py
|
||||
|
||||
# Veya doğrudan uvicorn ile
|
||||
uvicorn asgi_app:app --host 0.0.0.0 --port 8000
|
||||
```
|
||||
|
||||
Sunucu şu adreslerde kullanılabilir olacak:
|
||||
- MCP Endpoint: `http://localhost:8000/mcp/`
|
||||
- Sağlık Kontrolü: `http://localhost:8000/health`
|
||||
- API Durumu: `http://localhost:8000/status`
|
||||
|
||||
## Yerel Geliştirme
|
||||
|
||||
### Otomatik Yeniden Yükleme ile Geliştirme Sunucusu
|
||||
|
||||
```bash
|
||||
python run_asgi.py --reload --log-level debug
|
||||
```
|
||||
|
||||
### FastAPI Entegrasyonunu Kullanma
|
||||
|
||||
Ek REST API endpoint'leri için:
|
||||
|
||||
```bash
|
||||
uvicorn fastapi_app:app --reload
|
||||
```
|
||||
|
||||
Bu şunları sağlar:
|
||||
- `/docs` adresinde interaktif API dokümantasyonu
|
||||
- `/api/tools` adresinde araç listesi
|
||||
- `/api/databases` adresinde veritabanı bilgileri
|
||||
|
||||
### Ortam Değişkenleri
|
||||
|
||||
`.env.example` dosyasını temel alarak bir `.env` dosyası oluşturun:
|
||||
|
||||
```bash
|
||||
cp .env.example .env
|
||||
```
|
||||
|
||||
Temel değişkenler:
|
||||
- `HOST`: Sunucu host adresi (varsayılan: 127.0.0.1)
|
||||
- `PORT`: Sunucu portu (varsayılan: 8000)
|
||||
- `ALLOWED_ORIGINS`: CORS kökenleri (virgülle ayrılmış)
|
||||
- `LOG_LEVEL`: Log seviyesi (debug, info, warning, error)
|
||||
|
||||
## Production Dağıtımı
|
||||
|
||||
### 1. Uvicorn ile Çoklu Worker Kullanımı
|
||||
|
||||
```bash
|
||||
python run_asgi.py --host 0.0.0.0 --port 8000 --workers 4
|
||||
```
|
||||
|
||||
### 2. Gunicorn Kullanımı
|
||||
|
||||
```bash
|
||||
pip install gunicorn
|
||||
gunicorn asgi_app:app -w 4 -k uvicorn.workers.UvicornWorker --bind 0.0.0.0:8000
|
||||
```
|
||||
|
||||
### 3. Nginx Reverse Proxy ile
|
||||
|
||||
1. Nginx'i yükleyin
|
||||
2. Sağlanan `nginx.conf` dosyasını kullanın:
|
||||
|
||||
```bash
|
||||
sudo cp nginx.conf /etc/nginx/sites-available/yargi-mcp
|
||||
sudo ln -s /etc/nginx/sites-available/yargi-mcp /etc/nginx/sites-enabled/
|
||||
sudo nginx -t
|
||||
sudo systemctl reload nginx
|
||||
```
|
||||
|
||||
### 4. Systemd Servisi
|
||||
|
||||
`/etc/systemd/system/yargi-mcp.service` dosyasını oluşturun:
|
||||
|
||||
```ini
|
||||
[Unit]
|
||||
Description=Yargı MCP Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=exec
|
||||
User=www-data
|
||||
WorkingDirectory=/opt/yargi-mcp
|
||||
Environment="PATH=/opt/yargi-mcp/venv/bin"
|
||||
ExecStart=/opt/yargi-mcp/venv/bin/uvicorn asgi_app:app --host 0.0.0.0 --port 8000 --workers 4
|
||||
Restart=on-failure
|
||||
RestartSec=5
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
```
|
||||
|
||||
Etkinleştirin ve başlatın:
|
||||
|
||||
```bash
|
||||
sudo systemctl enable yargi-mcp
|
||||
sudo systemctl start yargi-mcp
|
||||
```
|
||||
|
||||
## Cloud Dağıtımı
|
||||
|
||||
### Heroku
|
||||
|
||||
1. `Procfile` oluşturun:
|
||||
```
|
||||
web: uvicorn asgi_app:app --host 0.0.0.0 --port $PORT
|
||||
```
|
||||
|
||||
2. Dağıtın:
|
||||
```bash
|
||||
heroku create uygulama-isminiz
|
||||
git push heroku main
|
||||
```
|
||||
|
||||
### Railway
|
||||
|
||||
1. `railway.json` ekleyin:
|
||||
```json
|
||||
{
|
||||
"build": {
|
||||
"builder": "NIXPACKS"
|
||||
},
|
||||
"deploy": {
|
||||
"startCommand": "uvicorn asgi_app:app --host 0.0.0.0 --port $PORT"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
2. Railway CLI veya GitHub entegrasyonu ile dağıtın
|
||||
|
||||
### Google Cloud Run
|
||||
|
||||
1. Container oluşturun:
|
||||
```bash
|
||||
docker build -t yargi-mcp .
|
||||
docker tag yargi-mcp gcr.io/PROJE_ADINIZ/yargi-mcp
|
||||
docker push gcr.io/PROJE_ADINIZ/yargi-mcp
|
||||
```
|
||||
|
||||
2. Dağıtın:
|
||||
```bash
|
||||
gcloud run deploy yargi-mcp \
|
||||
--image gcr.io/PROJE_ADINIZ/yargi-mcp \
|
||||
--platform managed \
|
||||
--region us-central1 \
|
||||
--allow-unauthenticated
|
||||
```
|
||||
|
||||
### AWS Lambda (Mangum kullanarak)
|
||||
|
||||
1. Mangum'u yükleyin:
|
||||
```bash
|
||||
pip install mangum
|
||||
```
|
||||
|
||||
2. `lambda_handler.py` oluşturun:
|
||||
```python
|
||||
from mangum import Mangum
|
||||
from asgi_app import app
|
||||
|
||||
handler = Mangum(app, lifespan="off")
|
||||
```
|
||||
|
||||
3. AWS SAM veya Serverless Framework kullanarak dağıtın
|
||||
|
||||
## Docker Dağıtımı
|
||||
|
||||
### Tek Container
|
||||
|
||||
```bash
|
||||
# Oluşturun
|
||||
docker build -t yargi-mcp .
|
||||
|
||||
# Çalıştırın
|
||||
docker run -p 8000:8000 --env-file .env yargi-mcp
|
||||
```
|
||||
|
||||
### Docker Compose
|
||||
|
||||
```bash
|
||||
# Geliştirme
|
||||
docker-compose up
|
||||
|
||||
# Nginx ile Production
|
||||
docker-compose --profile production up
|
||||
|
||||
# Redis önbellekleme ile
|
||||
docker-compose --profile with-cache up
|
||||
```
|
||||
|
||||
### Kubernetes
|
||||
|
||||
Deployment YAML oluşturun:
|
||||
|
||||
```yaml
|
||||
apiVersion: apps/v1
|
||||
kind: Deployment
|
||||
metadata:
|
||||
name: yargi-mcp
|
||||
spec:
|
||||
replicas: 3
|
||||
selector:
|
||||
matchLabels:
|
||||
app: yargi-mcp
|
||||
template:
|
||||
metadata:
|
||||
labels:
|
||||
app: yargi-mcp
|
||||
spec:
|
||||
containers:
|
||||
- name: yargi-mcp
|
||||
image: yargi-mcp:latest
|
||||
ports:
|
||||
- containerPort: 8000
|
||||
env:
|
||||
- name: HOST
|
||||
value: "0.0.0.0"
|
||||
- name: PORT
|
||||
value: "8000"
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: 8000
|
||||
initialDelaySeconds: 10
|
||||
periodSeconds: 30
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: yargi-mcp-service
|
||||
spec:
|
||||
selector:
|
||||
app: yargi-mcp
|
||||
ports:
|
||||
- port: 80
|
||||
targetPort: 8000
|
||||
type: LoadBalancer
|
||||
```
|
||||
|
||||
## Güvenlik Hususları
|
||||
|
||||
### 1. Kimlik Doğrulama
|
||||
|
||||
`API_TOKEN` ortam değişkenini ayarlayarak token kimlik doğrulamasını etkinleştirin:
|
||||
|
||||
```bash
|
||||
export API_TOKEN=gizli-token-degeri
|
||||
```
|
||||
|
||||
Ardından isteklere ekleyin:
|
||||
```bash
|
||||
curl -H "Authorization: Bearer gizli-token-degeri" http://localhost:8000/api/tools
|
||||
```
|
||||
|
||||
### 2. HTTPS/SSL
|
||||
|
||||
Production için her zaman HTTPS kullanın:
|
||||
|
||||
1. SSL sertifikası edinin (Let's Encrypt vb.)
|
||||
2. Nginx veya cloud sağlayıcıda yapılandırın
|
||||
3. `ALLOWED_ORIGINS` değerini https:// kullanacak şekilde güncelleyin
|
||||
|
||||
### 3. Rate Limiting (Hız Sınırlama)
|
||||
|
||||
Sağlanan Nginx yapılandırması rate limiting içerir:
|
||||
- API endpoint'leri: 10 istek/saniye
|
||||
- MCP endpoint: 100 istek/saniye
|
||||
|
||||
### 4. CORS Yapılandırması
|
||||
|
||||
Production için belirli kaynaklara izin verin:
|
||||
|
||||
```bash
|
||||
ALLOWED_ORIGINS=https://app.sizindomain.com,https://www.sizindomain.com
|
||||
```
|
||||
|
||||
## İzleme
|
||||
|
||||
### Sağlık Kontrolleri
|
||||
|
||||
`/health` endpoint'ini izleyin:
|
||||
|
||||
```bash
|
||||
curl http://localhost:8000/health
|
||||
```
|
||||
|
||||
Yanıt:
|
||||
```json
|
||||
{
|
||||
"status": "healthy",
|
||||
"timestamp": "2024-12-26T10:00:00",
|
||||
"uptime_seconds": 3600,
|
||||
"tools_operational": true
|
||||
}
|
||||
```
|
||||
|
||||
### Loglama
|
||||
|
||||
Ortam değişkeni ile log seviyesini yapılandırın:
|
||||
|
||||
```bash
|
||||
LOG_LEVEL=info # veya debug, warning, error
|
||||
```
|
||||
|
||||
Loglar şuraya yazılır:
|
||||
- Konsol (stdout)
|
||||
- `logs/mcp_server.log` dosyası
|
||||
|
||||
### Metrikler (Opsiyonel)
|
||||
|
||||
OpenTelemetry desteği için:
|
||||
|
||||
```bash
|
||||
pip install opentelemetry-instrumentation-fastapi
|
||||
```
|
||||
|
||||
Ortam değişkenlerini ayarlayın:
|
||||
```bash
|
||||
OTEL_EXPORTER_OTLP_ENDPOINT=http://localhost:4317
|
||||
OTEL_SERVICE_NAME=yargi-mcp-server
|
||||
```
|
||||
|
||||
## Sorun Giderme
|
||||
|
||||
### Port Zaten Kullanımda
|
||||
|
||||
```bash
|
||||
# 8000 portunu kullanan işlemi bulun
|
||||
lsof -i :8000
|
||||
|
||||
# İşlemi sonlandırın
|
||||
kill -9 <PID>
|
||||
```
|
||||
|
||||
### İzin Hataları
|
||||
|
||||
Dosya izinlerinin doğru olduğundan emin olun:
|
||||
|
||||
```bash
|
||||
chmod +x run_asgi.py
|
||||
chown -R www-data:www-data /opt/yargi-mcp
|
||||
```
|
||||
|
||||
### Bellek Sorunları
|
||||
|
||||
Büyük belge işleme için worker belleğini artırın:
|
||||
|
||||
```bash
|
||||
# systemd servisinde
|
||||
Environment="PYTHONMALLOC=malloc"
|
||||
LimitNOFILE=65536
|
||||
```
|
||||
|
||||
### Zaman Aşımı Sorunları
|
||||
|
||||
Zaman aşımlarını ayarlayın:
|
||||
1. Uvicorn: `--timeout-keep-alive 75`
|
||||
2. Nginx: `proxy_read_timeout 300s;`
|
||||
3. Cloud sağlayıcılar: Platform özel zaman aşımı ayarlarını kontrol edin
|
||||
|
||||
## Performans Ayarlama
|
||||
|
||||
### 1. Worker İşlemleri
|
||||
|
||||
- Geliştirme: 1 worker
|
||||
- Production: CPU çekirdeği başına 2-4 worker
|
||||
|
||||
### 2. Bağlantı Havuzlama
|
||||
|
||||
Sunucu varsayılan olarak httpx ile bağlantı havuzlama kullanır.
|
||||
|
||||
### 3. Önbellekleme (Gelecek Geliştirme)
|
||||
|
||||
Redis önbellekleme docker-compose ile etkinleştirilebilir:
|
||||
|
||||
```bash
|
||||
docker-compose --profile with-cache up
|
||||
```
|
||||
|
||||
### 4. Veritabanı Zaman Aşımları
|
||||
|
||||
`.env` dosyasında veritabanı başına zaman aşımlarını ayarlayın:
|
||||
|
||||
```bash
|
||||
YARGITAY_TIMEOUT=60
|
||||
DANISTAY_TIMEOUT=60
|
||||
ANAYASA_TIMEOUT=90
|
||||
```
|
||||
|
||||
## Destek
|
||||
|
||||
Sorunlar ve sorular için:
|
||||
- GitHub Issues: https://github.com/saidsurucu/yargi-mcp/issues
|
||||
- Dokümantasyon: README.md dosyasına bakın
|
||||
@@ -1,5 +1,6 @@
|
||||
# emsal_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
# from bs4 import BeautifulSoup # Uncomment if needed for advanced HTML pre-processing
|
||||
from typing import Dict, Any, List, Optional
|
||||
@@ -153,7 +154,7 @@ class EmsalApiClient:
|
||||
logger.warning(f"EmsalApiClient: Received empty or non-string HTML in 'data' field for Emsal ID {id}.")
|
||||
return EmsalDocumentMarkdown(id=id, markdown_content=None, source_url=source_url)
|
||||
|
||||
markdown_content = self._clean_html_and_convert_to_markdown_emsal(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._clean_html_and_convert_to_markdown_emsal, html_content_from_api)
|
||||
|
||||
return EmsalDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
@@ -84,7 +84,7 @@ class EmsalApiResponseInnerData(BaseModel):
|
||||
|
||||
class EmsalApiResponse(BaseModel):
|
||||
"""Model for the complete search response from the Emsal API."""
|
||||
data: EmsalApiResponseInnerData
|
||||
data: Optional[EmsalApiResponseInnerData] = None
|
||||
metadata: Optional[Dict[str, Any]] = Field(None, description="Optional metadata (Meta Veri) from API, if any.")
|
||||
|
||||
class EmsalDocumentMarkdown(BaseModel):
|
||||
|
||||
@@ -1,34 +0,0 @@
|
||||
# fly.toml app configuration file generated for yargi-mcp on 2025-06-29T00:23:47+03:00
|
||||
#
|
||||
# See https://fly.io/docs/reference/configuration/ for information about how to use this file.
|
||||
#
|
||||
|
||||
app = 'yargi-mcp'
|
||||
primary_region = 'fra'
|
||||
|
||||
[env]
|
||||
ENABLE_AUTH = "true"
|
||||
HOST = "0.0.0.0"
|
||||
PORT = "8000"
|
||||
LOG_LEVEL = "info"
|
||||
|
||||
[build]
|
||||
|
||||
[http_service]
|
||||
internal_port = 8000
|
||||
force_https = true
|
||||
auto_stop_machines = 'off'
|
||||
auto_start_machines = true
|
||||
min_machines_running = 1
|
||||
processes = ['app']
|
||||
|
||||
[[vm]]
|
||||
memory = '1gb'
|
||||
cpu_kind = 'shared'
|
||||
cpus = 1
|
||||
|
||||
[checks.http_health] # keep MCP /health live
|
||||
type = "http"
|
||||
interval = "30s"
|
||||
timeout = "10s"
|
||||
path = "/health"
|
||||
@@ -0,0 +1 @@
|
||||
# gib_mcp_module/__init__.py
|
||||
@@ -0,0 +1,355 @@
|
||||
# gib_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
import io
|
||||
import logging
|
||||
import math
|
||||
from typing import Optional, Any, Dict
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
GibSearchRequest,
|
||||
GibOzelgeSummary,
|
||||
GibSearchResult,
|
||||
GibDocumentMarkdown,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
|
||||
class GibApiClient:
|
||||
"""
|
||||
API client for searching and retrieving GİB özelgeler (Turkish Revenue
|
||||
Administration tax rulings) via the public gib.gov.tr JSON API.
|
||||
|
||||
The endpoint is a single POST list endpoint; document retrieval is done
|
||||
by filtering the same endpoint with an exact `id`.
|
||||
"""
|
||||
|
||||
BASE_URL = "https://gib.gov.tr/api"
|
||||
LIST_PATH = "/gibportal/mevzuat/ozelge/list"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
# Fixed filter values required by the backend
|
||||
_REQUIRED_STATUS = 2
|
||||
_REQUIRED_DELETED = False
|
||||
_REQUIRED_KTYPE = 99 # ktype=99 selects özelge
|
||||
_SORT_FIELD = "ozelgeTarih"
|
||||
_SORT_TYPE = "DESC"
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en;q=0.7",
|
||||
"Content-Type": "application/json",
|
||||
"User-Agent": "Mozilla/5.0 (compatible; yargi-mcp/1.0; +https://github.com/saidsurucu/yargi-mcp)",
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=True,
|
||||
follow_redirects=True,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _normalize_date(value: str, end_of_day: bool = False) -> Optional[str]:
|
||||
"""
|
||||
Accept 'YYYY-MM-DD' or full ISO 8601; always return full ISO 8601.
|
||||
|
||||
GİB backend rejects date-only strings.
|
||||
"""
|
||||
if not value:
|
||||
return None
|
||||
v = value.strip()
|
||||
if not v:
|
||||
return None
|
||||
# Already ISO with time component
|
||||
if "T" in v:
|
||||
return v
|
||||
# Simple YYYY-MM-DD - expand to start/end of day
|
||||
suffix = "T23:59:59.999Z" if end_of_day else "T00:00:00.000Z"
|
||||
return f"{v}{suffix}"
|
||||
|
||||
def _build_search_body(self, params: GibSearchRequest) -> Dict[str, Any]:
|
||||
body: Dict[str, Any] = {
|
||||
"status": self._REQUIRED_STATUS,
|
||||
"deleted": self._REQUIRED_DELETED,
|
||||
"ktype": self._REQUIRED_KTYPE,
|
||||
}
|
||||
|
||||
keywords = params.keywords.strip()
|
||||
kanun_no = params.kanunNo.strip()
|
||||
# Frontend sets title/kanunNo/description to the SAME value; the backend
|
||||
# ORs across them. If the caller supplies both, combine them so kanun_no
|
||||
# still biases toward ruling text, while keywords remain primary.
|
||||
search_term = keywords or kanun_no
|
||||
if keywords and kanun_no and kanun_no not in keywords:
|
||||
search_term = f"{keywords} {kanun_no}"
|
||||
if search_term:
|
||||
body["title"] = search_term
|
||||
body["kanunNo"] = search_term
|
||||
body["description"] = search_term
|
||||
|
||||
if params.ozelgeNo.strip():
|
||||
body["ozelgeNo"] = params.ozelgeNo.strip()
|
||||
|
||||
if params.kanunId and params.kanunId > 0:
|
||||
body["kanunIds"] = [params.kanunId]
|
||||
|
||||
start_iso = self._normalize_date(params.ozelgeStartDate, end_of_day=False)
|
||||
end_iso = self._normalize_date(params.ozelgeEndDate, end_of_day=True)
|
||||
if start_iso:
|
||||
body["ozelgeStartDate"] = start_iso
|
||||
if end_iso:
|
||||
body["ozelgeEndDate"] = end_iso
|
||||
|
||||
return body
|
||||
|
||||
def _build_query_params(self, page_1_indexed: int, page_size: int) -> Dict[str, Any]:
|
||||
# API expects 0-indexed page
|
||||
zero_indexed = max(0, page_1_indexed - 1)
|
||||
return {
|
||||
"page": zero_indexed,
|
||||
"size": page_size,
|
||||
"sortFieldName": self._SORT_FIELD,
|
||||
"sortType": self._SORT_TYPE,
|
||||
}
|
||||
|
||||
@staticmethod
|
||||
def _to_summary(item: Dict[str, Any]) -> Optional[GibOzelgeSummary]:
|
||||
if not isinstance(item, dict):
|
||||
return None
|
||||
raw_id = item.get("id")
|
||||
if raw_id is None:
|
||||
return None
|
||||
try:
|
||||
ozelge_id = int(raw_id)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
return GibOzelgeSummary(
|
||||
id=ozelge_id,
|
||||
ozelgeNo=item.get("ozelgeNo"),
|
||||
ozelgeTarih=item.get("ozelgeTarih"),
|
||||
title=item.get("title"),
|
||||
kanunNo=item.get("kanunNo"),
|
||||
kanunTitle=item.get("kanunTitle"),
|
||||
siteLink=item.get("siteLink"),
|
||||
)
|
||||
|
||||
async def search_ozelge(self, params: GibSearchRequest) -> GibSearchResult:
|
||||
"""Search GİB özelgeler."""
|
||||
body = self._build_search_body(params)
|
||||
query = self._build_query_params(params.page, params.pageSize)
|
||||
logger.info(
|
||||
"GibApiClient: search page=%s size=%s body_keys=%s",
|
||||
params.page, params.pageSize, sorted(body.keys()),
|
||||
)
|
||||
|
||||
try:
|
||||
resp = await self.http_client.post(self.LIST_PATH, params=query, json=body)
|
||||
resp.raise_for_status()
|
||||
payload = resp.json()
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error("GibApiClient: HTTP %s during search", e.response.status_code)
|
||||
return GibSearchResult(
|
||||
ozelgeler=[],
|
||||
total_results=0,
|
||||
total_pages=0,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error("GibApiClient: search request failed: %s", e)
|
||||
return GibSearchResult(
|
||||
ozelgeler=[],
|
||||
total_results=0,
|
||||
total_pages=0,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
|
||||
container = (payload or {}).get("resultContainer") or {}
|
||||
raw_items = container.get("content") or []
|
||||
|
||||
summaries = []
|
||||
for raw in raw_items:
|
||||
summary = self._to_summary(raw)
|
||||
if summary is not None:
|
||||
summaries.append(summary)
|
||||
|
||||
total_results = container.get("totalElements") or 0
|
||||
total_pages = container.get("totalPages") or 0
|
||||
try:
|
||||
total_results = int(total_results)
|
||||
except (TypeError, ValueError):
|
||||
total_results = 0
|
||||
try:
|
||||
total_pages = int(total_pages)
|
||||
except (TypeError, ValueError):
|
||||
total_pages = 0
|
||||
|
||||
return GibSearchResult(
|
||||
ozelgeler=summaries,
|
||||
total_results=total_results,
|
||||
total_pages=total_pages,
|
||||
current_page=params.page,
|
||||
page_size=params.pageSize,
|
||||
)
|
||||
|
||||
def _convert_html_to_markdown(self, html_content: str) -> Optional[str]:
|
||||
"""Convert HTML content to Markdown using MarkItDown with BytesIO."""
|
||||
if not html_content:
|
||||
return None
|
||||
try:
|
||||
html_bytes = html_content.encode("utf-8")
|
||||
html_stream = io.BytesIO(html_bytes)
|
||||
md_converter = MarkItDown(enable_plugins=False)
|
||||
result = md_converter.convert(html_stream)
|
||||
return result.text_content
|
||||
except Exception as e:
|
||||
logger.error("GibApiClient: HTML→Markdown conversion failed: %s", e)
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _build_header_block(item: Dict[str, Any]) -> str:
|
||||
"""Build a small Markdown header block summarising the ruling metadata."""
|
||||
parts = []
|
||||
title = item.get("title")
|
||||
if title:
|
||||
parts.append(f"# {title}")
|
||||
meta_lines = []
|
||||
if item.get("ozelgeNo"):
|
||||
meta_lines.append(f"**Sayı:** {item['ozelgeNo']}")
|
||||
if item.get("ozelgeTarih"):
|
||||
meta_lines.append(f"**Tarih:** {item['ozelgeTarih']}")
|
||||
if item.get("kanunTitle"):
|
||||
kanun_no = item.get("kanunNo")
|
||||
if kanun_no:
|
||||
meta_lines.append(f"**Kanun:** {item['kanunTitle']} ({kanun_no})")
|
||||
else:
|
||||
meta_lines.append(f"**Kanun:** {item['kanunTitle']}")
|
||||
if item.get("siteLink"):
|
||||
meta_lines.append(f"**Kaynak:** {item['siteLink']}")
|
||||
if meta_lines:
|
||||
parts.append("\n".join(meta_lines))
|
||||
return "\n\n".join(parts).strip()
|
||||
|
||||
async def get_ozelge_document(
|
||||
self, ozelge_id: int, page_number: int = 1
|
||||
) -> GibDocumentMarkdown:
|
||||
"""Retrieve a single özelge and return its paginated Markdown form."""
|
||||
logger.info(
|
||||
"GibApiClient: fetching özelge id=%s page=%s", ozelge_id, page_number
|
||||
)
|
||||
|
||||
if not isinstance(ozelge_id, int) or ozelge_id <= 0:
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id if isinstance(ozelge_id, int) else 0,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="ozelge_id must be a positive integer",
|
||||
)
|
||||
|
||||
body = {
|
||||
"status": self._REQUIRED_STATUS,
|
||||
"deleted": self._REQUIRED_DELETED,
|
||||
"ktype": self._REQUIRED_KTYPE,
|
||||
"id": ozelge_id,
|
||||
}
|
||||
query = {"page": 0, "size": 1}
|
||||
|
||||
try:
|
||||
resp = await self.http_client.post(self.LIST_PATH, params=query, json=body)
|
||||
resp.raise_for_status()
|
||||
payload = resp.json()
|
||||
except httpx.HTTPStatusError as e:
|
||||
msg = f"HTTP {e.response.status_code} when fetching özelge {ozelge_id}"
|
||||
logger.error("GibApiClient: %s", msg)
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=msg,
|
||||
)
|
||||
except Exception as e:
|
||||
msg = f"Request failed: {e}"
|
||||
logger.error("GibApiClient: %s", msg)
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=msg,
|
||||
)
|
||||
|
||||
container = (payload or {}).get("resultContainer") or {}
|
||||
content = container.get("content") or []
|
||||
if not content:
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message=f"Özelge {ozelge_id} not found",
|
||||
)
|
||||
|
||||
item = content[0] if isinstance(content[0], dict) else {}
|
||||
description_html = item.get("description") or ""
|
||||
markdown_body = (await asyncio.to_thread(self._convert_html_to_markdown, description_html)) or ""
|
||||
header_block = self._build_header_block(item)
|
||||
|
||||
if header_block and markdown_body:
|
||||
full_markdown = f"{header_block}\n\n---\n\n{markdown_body}"
|
||||
else:
|
||||
full_markdown = header_block or markdown_body
|
||||
|
||||
if not full_markdown.strip():
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
ozelge_no=item.get("ozelgeNo"),
|
||||
title=item.get("title"),
|
||||
ozelge_tarih=item.get("ozelgeTarih"),
|
||||
kanun_title=item.get("kanunTitle"),
|
||||
kanun_no=item.get("kanunNo"),
|
||||
site_link=item.get("siteLink"),
|
||||
current_page=page_number,
|
||||
total_pages=0,
|
||||
is_paginated=False,
|
||||
error_message="Document body is empty",
|
||||
)
|
||||
|
||||
total_pages = max(
|
||||
1, math.ceil(len(full_markdown) / self.DOCUMENT_MARKDOWN_CHUNK_SIZE)
|
||||
)
|
||||
current_page_clamped = max(1, min(page_number, total_pages))
|
||||
start = (current_page_clamped - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end = start + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
chunk = full_markdown[start:end]
|
||||
|
||||
return GibDocumentMarkdown(
|
||||
ozelge_id=ozelge_id,
|
||||
ozelge_no=item.get("ozelgeNo"),
|
||||
title=item.get("title"),
|
||||
ozelge_tarih=item.get("ozelgeTarih"),
|
||||
kanun_title=item.get("kanunTitle"),
|
||||
kanun_no=item.get("kanunNo"),
|
||||
site_link=item.get("siteLink"),
|
||||
markdown_chunk=chunk,
|
||||
current_page=current_page_clamped,
|
||||
total_pages=total_pages,
|
||||
is_paginated=total_pages > 1,
|
||||
error_message=None,
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
if hasattr(self, "http_client") and self.http_client and not self.http_client.is_closed:
|
||||
await self.http_client.aclose()
|
||||
logger.info("GibApiClient: HTTP client session closed.")
|
||||
@@ -0,0 +1,64 @@
|
||||
# gib_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List, Optional
|
||||
|
||||
|
||||
class GibSearchRequest(BaseModel):
|
||||
"""
|
||||
Request model for searching GİB özelgeler (Turkish Revenue Administration tax rulings).
|
||||
|
||||
GİB (Gelir İdaresi Başkanlığı) publishes official tax-ruling letters
|
||||
("özelge") responding to taxpayer questions on VAT, income tax,
|
||||
corporate tax, stamp duty, and other tax matters. 18,000+ rulings
|
||||
are searchable via the public gib.gov.tr API.
|
||||
"""
|
||||
keywords: str = Field("", description="Keywords searched across title, kanunNo and description (Turkish)")
|
||||
ozelgeNo: str = Field("", description="Exact özelge reference number (e.g., 'E-40247694-130-15524')")
|
||||
kanunNo: str = Field("", description="Law number filter, e.g. '3065' for KDV")
|
||||
kanunId: int = Field(0, description="Optional numeric law ID filter (0=ignore)")
|
||||
ozelgeStartDate: str = Field("", description="Start date YYYY-MM-DD or full ISO 8601")
|
||||
ozelgeEndDate: str = Field("", description="End date YYYY-MM-DD or full ISO 8601")
|
||||
page: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)")
|
||||
|
||||
|
||||
class GibOzelgeSummary(BaseModel):
|
||||
"""Summary of a single GİB özelge from search results (no full HTML)."""
|
||||
id: int = Field(..., description="Numeric özelge ID for document retrieval")
|
||||
ozelgeNo: Optional[str] = Field(None, description="Official ruling reference number")
|
||||
ozelgeTarih: Optional[str] = Field(None, description="Ruling date (ISO datetime)")
|
||||
title: Optional[str] = Field(None, description="Subject/title of the ruling")
|
||||
kanunNo: Optional[str] = Field(None, description="Law number (e.g., '3065')")
|
||||
kanunTitle: Optional[str] = Field(None, description="Law title (e.g., 'KATMA DEĞER VERGİSİ KANUNU')")
|
||||
siteLink: Optional[str] = Field(None, description="Direct URL to the ruling on gib.gov.tr")
|
||||
|
||||
|
||||
class GibSearchResult(BaseModel):
|
||||
"""Response model for GİB özelge search results."""
|
||||
ozelgeler: List[GibOzelgeSummary] = Field(default_factory=list, description="Matching özelge summaries")
|
||||
total_results: int = Field(0, description="Total number of matching özelgeler across all pages")
|
||||
total_pages: int = Field(0, description="Total number of pages for this query")
|
||||
current_page: int = Field(1, description="Current page (1-indexed)")
|
||||
page_size: int = Field(10, description="Results per page")
|
||||
|
||||
|
||||
class GibDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
GİB özelge document converted to paginated Markdown.
|
||||
|
||||
Long rulings are split into 5000-character chunks; request successive
|
||||
pages via page_number to read the full text.
|
||||
"""
|
||||
ozelge_id: int = Field(..., description="Numeric özelge ID")
|
||||
ozelge_no: Optional[str] = Field(None, description="Official ruling reference number")
|
||||
title: Optional[str] = Field(None, description="Subject/title of the ruling")
|
||||
ozelge_tarih: Optional[str] = Field(None, description="Ruling date (ISO datetime)")
|
||||
kanun_title: Optional[str] = Field(None, description="Related law title")
|
||||
kanun_no: Optional[str] = Field(None, description="Related law number")
|
||||
site_link: Optional[str] = Field(None, description="Direct URL to the ruling on gib.gov.tr")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Current 5000-character Markdown chunk")
|
||||
current_page: int = Field(1, description="Current page number (1-indexed)")
|
||||
total_pages: int = Field(0, description="Total pages for the full Markdown content")
|
||||
is_paginated: bool = Field(False, description="True if split across multiple pages")
|
||||
error_message: Optional[str] = Field(None, description="Populated when retrieval failed")
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,507 @@
|
||||
# kik_mcp_module/client_v2.py
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import httpx
|
||||
import logging
|
||||
import uuid
|
||||
import ssl
|
||||
import os
|
||||
from typing import Optional
|
||||
from datetime import datetime
|
||||
|
||||
# Cryptography imports for AES-256-CBC encryption of document IDs
|
||||
try:
|
||||
from cryptography.hazmat.primitives.ciphers import Cipher, algorithms, modes
|
||||
from cryptography.hazmat.backends import default_backend
|
||||
HAS_CRYPTOGRAPHY = True
|
||||
except ImportError:
|
||||
HAS_CRYPTOGRAPHY = False
|
||||
|
||||
from .models_v2 import (
|
||||
KikV2DecisionType, KikV2SearchPayload, KikV2SearchPayloadDk, KikV2SearchPayloadMk,
|
||||
KikV2RequestData, KikV2QueryRequest, KikV2KeyValuePair,
|
||||
KikV2SearchResponse, KikV2SearchResponseDk, KikV2SearchResponseMk,
|
||||
KikV2SearchResult, KikV2CompactDecision, KikV2DocumentMarkdown
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
class KikV2ApiClient:
|
||||
"""
|
||||
New KIK v2 API Client for https://ekapv2.kik.gov.tr
|
||||
|
||||
This client uses the modern JSON-based API endpoint that provides
|
||||
better structured data compared to the legacy form-based API.
|
||||
"""
|
||||
|
||||
BASE_URL = "https://ekapv2.kik.gov.tr"
|
||||
|
||||
# Endpoint mappings for different decision types
|
||||
ENDPOINTS = {
|
||||
KikV2DecisionType.UYUSMAZLIK: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlari",
|
||||
KikV2DecisionType.DUZENLEYICI: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariDk",
|
||||
KikV2DecisionType.MAHKEME: "/b_ihalearaclari/api/KurulKararlari/GetKurulKararlariMk"
|
||||
}
|
||||
|
||||
# AES-256-CBC encryption key for document ID encryption (reverse engineered from ekapv2.kik.gov.tr Angular app)
|
||||
# This key is used to encrypt numeric gundemMaddesiId values to 64-character hex hashes for document URLs
|
||||
DOCUMENT_ID_ENCRYPTION_KEY = bytes([
|
||||
236, 193, 164, 43, 12, 135, 121, 170, 4, 244, 123, 219, 82, 158, 124, 174,
|
||||
174, 228, 219, 174, 208, 104, 174, 120, 32, 76, 250, 4, 143, 159, 211, 176
|
||||
])
|
||||
|
||||
# AES-192-CBC key (environment.r8fact) used by the Angular HTTP interceptor to sign every
|
||||
# request. The server decrypts X-Custom-Request-Ts and rejects stale timestamps with
|
||||
# HTTP 401 "İstek zaman aşımına uğradı.", so these headers MUST be generated per-request
|
||||
# with the current timestamp (see _generate_security_headers).
|
||||
REQUEST_SIGNING_KEY = b"Qm2LtXR0aByP69vZNKef4wMJ" # UTF-8 bytes, 24 chars -> AES-192
|
||||
|
||||
@staticmethod
|
||||
def encrypt_document_id(numeric_id: str) -> str:
|
||||
"""
|
||||
Encrypt a numeric KİK gundemMaddesiId to the 64-character hex hash
|
||||
used in document URLs.
|
||||
|
||||
Algorithm: AES-256-CBC with PKCS7 padding
|
||||
Output format: IV (16 bytes hex) + Ciphertext (16 bytes hex) = 64 chars
|
||||
|
||||
Args:
|
||||
numeric_id: The numeric document ID from search results (e.g., "177280")
|
||||
|
||||
Returns:
|
||||
64-character hex string for use in document URL KararId parameter
|
||||
"""
|
||||
if not HAS_CRYPTOGRAPHY:
|
||||
raise ImportError("cryptography library required for document ID encryption")
|
||||
|
||||
# Generate random IV (16 bytes)
|
||||
iv = os.urandom(16)
|
||||
|
||||
# Create AES-CBC cipher with the encryption key
|
||||
cipher = Cipher(
|
||||
algorithms.AES(KikV2ApiClient.DOCUMENT_ID_ENCRYPTION_KEY),
|
||||
modes.CBC(iv),
|
||||
backend=default_backend()
|
||||
)
|
||||
encryptor = cipher.encryptor()
|
||||
|
||||
# Encode plaintext and apply PKCS7 padding
|
||||
plaintext = numeric_id.encode('utf-8')
|
||||
block_size = 16
|
||||
padding_len = block_size - (len(plaintext) % block_size)
|
||||
padded_plaintext = plaintext + bytes([padding_len] * padding_len)
|
||||
|
||||
# Encrypt
|
||||
ciphertext = encryptor.update(padded_plaintext) + encryptor.finalize()
|
||||
|
||||
# Return IV + ciphertext as lowercase hex (64 characters total)
|
||||
return iv.hex() + ciphertext.hex()
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
# Create SSL context with legacy server support
|
||||
ssl_context = ssl.create_default_context()
|
||||
ssl_context.check_hostname = False
|
||||
ssl_context.verify_mode = ssl.CERT_NONE
|
||||
|
||||
# Enable legacy server connect option for older SSL implementations (Python 3.12+)
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
|
||||
# Set broader cipher suite support including legacy ciphers
|
||||
ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
verify=ssl_context,
|
||||
headers={
|
||||
"Accept": "application/json",
|
||||
"Accept-Language": "tr",
|
||||
"Content-Type": "application/json",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": f"{self.BASE_URL}/sorgulamalar/kurul-kararlari",
|
||||
"Sec-Fetch-Dest": "empty",
|
||||
"Sec-Fetch-Mode": "cors",
|
||||
"Sec-Fetch-Site": "same-origin",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36",
|
||||
"api-version": "v1",
|
||||
"sec-ch-ua": '"Not;A=Brand";v="99", "Google Chrome";v="139", "Chromium";v="139"',
|
||||
"sec-ch-ua-mobile": "?0",
|
||||
"sec-ch-ua-platform": '"macOS"'
|
||||
},
|
||||
timeout=request_timeout
|
||||
)
|
||||
|
||||
# Generate security headers (these might need to be updated based on API requirements)
|
||||
self.security_headers = self._generate_security_headers()
|
||||
|
||||
def _sign_request_value(self, plaintext: str, iv: bytes) -> str:
|
||||
"""AES-192-CBC encrypt a value with the request signing key, return base64 ciphertext."""
|
||||
cipher = Cipher(
|
||||
algorithms.AES(self.REQUEST_SIGNING_KEY),
|
||||
modes.CBC(iv),
|
||||
backend=default_backend()
|
||||
)
|
||||
encryptor = cipher.encryptor()
|
||||
data = plaintext.encode("utf-8")
|
||||
block_size = 16
|
||||
padding_len = block_size - (len(data) % block_size)
|
||||
padded = data + bytes([padding_len] * padding_len)
|
||||
ciphertext = encryptor.update(padded) + encryptor.finalize()
|
||||
return base64.b64encode(ciphertext).decode("ascii")
|
||||
|
||||
def _generate_security_headers(self) -> dict:
|
||||
"""
|
||||
Generate the custom security headers required by the KIK v2 API.
|
||||
|
||||
Mirrors the Angular HTTP interceptor on ekapv2.kik.gov.tr: a random GUID and a
|
||||
current-timestamp (epoch milliseconds) are AES-192-CBC encrypted with environment.r8fact
|
||||
using a fresh random IV. The IV is sent as -Siv, the encrypted timestamp as -Ts, and the
|
||||
encrypted GUID as -R8id. The server validates the decrypted timestamp's freshness, so these
|
||||
MUST be regenerated on every request; stale values yield HTTP 401 "İstek zaman aşımına uğradı.".
|
||||
"""
|
||||
if not HAS_CRYPTOGRAPHY:
|
||||
raise ImportError("cryptography library required for KIK v2 request signing")
|
||||
|
||||
request_guid = str(uuid.uuid4())
|
||||
iv = os.urandom(16)
|
||||
timestamp_ms = str(int(datetime.now().timestamp() * 1000))
|
||||
|
||||
return {
|
||||
"X-Custom-Request-Guid": request_guid,
|
||||
"X-Custom-Request-R8id": self._sign_request_value(request_guid, iv),
|
||||
"X-Custom-Request-Siv": base64.b64encode(iv).decode("ascii"),
|
||||
"X-Custom-Request-Ts": self._sign_request_value(timestamp_ms, iv),
|
||||
}
|
||||
|
||||
def _build_search_payload(self,
|
||||
decision_type: KikV2DecisionType,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = ""):
|
||||
"""Build the search payload for KIK v2 API."""
|
||||
|
||||
key_value_pairs = []
|
||||
|
||||
# Add non-empty search criteria
|
||||
if karar_metni:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=karar_metni))
|
||||
|
||||
if karar_no:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararNo", value=karar_no))
|
||||
|
||||
if basvuran:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BasvuranAdi", value=basvuran))
|
||||
|
||||
if idare_adi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="IdareAdi", value=idare_adi))
|
||||
|
||||
if baslangic_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BaslangicTarihi", value=baslangic_tarihi))
|
||||
|
||||
if bitis_tarihi:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="BitisTarihi", value=bitis_tarihi))
|
||||
|
||||
# If no search criteria provided, use a generic search
|
||||
if not key_value_pairs:
|
||||
key_value_pairs.append(KikV2KeyValuePair(key="KararMetni", value=""))
|
||||
|
||||
query_request = KikV2QueryRequest(keyValueOfstringanyType=key_value_pairs)
|
||||
request_data = KikV2RequestData(keyValuePairs=query_request)
|
||||
|
||||
# Return appropriate payload based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
return KikV2SearchPayload(sorgulaKurulKararlari=request_data)
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
return KikV2SearchPayloadDk(sorgulaKurulKararlariDk=request_data)
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
return KikV2SearchPayloadMk(sorgulaKurulKararlariMk=request_data)
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
async def search_decisions(self,
|
||||
decision_type: KikV2DecisionType = KikV2DecisionType.UYUSMAZLIK,
|
||||
karar_metni: str = "",
|
||||
karar_no: str = "",
|
||||
basvuran: str = "",
|
||||
idare_adi: str = "",
|
||||
baslangic_tarihi: str = "",
|
||||
bitis_tarihi: str = "") -> KikV2SearchResult:
|
||||
"""
|
||||
Search KIK decisions using the v2 API.
|
||||
|
||||
Args:
|
||||
decision_type: Type of decision to search (uyusmazlik/duzenleyici/mahkeme)
|
||||
karar_metni: Decision text search
|
||||
karar_no: Decision number (e.g., "2025/UH.II-1801")
|
||||
basvuran: Applicant name
|
||||
idare_adi: Administration name
|
||||
baslangic_tarihi: Start date (YYYY-MM-DD format)
|
||||
bitis_tarihi: End date (YYYY-MM-DD format)
|
||||
|
||||
Returns:
|
||||
KikV2SearchResult with compact decision list
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Searching {decision_type.value} decisions with criteria - karar_metni: '{karar_metni}', karar_no: '{karar_no}', basvuran: '{basvuran}'")
|
||||
|
||||
try:
|
||||
# Build request payload
|
||||
payload = self._build_search_payload(
|
||||
decision_type=decision_type,
|
||||
karar_metni=karar_metni,
|
||||
karar_no=karar_no,
|
||||
basvuran=basvuran,
|
||||
idare_adi=idare_adi,
|
||||
baslangic_tarihi=baslangic_tarihi,
|
||||
bitis_tarihi=bitis_tarihi
|
||||
)
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Get the appropriate endpoint for this decision type
|
||||
endpoint = self.ENDPOINTS[decision_type]
|
||||
|
||||
# Make API request
|
||||
response = await self.http_client.post(
|
||||
endpoint,
|
||||
json=payload.model_dump(),
|
||||
headers=headers
|
||||
)
|
||||
|
||||
response.raise_for_status()
|
||||
response_data = response.json()
|
||||
|
||||
logger.debug(f"KikV2ApiClient: Raw API response structure: {type(response_data)}")
|
||||
|
||||
# Parse the API response based on decision type
|
||||
if decision_type == KikV2DecisionType.UYUSMAZLIK:
|
||||
api_response = KikV2SearchResponse(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariResponse.SorgulaKurulKararlariResult
|
||||
elif decision_type == KikV2DecisionType.DUZENLEYICI:
|
||||
api_response = KikV2SearchResponseDk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariDkResponse.SorgulaKurulKararlariDkResult
|
||||
elif decision_type == KikV2DecisionType.MAHKEME:
|
||||
api_response = KikV2SearchResponseMk(**response_data)
|
||||
result_data = api_response.SorgulaKurulKararlariMkResponse.SorgulaKurulKararlariMkResult
|
||||
else:
|
||||
raise ValueError(f"Unsupported decision type: {decision_type}")
|
||||
|
||||
# Check for API errors
|
||||
if result_data.hataKodu and result_data.hataKodu != "0":
|
||||
logger.warning(f"KikV2ApiClient: API returned error - Code: {result_data.hataKodu}, Message: {result_data.hataMesaji}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code=result_data.hataKodu,
|
||||
error_message=result_data.hataMesaji
|
||||
)
|
||||
|
||||
# Convert to compact format
|
||||
compact_decisions = []
|
||||
total_count = 0
|
||||
|
||||
for decision_group in result_data.KurulKararTutanakDetayListesi:
|
||||
for decision_detail in decision_group.KurulKararTutanakDetayi:
|
||||
compact_decision = KikV2CompactDecision(
|
||||
kararNo=decision_detail.kararNo,
|
||||
kararTarihi=decision_detail.kararTarihi,
|
||||
basvuran=decision_detail.basvuran,
|
||||
idareAdi=decision_detail.idareAdi,
|
||||
basvuruKonusu=decision_detail.basvuruKonusu,
|
||||
gundemMaddesiId=decision_detail.gundemMaddesiId,
|
||||
decision_type=decision_type.value
|
||||
)
|
||||
compact_decisions.append(compact_decision)
|
||||
total_count += 1
|
||||
|
||||
logger.info(f"KikV2ApiClient: Found {total_count} decisions")
|
||||
|
||||
return KikV2SearchResult(
|
||||
decisions=compact_decisions,
|
||||
total_records=total_count,
|
||||
page=1,
|
||||
error_code="0",
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"KikV2ApiClient: HTTP error during search: {e.response.status_code} - {e.response.text}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="HTTP_ERROR",
|
||||
error_message=f"HTTP {e.response.status_code}: {e.response.text}"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Unexpected error during search: {str(e)}")
|
||||
return KikV2SearchResult(
|
||||
decisions=[],
|
||||
total_records=0,
|
||||
page=1,
|
||||
error_code="UNEXPECTED_ERROR",
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def get_document_markdown(self, document_id: str) -> KikV2DocumentMarkdown:
|
||||
"""
|
||||
Get KİK decision document content in Markdown format.
|
||||
|
||||
This method uses a two-step process:
|
||||
1. Call GetSorgulamaUrl endpoint to get the actual document URL
|
||||
2. Use httpx to fetch the document content
|
||||
|
||||
Args:
|
||||
document_id: The gundemMaddesiId from search results
|
||||
|
||||
Returns:
|
||||
KikV2DocumentMarkdown with document content converted to Markdown
|
||||
"""
|
||||
|
||||
logger.info(f"KikV2ApiClient: Getting document for ID: {document_id}")
|
||||
|
||||
if not document_id or not document_id.strip():
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Document ID is required"
|
||||
)
|
||||
|
||||
try:
|
||||
# Step 1: Get the actual document URL using GetSorgulamaUrl endpoint
|
||||
logger.info(f"KikV2ApiClient: Step 1 - Getting document URL for ID: {document_id}")
|
||||
|
||||
# Update security headers for this request
|
||||
headers = {**self.http_client.headers, **self._generate_security_headers()}
|
||||
|
||||
# Call GetSorgulamaUrl to get the real document URL
|
||||
url_payload = {"sorguSayfaTipi": 2} # As shown in curl example
|
||||
|
||||
url_response = await self.http_client.post(
|
||||
"/b_ihalearaclari/api/KurulKararlari/GetSorgulamaUrl",
|
||||
json=url_payload,
|
||||
headers=headers
|
||||
)
|
||||
|
||||
url_response.raise_for_status()
|
||||
url_data = url_response.json()
|
||||
|
||||
# Get the base document URL from API response
|
||||
base_document_url = url_data.get("sorgulamaUrl", "")
|
||||
if not base_document_url:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url="",
|
||||
error_message="Could not get document URL from GetSorgulamaUrl API"
|
||||
)
|
||||
|
||||
# If document_id is numeric, encrypt it to get the KararId hash
|
||||
# The web interface uses AES-256-CBC encrypted hashes for document URLs
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID {document_id} to hash: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt document ID, using as-is: {enc_error}")
|
||||
|
||||
# Construct full document URL with the encrypted KararId
|
||||
document_url = f"{base_document_url}?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Retrieved document URL: {document_url}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error getting document URL for ID {document_id}: {str(e)}")
|
||||
# Fallback to old method if GetSorgulamaUrl fails
|
||||
# Also encrypt numeric IDs in fallback path
|
||||
karar_id = document_id
|
||||
if document_id.isdigit():
|
||||
try:
|
||||
karar_id = self.encrypt_document_id(document_id)
|
||||
logger.info(f"KikV2ApiClient: Encrypted numeric ID in fallback: {karar_id}")
|
||||
except Exception as enc_error:
|
||||
logger.warning(f"KikV2ApiClient: Could not encrypt in fallback: {enc_error}")
|
||||
document_url = f"https://ekap.kik.gov.tr/EKAP/Vatandas/KurulKararGoster.aspx?KararId={karar_id}"
|
||||
logger.info(f"KikV2ApiClient: Falling back to direct URL: {document_url}")
|
||||
|
||||
try:
|
||||
# Step 2: Use httpx to get the document content
|
||||
logger.info(f"KikV2ApiClient: Step 2 - Using httpx to retrieve document from: {document_url}")
|
||||
|
||||
# Create a separate httpx client for document retrieval with HTML headers
|
||||
doc_ssl_context = ssl.create_default_context()
|
||||
doc_ssl_context.check_hostname = False
|
||||
doc_ssl_context.verify_mode = ssl.CERT_NONE
|
||||
if hasattr(ssl, 'OP_LEGACY_SERVER_CONNECT'):
|
||||
doc_ssl_context.options |= ssl.OP_LEGACY_SERVER_CONNECT
|
||||
doc_ssl_context.set_ciphers('ALL:!aNULL:!eNULL:!EXPORT:!DES:!RC4:!MD5:!PSK:!SRP:!CAMELLIA')
|
||||
|
||||
async with httpx.AsyncClient(
|
||||
verify=doc_ssl_context,
|
||||
headers={
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "tr,en-US;q=0.5",
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/139.0.0.0 Safari/537.36"
|
||||
},
|
||||
timeout=60.0,
|
||||
follow_redirects=True
|
||||
) as doc_client:
|
||||
response = await doc_client.get(document_url)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
logger.info(f"KikV2ApiClient: Retrieved content via httpx, length: {len(html_content)}")
|
||||
|
||||
# Convert HTML to Markdown using MarkItDown with BytesIO
|
||||
try:
|
||||
from markitdown import MarkItDown
|
||||
from io import BytesIO
|
||||
|
||||
md = MarkItDown()
|
||||
html_bytes = html_content.encode('utf-8')
|
||||
html_stream = BytesIO(html_bytes)
|
||||
|
||||
# markitdown is sync; offload to thread so HTML parsing doesn't
|
||||
# block the event-loop / other in-flight MCP requests.
|
||||
result = await asyncio.to_thread(md.convert_stream, html_stream, file_extension=".html")
|
||||
markdown_content = result.text_content
|
||||
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content=markdown_content,
|
||||
source_url=document_url,
|
||||
error_message=""
|
||||
)
|
||||
|
||||
except ImportError:
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="MarkItDown library not available",
|
||||
source_url=document_url,
|
||||
error_message="MarkItDown library not installed"
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"KikV2ApiClient: Error retrieving document {document_id}: {str(e)}")
|
||||
return KikV2DocumentMarkdown(
|
||||
document_id=document_id,
|
||||
kararNo="",
|
||||
markdown_content="",
|
||||
source_url=document_url,
|
||||
error_message=str(e)
|
||||
)
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("KikV2ApiClient: HTTP client session closed.")
|
||||
@@ -1,74 +0,0 @@
|
||||
# kik_mcp_module/models.py
|
||||
from pydantic import BaseModel, Field, HttpUrl, computed_field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from enum import Enum
|
||||
import base64 # Base64 encoding/decoding için
|
||||
|
||||
class KikKararTipi(str, Enum):
|
||||
"""Enum for KIK (Public Procurement Authority) Decision Types."""
|
||||
UYUSMAZLIK = "rbUyusmazlik"
|
||||
DUZENLEYICI = "rbDuzenleyici"
|
||||
MAHKEME = "rbMahkeme"
|
||||
|
||||
class KikSearchRequest(BaseModel):
|
||||
"""Model for KIK Decision search criteria."""
|
||||
karar_tipi: KikKararTipi = Field(KikKararTipi.UYUSMAZLIK, description="Type")
|
||||
karar_no: str = Field("", description="No")
|
||||
karar_tarihi_baslangic: str = Field("", description="Start", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
karar_tarihi_bitis: str = Field("", description="End", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
resmi_gazete_sayisi: str = Field("", description="Gazette")
|
||||
resmi_gazete_tarihi: str = Field("", description="Date", pattern=r"^\d{2}\.\d{2}\.\d{4}$|^$")
|
||||
basvuru_konusu_ihale: str = Field("", description="Subject")
|
||||
basvuru_sahibi: str = Field("", description="Applicant")
|
||||
ihaleyi_yapan_idare: str = Field("", description="Entity")
|
||||
yil: str = Field("", description="Year")
|
||||
karar_metni: str = Field("", description="Text")
|
||||
page: int = Field(1, ge=1, description="Page")
|
||||
|
||||
class KikDecisionEntry(BaseModel):
|
||||
"""Represents a single decision entry from KIK search results."""
|
||||
preview_event_target: str = Field(..., description="Event target")
|
||||
karar_no_str: str = Field(..., alias="kararNo", description="Decision number")
|
||||
karar_tipi: KikKararTipi = Field(..., description="Decision type")
|
||||
|
||||
karar_tarihi_str: str = Field(..., alias="kararTarihi", description="Date")
|
||||
idare_str: str = Field("", alias="idare", description="Entity")
|
||||
basvuru_sahibi_str: str = Field("", alias="basvuruSahibi", description="Applicant")
|
||||
ihale_konusu_str: str = Field("", alias="ihaleKonusu", description="Subject")
|
||||
|
||||
@computed_field
|
||||
@property
|
||||
def karar_id(self) -> str:
|
||||
"""
|
||||
A Base64 encoded unique ID for the decision, combining decision type and number.
|
||||
Format before encoding: "{karar_tipi.value}|{karar_no_str}"
|
||||
"""
|
||||
combined_key = f"{self.karar_tipi.value}|{self.karar_no_str}"
|
||||
return base64.b64encode(combined_key.encode('utf-8')).decode('utf-8')
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikSearchResult(BaseModel):
|
||||
"""Model for KIK search results."""
|
||||
decisions: List[KikDecisionEntry]
|
||||
total_records: int = 0
|
||||
current_page: int = 1
|
||||
|
||||
class KikDocumentMarkdown(BaseModel):
|
||||
"""
|
||||
KIK decision document, with Markdown content potentially paginated.
|
||||
"""
|
||||
retrieved_with_karar_id: Optional[str] = Field(None, description="Request ID")
|
||||
retrieved_karar_no: Optional[str] = Field(None, description="Decision number")
|
||||
retrieved_karar_tipi: Optional[KikKararTipi] = Field(None, description="Decision type")
|
||||
|
||||
karar_id_param_from_url: Optional[str] = Field(None, alias="kararIdParam", description="Internal ID")
|
||||
markdown_chunk: Optional[str] = Field(None, description="Content")
|
||||
source_url: Optional[str] = Field(None, description="Source URL")
|
||||
error_message: Optional[str] = Field(None, description="Error")
|
||||
current_page: int = Field(1, description="Page")
|
||||
total_pages: int = Field(1, description="Total pages")
|
||||
is_paginated: bool = Field(False, description="Paginated")
|
||||
full_content_char_count: Optional[int] = Field(None, description="Char count")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
@@ -0,0 +1,147 @@
|
||||
# kik_mcp_module/models_v2.py
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
from typing import List, Optional
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
|
||||
# New KIK v2 API Models
|
||||
|
||||
class KikV2DecisionType(str, Enum):
|
||||
"""KIK v2 Decision Types with corresponding endpoints."""
|
||||
UYUSMAZLIK = "uyusmazlik" # Disputes - GetKurulKararlari
|
||||
DUZENLEYICI = "duzenleyici" # Regulatory - GetKurulKararlariDk
|
||||
MAHKEME = "mahkeme" # Court - GetKurulKararlariMk
|
||||
|
||||
class KikV2SearchRequest(BaseModel):
|
||||
"""Model for KIK v2 API search request."""
|
||||
KararMetni: str = Field("", description="Decision text search query")
|
||||
KararNo: str = Field("", description="Decision number (e.g., '2025/UH.II-1801')")
|
||||
BasvuranAdi: str = Field("", description="Applicant name")
|
||||
IdareAdi: str = Field("", description="Administration name")
|
||||
BaslangicTarihi: str = Field("", description="Start date (YYYY-MM-DD)")
|
||||
BitisTarihi: str = Field("", description="End date (YYYY-MM-DD)")
|
||||
|
||||
class KikV2KeyValuePair(BaseModel):
|
||||
"""Key-value pair for KIK v2 API request."""
|
||||
key: str
|
||||
value: str
|
||||
|
||||
class KikV2QueryRequest(BaseModel):
|
||||
"""Nested query structure for KIK v2 API."""
|
||||
keyValueOfstringanyType: List[KikV2KeyValuePair]
|
||||
|
||||
class KikV2RequestData(BaseModel):
|
||||
"""Main request data structure for KIK v2 API."""
|
||||
keyValuePairs: KikV2QueryRequest
|
||||
|
||||
# Request Payloads for different decision types
|
||||
class KikV2SearchPayload(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Uyuşmazlık (Disputes)."""
|
||||
sorgulaKurulKararlari: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadDk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Düzenleyici (Regulatory)."""
|
||||
sorgulaKurulKararlariDk: KikV2RequestData
|
||||
|
||||
class KikV2SearchPayloadMk(BaseModel):
|
||||
"""Complete payload for KIK v2 API search - Mahkeme (Court)."""
|
||||
sorgulaKurulKararlariMk: KikV2RequestData
|
||||
|
||||
# Response Models
|
||||
|
||||
class KikV2DecisionDetail(BaseModel):
|
||||
"""Individual decision detail from KIK v2 API response."""
|
||||
resmiGazeteMukerrerSayi: str = Field("", description="Official Gazette duplicate number")
|
||||
itiraz: str = Field("", description="Objection")
|
||||
yayinlanmaTarihi: str = Field("", description="Publication date")
|
||||
idareAdi: str = Field("", description="Administration name")
|
||||
uzmanTCKN: str = Field("", description="Expert TCKN")
|
||||
resmiGazeteTarihi: str = Field("", description="Official Gazette date")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
kararTurKod: str = Field("", description="Decision type code")
|
||||
kararTurAciklama: str = Field("", description="Decision type description")
|
||||
karar: str = Field("", description="Decision text")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
resmiGazeteSayisi: str = Field("", description="Official Gazette number")
|
||||
inceleme: str = Field("", description="Review")
|
||||
basvuruTarihi: str = Field("", description="Application date")
|
||||
kararNitelikKod: str = Field("", description="Decision nature code")
|
||||
resmiGazeteMukerrer: str = Field("", description="Official Gazette duplicate")
|
||||
basvuruSayisi: str = Field("", description="Application number")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
kararNitelik: str = Field("", description="Decision nature")
|
||||
uyusmazlikKararNo: str = Field("", description="Dispute decision number")
|
||||
kurulNo: str = Field("", description="Board number")
|
||||
gundemMaddesiSiraNo: str = Field("", description="Agenda item sequence")
|
||||
kararTarihi: str = Field("", description="Decision date (ISO format)")
|
||||
dosyaBirimKodu: str = Field("", description="File unit code")
|
||||
gundemMaddesiId: str = Field("", description="Agenda item ID")
|
||||
|
||||
class KikV2DecisionGroup(BaseModel):
|
||||
"""Group of decision details."""
|
||||
KurulKararTutanakDetayi: List[KikV2DecisionDetail] = Field(alias="kurulKararTutanakDetayi")
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultData(BaseModel):
|
||||
"""Search result data structure."""
|
||||
hataKodu: str = Field("", description="Error code")
|
||||
hataMesaji: str = Field("", description="Error message")
|
||||
KurulKararTutanakDetayListesi: List[KikV2DecisionGroup]
|
||||
|
||||
model_config = ConfigDict(populate_by_name=True)
|
||||
|
||||
class KikV2SearchResultWrapper(BaseModel):
|
||||
"""Wrapper for search result."""
|
||||
SorgulaKurulKararlariResult: KikV2SearchResultData
|
||||
|
||||
# Base Response Models
|
||||
class KikV2SearchResponse(BaseModel):
|
||||
"""Complete KIK v2 API search response for Uyuşmazlık (Disputes)."""
|
||||
SorgulaKurulKararlariResponse: KikV2SearchResultWrapper
|
||||
|
||||
# Düzenleyici Kararlar (Regulatory Decisions) Response Models
|
||||
class KikV2SearchResultWrapperDk(BaseModel):
|
||||
"""Wrapper for regulatory decisions search result."""
|
||||
SorgulaKurulKararlariDkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseDk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Düzenleyici (Regulatory) decisions."""
|
||||
SorgulaKurulKararlariDkResponse: KikV2SearchResultWrapperDk
|
||||
|
||||
# Mahkeme Kararlar (Court Decisions) Response Models
|
||||
class KikV2SearchResultWrapperMk(BaseModel):
|
||||
"""Wrapper for court decisions search result."""
|
||||
SorgulaKurulKararlariMkResult: KikV2SearchResultData
|
||||
|
||||
class KikV2SearchResponseMk(BaseModel):
|
||||
"""Complete KIK v2 API search response for Mahkeme (Court) decisions."""
|
||||
SorgulaKurulKararlariMkResponse: KikV2SearchResultWrapperMk
|
||||
|
||||
# Simplified Models for MCP Tools
|
||||
|
||||
class KikV2CompactDecision(BaseModel):
|
||||
"""Compact decision format for MCP tool responses."""
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
kararTarihi: str = Field("", description="Decision date")
|
||||
basvuran: str = Field("", description="Applicant")
|
||||
idareAdi: str = Field("", description="Administration")
|
||||
basvuruKonusu: str = Field("", description="Application subject")
|
||||
gundemMaddesiId: str = Field("", description="Document ID for retrieval")
|
||||
decision_type: str = Field("", description="Decision type (uyusmazlik/duzenleyici/mahkeme)")
|
||||
|
||||
class KikV2SearchResult(BaseModel):
|
||||
"""Compact search results for MCP tools."""
|
||||
decisions: List[KikV2CompactDecision]
|
||||
total_records: int = Field(0, description="Total number of decisions found")
|
||||
page: int = Field(1, description="Current page number")
|
||||
error_code: str = Field("", description="API error code")
|
||||
error_message: str = Field("", description="API error message")
|
||||
|
||||
class KikV2DocumentMarkdown(BaseModel):
|
||||
"""Document content in Markdown format."""
|
||||
document_id: str = Field("", description="Document ID")
|
||||
kararNo: str = Field("", description="Decision number")
|
||||
markdown_content: str = Field("", description="Decision content in Markdown")
|
||||
source_url: str = Field("", description="Source URL")
|
||||
error_message: str = Field("", description="Error message if retrieval failed")
|
||||
@@ -1,5 +1,6 @@
|
||||
# kvkk_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Dict, Any
|
||||
@@ -291,7 +292,7 @@ class KvkkApiClient:
|
||||
# Convert HTML content to Markdown
|
||||
full_markdown_content = None
|
||||
if extracted_data["html_content"]:
|
||||
full_markdown_content = self._convert_html_to_markdown(extracted_data["html_content"])
|
||||
full_markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, extracted_data["html_content"])
|
||||
|
||||
if not full_markdown_content:
|
||||
return KvkkDocumentMarkdown(
|
||||
|
||||
@@ -1,28 +0,0 @@
|
||||
"""
|
||||
MCP Auth Toolkit - OAuth 2.1 + Authorization for Model Context Protocol Servers
|
||||
Integrated with Clerk Authentication
|
||||
"""
|
||||
|
||||
from .middleware import (
|
||||
AuthContext,
|
||||
FastMCPAuthWrapper,
|
||||
MCPAuthMiddleware,
|
||||
auth_required,
|
||||
)
|
||||
from .oauth import OAuthConfig, OAuthProvider
|
||||
from .policy import PolicyEngine, ToolPolicy, create_default_policies
|
||||
from .storage import PersistentStorage
|
||||
|
||||
__version__ = "0.1.0"
|
||||
__all__ = [
|
||||
"OAuthProvider",
|
||||
"OAuthConfig",
|
||||
"AuthContext",
|
||||
"auth_required",
|
||||
"create_default_policies",
|
||||
"MCPAuthMiddleware",
|
||||
"FastMCPAuthWrapper",
|
||||
"PolicyEngine",
|
||||
"ToolPolicy",
|
||||
"PersistentStorage",
|
||||
]
|
||||
@@ -1,73 +0,0 @@
|
||||
"""
|
||||
Clerk OAuth configuration for MCP Auth Toolkit
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
from .oauth import OAuthConfig
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def create_clerk_oauth_config() -> OAuthConfig:
|
||||
"""Create OAuth configuration for Clerk integration using SDK"""
|
||||
|
||||
# Get Clerk configuration from environment
|
||||
clerk_domain = os.getenv("CLERK_DOMAIN", "accounts.yargimcp.com")
|
||||
clerk_publishable_key = os.getenv("CLERK_PUBLISHABLE_KEY")
|
||||
clerk_secret_key = os.getenv("CLERK_SECRET_KEY")
|
||||
|
||||
if not clerk_publishable_key or not clerk_secret_key:
|
||||
raise ValueError("CLERK_PUBLISHABLE_KEY and CLERK_SECRET_KEY are required")
|
||||
|
||||
# For Clerk with custom domains, we use our adapter endpoints
|
||||
# This allows us to handle the custom domain flow properly
|
||||
base_url = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
config = OAuthConfig(
|
||||
client_id=clerk_publishable_key,
|
||||
client_secret=clerk_secret_key,
|
||||
# Use our adapter endpoints instead of Clerk's direct endpoints
|
||||
authorization_endpoint=f"{base_url}/authorize",
|
||||
token_endpoint=f"{base_url}/token",
|
||||
# Keep Clerk's JWKS for token validation
|
||||
jwks_uri=f"https://{clerk_domain}/.well-known/jwks.json",
|
||||
issuer=base_url, # We're the issuer for MCP tokens
|
||||
scopes=["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"]
|
||||
)
|
||||
|
||||
logger.info(f"Created Clerk OAuth config with adapter endpoints")
|
||||
logger.info(f"Clerk domain: {clerk_domain}")
|
||||
logger.debug(f"Authorization endpoint: {config.authorization_endpoint}")
|
||||
logger.debug(f"Token endpoint: {config.token_endpoint}")
|
||||
|
||||
return config
|
||||
|
||||
|
||||
def get_jwt_secret() -> str:
|
||||
"""Get JWT secret for token signing"""
|
||||
jwt_secret = os.getenv("JWT_SECRET_KEY")
|
||||
|
||||
if not jwt_secret:
|
||||
raise ValueError("JWT_SECRET_KEY environment variable is required")
|
||||
|
||||
return jwt_secret
|
||||
|
||||
|
||||
def create_mcp_server_config():
|
||||
"""Create complete MCP server configuration for Clerk integration"""
|
||||
|
||||
try:
|
||||
oauth_config = create_clerk_oauth_config()
|
||||
jwt_secret = get_jwt_secret()
|
||||
|
||||
return {
|
||||
"oauth_config": oauth_config,
|
||||
"jwt_secret": jwt_secret,
|
||||
"base_url": os.getenv("BASE_URL", "https://yargi-mcp.fly.dev"),
|
||||
"auth_enabled": os.getenv("ENABLE_AUTH", "true").lower() == "true"
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create MCP server config: {e}")
|
||||
raise
|
||||
@@ -1,315 +0,0 @@
|
||||
"""
|
||||
MCP server middleware for OAuth authentication and authorization
|
||||
"""
|
||||
|
||||
import functools
|
||||
import logging
|
||||
from collections.abc import Callable
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from fastmcp import FastMCP
|
||||
FASTMCP_AVAILABLE = True
|
||||
except ImportError:
|
||||
FASTMCP_AVAILABLE = False
|
||||
FastMCP = None
|
||||
logger.warning("FastMCP not available, some features will be disabled")
|
||||
|
||||
from .oauth import OAuthProvider
|
||||
from .policy import PolicyEngine
|
||||
|
||||
|
||||
@dataclass
|
||||
class AuthContext:
|
||||
"""Authentication context passed to MCP tools"""
|
||||
|
||||
user_id: str
|
||||
scopes: list[str]
|
||||
claims: dict[str, Any]
|
||||
token: str
|
||||
|
||||
|
||||
class MCPAuthMiddleware:
|
||||
"""Authentication middleware for MCP servers"""
|
||||
|
||||
def __init__(self, oauth_provider: OAuthProvider, policy_engine: PolicyEngine):
|
||||
self.oauth_provider = oauth_provider
|
||||
self.policy_engine = policy_engine
|
||||
|
||||
def authenticate_request(self, authorization_header: str) -> AuthContext | None:
|
||||
"""Extract and validate auth token from request"""
|
||||
|
||||
if not authorization_header:
|
||||
logger.debug("No authorization header provided")
|
||||
return None
|
||||
|
||||
if not authorization_header.startswith("Bearer "):
|
||||
logger.debug("Authorization header does not start with 'Bearer '")
|
||||
return None
|
||||
|
||||
token = authorization_header[7:] # Remove 'Bearer ' prefix
|
||||
|
||||
token_info = self.oauth_provider.introspect_token(token)
|
||||
|
||||
if not token_info.get("active"):
|
||||
logger.warning("Token is not active")
|
||||
return None
|
||||
|
||||
logger.debug(f"Authenticated user: {token_info.get('sub', 'unknown')}")
|
||||
|
||||
return AuthContext(
|
||||
user_id=token_info.get("sub", "unknown"),
|
||||
scopes=token_info.get("mcp_tool_scopes", []),
|
||||
claims=token_info,
|
||||
token=token,
|
||||
)
|
||||
|
||||
def authorize_tool_call(
|
||||
self, tool_name: str, auth_context: AuthContext
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Check if user can call the specified tool"""
|
||||
|
||||
return self.policy_engine.authorize_tool_call(
|
||||
tool_name=tool_name,
|
||||
user_scopes=auth_context.scopes,
|
||||
user_claims=auth_context.claims,
|
||||
)
|
||||
|
||||
|
||||
def auth_required(
|
||||
oauth_provider: OAuthProvider,
|
||||
policy_engine: PolicyEngine,
|
||||
tool_name: str | None = None,
|
||||
):
|
||||
"""
|
||||
Decorator to require authentication for MCP tool functions
|
||||
|
||||
Usage:
|
||||
@auth_required(oauth_provider, policy_engine, "search_yargitay")
|
||||
def my_tool_function(context: AuthContext, ...):
|
||||
pass
|
||||
"""
|
||||
|
||||
def decorator(func: Callable) -> Callable:
|
||||
middleware = MCPAuthMiddleware(oauth_provider, policy_engine)
|
||||
|
||||
@functools.wraps(func)
|
||||
async def wrapper(*args, **kwargs):
|
||||
# Extract authorization header from kwargs
|
||||
auth_header = kwargs.pop("authorization", None)
|
||||
|
||||
# Also check in args if it's a Request object
|
||||
if not auth_header and args:
|
||||
for arg in args:
|
||||
if hasattr(arg, 'headers'):
|
||||
auth_header = arg.headers.get("Authorization")
|
||||
break
|
||||
|
||||
if not auth_header:
|
||||
logger.warning(f"No authorization header for tool '{tool_name or func.__name__}'")
|
||||
raise PermissionError("Authorization header required")
|
||||
|
||||
auth_context = middleware.authenticate_request(auth_header)
|
||||
|
||||
if not auth_context:
|
||||
logger.warning(f"Authentication failed for tool '{tool_name or func.__name__}'")
|
||||
raise PermissionError("Invalid or expired token")
|
||||
|
||||
actual_tool_name = tool_name or func.__name__
|
||||
|
||||
authorized, reason = middleware.authorize_tool_call(
|
||||
actual_tool_name, auth_context
|
||||
)
|
||||
|
||||
if not authorized:
|
||||
logger.warning(f"Authorization failed for tool '{actual_tool_name}': {reason}")
|
||||
raise PermissionError(f"Access denied: {reason}")
|
||||
|
||||
# Add auth context to function call
|
||||
return await func(auth_context, *args, **kwargs)
|
||||
|
||||
return wrapper
|
||||
|
||||
return decorator
|
||||
|
||||
|
||||
class FastMCPAuthWrapper:
|
||||
"""Wrapper for FastMCP servers to add authentication"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
mcp_server: "FastMCP",
|
||||
oauth_provider: OAuthProvider,
|
||||
policy_engine: PolicyEngine,
|
||||
):
|
||||
if not FASTMCP_AVAILABLE:
|
||||
raise ImportError("FastMCP is required for FastMCPAuthWrapper")
|
||||
|
||||
self.mcp_server = mcp_server
|
||||
self.middleware = MCPAuthMiddleware(oauth_provider, policy_engine)
|
||||
self.oauth_provider = oauth_provider
|
||||
logger.info("Initializing FastMCP authentication wrapper")
|
||||
self._wrap_tools()
|
||||
|
||||
def _wrap_tools(self):
|
||||
"""Wrap all existing tools with auth middleware"""
|
||||
|
||||
# Try different FastMCP tool storage locations
|
||||
tool_registry = None
|
||||
|
||||
if hasattr(self.mcp_server, '_tools'):
|
||||
tool_registry = self.mcp_server._tools
|
||||
elif hasattr(self.mcp_server, 'tools'):
|
||||
tool_registry = self.mcp_server.tools
|
||||
elif hasattr(self.mcp_server, '_tool_registry'):
|
||||
tool_registry = self.mcp_server._tool_registry
|
||||
elif hasattr(self.mcp_server, '_handlers') and hasattr(self.mcp_server._handlers, 'tools'):
|
||||
tool_registry = self.mcp_server._handlers.tools
|
||||
|
||||
if not tool_registry:
|
||||
logger.warning("FastMCP server tool registry not found, tools will not be automatically wrapped")
|
||||
logger.debug(f"Available server attributes: {dir(self.mcp_server)}")
|
||||
return
|
||||
|
||||
logger.debug(f"Found tool registry with {len(tool_registry)} tools")
|
||||
original_tools = dict(tool_registry)
|
||||
wrapped_count = 0
|
||||
|
||||
for tool_name, tool_func in original_tools.items():
|
||||
try:
|
||||
wrapped_func = self._create_auth_wrapper(tool_name, tool_func)
|
||||
tool_registry[tool_name] = wrapped_func
|
||||
wrapped_count += 1
|
||||
logger.debug(f"Wrapped tool: {tool_name}")
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to wrap tool {tool_name}: {e}")
|
||||
|
||||
logger.info(f"Successfully wrapped {wrapped_count} tools with authentication")
|
||||
|
||||
def _create_auth_wrapper(self, tool_name: str, original_func: Callable) -> Callable:
|
||||
"""Create auth wrapper for a specific tool"""
|
||||
|
||||
@functools.wraps(original_func)
|
||||
async def auth_wrapper(*args, **kwargs):
|
||||
# Extract authorization from various sources
|
||||
auth_header = None
|
||||
|
||||
# Check kwargs first
|
||||
auth_header = kwargs.pop("authorization", None)
|
||||
|
||||
# Check if first argument is a Request object
|
||||
if not auth_header and args:
|
||||
first_arg = args[0]
|
||||
if hasattr(first_arg, 'headers'):
|
||||
auth_header = first_arg.headers.get("Authorization")
|
||||
|
||||
if not auth_header:
|
||||
logger.warning(f"No authorization header for tool '{tool_name}'")
|
||||
raise PermissionError("Authorization required")
|
||||
|
||||
auth_context = self.middleware.authenticate_request(auth_header)
|
||||
|
||||
if not auth_context:
|
||||
logger.warning(f"Authentication failed for tool '{tool_name}'")
|
||||
raise PermissionError("Invalid token")
|
||||
|
||||
authorized, reason = self.middleware.authorize_tool_call(
|
||||
tool_name, auth_context
|
||||
)
|
||||
|
||||
if not authorized:
|
||||
logger.warning(f"Authorization failed for tool '{tool_name}': {reason}")
|
||||
raise PermissionError(f"Access denied: {reason}")
|
||||
|
||||
# Add auth context to kwargs
|
||||
kwargs["auth_context"] = auth_context
|
||||
logger.debug(f"Calling tool '{tool_name}' for user {auth_context.user_id}")
|
||||
|
||||
return await original_func(*args, **kwargs)
|
||||
|
||||
return auth_wrapper
|
||||
|
||||
def add_oauth_endpoints(self):
|
||||
"""Add OAuth endpoints to the MCP server"""
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Initiate OAuth 2.1 authorization flow with PKCE",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_authorize(redirect_uri: str, scopes: Optional[str] = None):
|
||||
"""OAuth authorization endpoint"""
|
||||
scope_list = scopes.split(" ") if scopes else None
|
||||
auth_url, pkce = self.oauth_provider.generate_authorization_url(
|
||||
redirect_uri=redirect_uri, scopes=scope_list
|
||||
)
|
||||
logger.info(f"Generated authorization URL for redirect_uri: {redirect_uri}")
|
||||
return {
|
||||
"authorization_url": auth_url,
|
||||
"code_verifier": pkce.verifier, # For PKCE flow
|
||||
"code_challenge": pkce.challenge,
|
||||
"instructions": "Use the authorization_url to complete OAuth flow, then exchange the returned code using oauth_token tool"
|
||||
}
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Exchange OAuth authorization code for access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_token(
|
||||
code: str,
|
||||
state: str,
|
||||
redirect_uri: str
|
||||
):
|
||||
"""OAuth token exchange endpoint"""
|
||||
try:
|
||||
result = await self.oauth_provider.exchange_code_for_token(
|
||||
code=code, state=state, redirect_uri=redirect_uri
|
||||
)
|
||||
logger.info("Successfully exchanged authorization code for token")
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.error(f"Token exchange failed: {e}")
|
||||
raise
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Validate and introspect OAuth access token",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_introspect(token: str):
|
||||
"""Token introspection endpoint"""
|
||||
result = self.oauth_provider.introspect_token(token)
|
||||
logger.debug(f"Token introspection: active={result.get('active', False)}")
|
||||
return result
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Revoke OAuth access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_revoke(token: str):
|
||||
"""Token revocation endpoint"""
|
||||
success = self.oauth_provider.revoke_token(token)
|
||||
logger.info(f"Token revocation: success={success}")
|
||||
return {"revoked": success}
|
||||
|
||||
@self.mcp_server.tool(
|
||||
description="Get list of tools available to authenticated user",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_user_tools(authorization: str):
|
||||
"""Get user's allowed tools based on scopes"""
|
||||
auth_context = self.middleware.authenticate_request(authorization)
|
||||
if not auth_context:
|
||||
raise PermissionError("Invalid token")
|
||||
|
||||
allowed_patterns = self.middleware.policy_engine.get_allowed_tools(auth_context.scopes)
|
||||
|
||||
return {
|
||||
"user_id": auth_context.user_id,
|
||||
"scopes": auth_context.scopes,
|
||||
"allowed_tool_patterns": allowed_patterns,
|
||||
"message": "Use these patterns to determine which tools you can access"
|
||||
}
|
||||
|
||||
logger.info("Added OAuth endpoints: oauth_authorize, oauth_token, oauth_introspect, oauth_revoke, oauth_user_tools")
|
||||
@@ -1,304 +0,0 @@
|
||||
"""
|
||||
OAuth 2.1 + PKCE implementation for MCP servers with Clerk integration
|
||||
"""
|
||||
|
||||
import base64
|
||||
import hashlib
|
||||
import secrets
|
||||
import time
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta
|
||||
from typing import Any, Optional
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import httpx
|
||||
import jwt
|
||||
from jwt.exceptions import PyJWTError, InvalidTokenError
|
||||
|
||||
from .storage import PersistentStorage
|
||||
|
||||
# Try to import Clerk SDK
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class OAuthConfig:
|
||||
"""OAuth provider configuration for Clerk"""
|
||||
|
||||
client_id: str
|
||||
client_secret: str
|
||||
authorization_endpoint: str
|
||||
token_endpoint: str
|
||||
jwks_uri: str | None = None
|
||||
issuer: str = "mcp-auth"
|
||||
scopes: list[str] = None
|
||||
|
||||
def __post_init__(self):
|
||||
if self.scopes is None:
|
||||
self.scopes = ["mcp:tools:read", "mcp:tools:write"]
|
||||
|
||||
|
||||
class PKCEChallenge:
|
||||
"""PKCE challenge/verifier pair for OAuth 2.1"""
|
||||
|
||||
def __init__(self):
|
||||
self.verifier = (
|
||||
base64.urlsafe_b64encode(secrets.token_bytes(32))
|
||||
.decode("utf-8")
|
||||
.rstrip("=")
|
||||
)
|
||||
|
||||
challenge_bytes = hashlib.sha256(self.verifier.encode("utf-8")).digest()
|
||||
self.challenge = (
|
||||
base64.urlsafe_b64encode(challenge_bytes).decode("utf-8").rstrip("=")
|
||||
)
|
||||
|
||||
|
||||
class OAuthProvider:
|
||||
"""OAuth 2.1 provider with PKCE support and Clerk integration"""
|
||||
|
||||
def __init__(self, config: OAuthConfig, jwt_secret: str):
|
||||
self.config = config
|
||||
self.jwt_secret = jwt_secret
|
||||
# Use persistent storage instead of memory
|
||||
self.storage = PersistentStorage()
|
||||
|
||||
# Initialize Clerk SDK if available
|
||||
self.clerk = None
|
||||
if CLERK_AVAILABLE and config.client_secret:
|
||||
try:
|
||||
self.clerk = Clerk(bearer_auth=config.client_secret)
|
||||
logger.info("Clerk SDK initialized successfully")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to initialize Clerk SDK: {e}")
|
||||
|
||||
logger.info("OAuth provider initialized with persistent storage")
|
||||
|
||||
def generate_authorization_url(
|
||||
self,
|
||||
redirect_uri: str,
|
||||
state: str | None = None,
|
||||
scopes: list[str] | None = None,
|
||||
) -> tuple[str, PKCEChallenge]:
|
||||
"""Generate OAuth authorization URL with PKCE for Clerk"""
|
||||
|
||||
pkce = PKCEChallenge()
|
||||
session_id = secrets.token_urlsafe(32)
|
||||
|
||||
if state is None:
|
||||
state = secrets.token_urlsafe(16)
|
||||
|
||||
if scopes is None:
|
||||
scopes = self.config.scopes
|
||||
|
||||
# Store session data with expiration
|
||||
session_data = {
|
||||
"pkce_verifier": pkce.verifier,
|
||||
"state": state,
|
||||
"redirect_uri": redirect_uri,
|
||||
"scopes": scopes,
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=10)).timestamp(),
|
||||
}
|
||||
self.storage.set_session(session_id, session_data)
|
||||
|
||||
# Build Clerk OAuth URL
|
||||
# Check if this is a custom domain (sign-in endpoint)
|
||||
if self.config.authorization_endpoint.endswith('/sign-in'):
|
||||
# For custom domains, Clerk expects redirect_url parameter
|
||||
params = {
|
||||
"redirect_url": redirect_uri,
|
||||
"state": f"{state}:{session_id}",
|
||||
}
|
||||
auth_url = f"{self.config.authorization_endpoint}?{urlencode(params)}"
|
||||
else:
|
||||
# Standard OAuth flow with PKCE
|
||||
params = {
|
||||
"response_type": "code",
|
||||
"client_id": self.config.client_id,
|
||||
"redirect_uri": redirect_uri,
|
||||
"scope": " ".join(scopes),
|
||||
"state": f"{state}:{session_id}", # Combine state with session ID
|
||||
"code_challenge": pkce.challenge,
|
||||
"code_challenge_method": "S256",
|
||||
}
|
||||
auth_url = f"{self.config.authorization_endpoint}?{urlencode(params)}"
|
||||
|
||||
logger.info(f"Generated OAuth URL with session {session_id[:8]}...")
|
||||
logger.debug(f"Auth URL: {auth_url}")
|
||||
return auth_url, pkce
|
||||
|
||||
async def exchange_code_for_token(
|
||||
self, code: str, state: str, redirect_uri: str
|
||||
) -> dict[str, Any]:
|
||||
"""Exchange authorization code for access token with Clerk"""
|
||||
|
||||
try:
|
||||
original_state, session_id = state.split(":", 1)
|
||||
except ValueError as e:
|
||||
logger.error(f"Invalid state format: {state}")
|
||||
raise ValueError("Invalid state format") from e
|
||||
|
||||
session = self.storage.get_session(session_id)
|
||||
if not session:
|
||||
logger.error(f"Session {session_id} not found")
|
||||
raise ValueError("Invalid session")
|
||||
|
||||
# Check session expiration
|
||||
if datetime.utcnow().timestamp() > session.get("expires_at", 0):
|
||||
self.storage.delete_session(session_id)
|
||||
logger.error(f"Session {session_id} expired")
|
||||
raise ValueError("Session expired")
|
||||
|
||||
if session["state"] != original_state:
|
||||
logger.error(f"State mismatch: expected {session['state']}, got {original_state}")
|
||||
raise ValueError("State mismatch")
|
||||
|
||||
if session["redirect_uri"] != redirect_uri:
|
||||
logger.error(f"Redirect URI mismatch: expected {session['redirect_uri']}, got {redirect_uri}")
|
||||
raise ValueError("Redirect URI mismatch")
|
||||
|
||||
# Prepare token exchange request for Clerk
|
||||
token_data = {
|
||||
"grant_type": "authorization_code",
|
||||
"client_id": self.config.client_id,
|
||||
"client_secret": self.config.client_secret,
|
||||
"code": code,
|
||||
"redirect_uri": redirect_uri,
|
||||
"code_verifier": session["pkce_verifier"],
|
||||
}
|
||||
|
||||
logger.info(f"Exchanging code with Clerk for session {session_id[:8]}...")
|
||||
|
||||
async with httpx.AsyncClient() as client:
|
||||
response = await client.post(
|
||||
self.config.token_endpoint,
|
||||
data=token_data,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded"},
|
||||
timeout=30.0,
|
||||
)
|
||||
|
||||
if response.status_code != 200:
|
||||
logger.error(f"Clerk token exchange failed: {response.status_code} - {response.text}")
|
||||
raise ValueError(f"Token exchange failed: {response.text}")
|
||||
|
||||
token_response = response.json()
|
||||
logger.info("Successfully exchanged code for Clerk token")
|
||||
|
||||
# Create MCP-scoped JWT token
|
||||
access_token = self._create_mcp_token(
|
||||
session["scopes"], token_response.get("access_token"), session_id
|
||||
)
|
||||
|
||||
# Store token for introspection
|
||||
token_id = secrets.token_urlsafe(16)
|
||||
token_data = {
|
||||
"access_token": access_token,
|
||||
"scopes": session["scopes"],
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(hours=1)).timestamp(),
|
||||
"session_id": session_id,
|
||||
"clerk_token": token_response.get("access_token"),
|
||||
}
|
||||
self.storage.set_token(token_id, token_data)
|
||||
|
||||
# Clean up session
|
||||
self.storage.delete_session(session_id)
|
||||
|
||||
return {
|
||||
"access_token": access_token,
|
||||
"token_type": "bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": " ".join(session["scopes"]),
|
||||
}
|
||||
|
||||
def validate_pkce(self, code_verifier: str, code_challenge: str) -> bool:
|
||||
"""Validate PKCE code challenge (RFC 7636)"""
|
||||
# S256 method
|
||||
verifier_hash = hashlib.sha256(code_verifier.encode()).digest()
|
||||
expected_challenge = base64.urlsafe_b64encode(verifier_hash).decode().rstrip('=')
|
||||
return expected_challenge == code_challenge
|
||||
|
||||
def _create_mcp_token(
|
||||
self, scopes: list[str], upstream_token: str, session_id: str
|
||||
) -> str:
|
||||
"""Create MCP-scoped JWT token with Clerk token embedded"""
|
||||
|
||||
now = int(time.time())
|
||||
payload = {
|
||||
"iss": self.config.issuer,
|
||||
"sub": session_id,
|
||||
"aud": "mcp-server",
|
||||
"iat": now,
|
||||
"exp": now + 3600, # 1 hour expiration
|
||||
"mcp_tool_scopes": scopes,
|
||||
"upstream_token": upstream_token,
|
||||
"clerk_integration": True,
|
||||
}
|
||||
|
||||
return jwt.encode(payload, self.jwt_secret, algorithm="HS256")
|
||||
|
||||
def introspect_token(self, token: str) -> dict[str, Any]:
|
||||
"""Introspect and validate MCP token"""
|
||||
|
||||
try:
|
||||
payload = jwt.decode(token, self.jwt_secret, algorithms=["HS256"])
|
||||
|
||||
# Check if token is expired
|
||||
if payload.get("exp", 0) < time.time():
|
||||
return {"active": False, "error": "token_expired"}
|
||||
|
||||
return {
|
||||
"active": True,
|
||||
"sub": payload.get("sub"),
|
||||
"aud": payload.get("aud"),
|
||||
"iss": payload.get("iss"),
|
||||
"exp": payload.get("exp"),
|
||||
"iat": payload.get("iat"),
|
||||
"mcp_tool_scopes": payload.get("mcp_tool_scopes", []),
|
||||
"upstream_token": payload.get("upstream_token"),
|
||||
"clerk_integration": payload.get("clerk_integration", False),
|
||||
}
|
||||
|
||||
except PyJWTError as e:
|
||||
logger.warning(f"Token validation failed: {e}")
|
||||
return {"active": False, "error": "invalid_token"}
|
||||
|
||||
def revoke_token(self, token: str) -> bool:
|
||||
"""Revoke a token"""
|
||||
|
||||
try:
|
||||
payload = jwt.decode(token, self.jwt_secret, algorithms=["HS256"])
|
||||
session_id = payload.get("sub")
|
||||
|
||||
# Remove all tokens associated with this session
|
||||
all_tokens = self.storage.get_tokens()
|
||||
tokens_to_remove = [
|
||||
token_id
|
||||
for token_id, token_data in all_tokens.items()
|
||||
if token_data.get("session_id") == session_id
|
||||
]
|
||||
|
||||
for token_id in tokens_to_remove:
|
||||
self.storage.delete_token(token_id)
|
||||
|
||||
logger.info(f"Revoked {len(tokens_to_remove)} tokens for session {session_id}")
|
||||
return True
|
||||
|
||||
except InvalidTokenError as e:
|
||||
logger.warning(f"Token revocation failed: {e}")
|
||||
return False
|
||||
|
||||
def cleanup_expired_sessions(self):
|
||||
"""Clean up expired sessions and tokens"""
|
||||
# This is now handled automatically by persistent storage
|
||||
self.storage.cleanup_expired_sessions()
|
||||
logger.debug("Cleanup completed via persistent storage")
|
||||
@@ -1,201 +0,0 @@
|
||||
"""
|
||||
Authorization policy engine for MCP tools
|
||||
"""
|
||||
|
||||
import re
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PolicyAction(Enum):
|
||||
ALLOW = "allow"
|
||||
DENY = "deny"
|
||||
|
||||
|
||||
@dataclass
|
||||
class ToolPolicy:
|
||||
"""Policy rule for MCP tool access"""
|
||||
|
||||
tool_pattern: str # regex pattern for tool names
|
||||
required_scopes: list[str]
|
||||
action: PolicyAction = PolicyAction.ALLOW
|
||||
conditions: dict[str, Any] | None = None
|
||||
|
||||
def matches_tool(self, tool_name: str) -> bool:
|
||||
"""Check if the policy applies to given tool"""
|
||||
return bool(re.match(self.tool_pattern, tool_name))
|
||||
|
||||
def evaluate_scopes(self, user_scopes: list[str]) -> bool:
|
||||
"""Check if user has required scopes"""
|
||||
return all(scope in user_scopes for scope in self.required_scopes)
|
||||
|
||||
|
||||
class PolicyEngine:
|
||||
"""Authorization policy engine for Turkish legal database tools"""
|
||||
|
||||
def __init__(self):
|
||||
self.policies: list[ToolPolicy] = []
|
||||
self.default_action = PolicyAction.DENY
|
||||
|
||||
def add_policy(self, policy: ToolPolicy):
|
||||
"""Add a policy rule"""
|
||||
self.policies.append(policy)
|
||||
logger.debug(f"Added policy: {policy.tool_pattern} -> {policy.required_scopes}")
|
||||
|
||||
def add_tool_scope_policy(
|
||||
self,
|
||||
tool_pattern: str,
|
||||
required_scopes: str | list[str],
|
||||
action: PolicyAction = PolicyAction.ALLOW,
|
||||
):
|
||||
"""Convenience method to add tool-scope policy"""
|
||||
if isinstance(required_scopes, str):
|
||||
required_scopes = [required_scopes]
|
||||
|
||||
policy = ToolPolicy(
|
||||
tool_pattern=tool_pattern, required_scopes=required_scopes, action=action
|
||||
)
|
||||
self.add_policy(policy)
|
||||
|
||||
def authorize_tool_call(
|
||||
self,
|
||||
tool_name: str,
|
||||
user_scopes: list[str],
|
||||
user_claims: dict[str, Any] | None = None,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""
|
||||
Authorize a tool call
|
||||
|
||||
Returns:
|
||||
(authorized: bool, reason: Optional[str])
|
||||
"""
|
||||
|
||||
logger.debug(f"Authorizing tool '{tool_name}' for user with scopes: {user_scopes}")
|
||||
|
||||
matching_policies = [
|
||||
policy for policy in self.policies if policy.matches_tool(tool_name)
|
||||
]
|
||||
|
||||
if not matching_policies:
|
||||
if self.default_action == PolicyAction.ALLOW:
|
||||
logger.debug(f"No policies found for '{tool_name}', allowing by default")
|
||||
return True, None
|
||||
else:
|
||||
logger.warning(f"No policies found for '{tool_name}', denying by default")
|
||||
return False, f"No policy found for tool '{tool_name}', default deny"
|
||||
|
||||
# Check for explicit deny policies first
|
||||
for policy in matching_policies:
|
||||
if policy.action == PolicyAction.DENY:
|
||||
if policy.evaluate_scopes(user_scopes):
|
||||
logger.warning(f"Explicit deny policy matched for '{tool_name}'")
|
||||
return False, f"Explicit deny policy for tool '{tool_name}'"
|
||||
|
||||
# Check allow policies
|
||||
allow_policies = [
|
||||
p for p in matching_policies if p.action == PolicyAction.ALLOW
|
||||
]
|
||||
|
||||
if not allow_policies:
|
||||
logger.warning(f"No allow policies found for '{tool_name}'")
|
||||
return False, f"No allow policies found for tool '{tool_name}'"
|
||||
|
||||
for policy in allow_policies:
|
||||
if policy.evaluate_scopes(user_scopes):
|
||||
if self._evaluate_conditions(policy.conditions, user_claims):
|
||||
logger.debug(f"Authorization granted for '{tool_name}'")
|
||||
return True, None
|
||||
|
||||
logger.warning(f"Insufficient scopes for '{tool_name}'. Required: {[p.required_scopes for p in allow_policies]}, User has: {user_scopes}")
|
||||
return False, f"Insufficient scopes for tool '{tool_name}'"
|
||||
|
||||
def _evaluate_conditions(
|
||||
self,
|
||||
conditions: dict[str, Any] | None,
|
||||
user_claims: dict[str, Any] | None,
|
||||
) -> bool:
|
||||
"""Evaluate additional policy conditions"""
|
||||
|
||||
if not conditions:
|
||||
return True
|
||||
|
||||
if not user_claims:
|
||||
logger.debug("No user claims provided, conditions evaluation failed")
|
||||
return False
|
||||
|
||||
for key, expected_value in conditions.items():
|
||||
user_value = user_claims.get(key)
|
||||
|
||||
if isinstance(expected_value, list):
|
||||
if user_value not in expected_value:
|
||||
logger.debug(f"Condition failed: {key} = {user_value} not in {expected_value}")
|
||||
return False
|
||||
elif user_value != expected_value:
|
||||
logger.debug(f"Condition failed: {key} = {user_value} != {expected_value}")
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
def get_allowed_tools(self, user_scopes: list[str]) -> list[str]:
|
||||
"""Get list of tool patterns user is allowed to call"""
|
||||
|
||||
allowed_tools = []
|
||||
|
||||
for policy in self.policies:
|
||||
if policy.action == PolicyAction.ALLOW and policy.evaluate_scopes(
|
||||
user_scopes
|
||||
):
|
||||
allowed_tools.append(policy.tool_pattern)
|
||||
|
||||
return allowed_tools
|
||||
|
||||
|
||||
def create_turkish_legal_policies() -> PolicyEngine:
|
||||
"""Create policy set for Turkish legal database MCP server"""
|
||||
|
||||
engine = PolicyEngine()
|
||||
|
||||
# Administrative tools (full access)
|
||||
engine.add_tool_scope_policy(".*", ["mcp:tools:admin"])
|
||||
|
||||
# Search tools - require read access
|
||||
engine.add_tool_scope_policy("search.*", ["mcp:tools:read"])
|
||||
|
||||
# Fetch/get document tools - require read access
|
||||
engine.add_tool_scope_policy("get_.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("fetch.*", ["mcp:tools:read"])
|
||||
|
||||
# Specific Turkish legal database tools
|
||||
engine.add_tool_scope_policy("search_yargitay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_danistay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_anayasa.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_rekabet.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_kik.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_emsal.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_uyusmazlik.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_sayistay.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_.*_bedesten", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_yerel_hukuk.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_istinaf_hukuk.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("search_kyb.*", ["mcp:tools:read"])
|
||||
|
||||
# Document retrieval tools
|
||||
engine.add_tool_scope_policy("get_.*_document.*", ["mcp:tools:read"])
|
||||
engine.add_tool_scope_policy("get_.*_markdown", ["mcp:tools:read"])
|
||||
|
||||
# Write operations (if any future tools need them)
|
||||
engine.add_tool_scope_policy("create_.*", ["mcp:tools:write"])
|
||||
engine.add_tool_scope_policy("update_.*", ["mcp:tools:write"])
|
||||
engine.add_tool_scope_policy("delete_.*", ["mcp:tools:write"])
|
||||
|
||||
logger.info("Created Turkish legal database policy engine")
|
||||
return engine
|
||||
|
||||
|
||||
def create_default_policies() -> PolicyEngine:
|
||||
"""Create a default policy set for MCP servers (backwards compatibility)"""
|
||||
return create_turkish_legal_policies()
|
||||
@@ -1,112 +0,0 @@
|
||||
"""
|
||||
Persistent storage for OAuth sessions and tokens
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
import logging
|
||||
from datetime import datetime
|
||||
from typing import Dict, Any, Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class PersistentStorage:
|
||||
"""File-based persistent storage for OAuth data"""
|
||||
|
||||
def __init__(self, storage_dir: str = None):
|
||||
if storage_dir is None:
|
||||
# Use system temp directory or environment variable
|
||||
storage_dir = os.environ.get('TEMP', tempfile.gettempdir())
|
||||
|
||||
self.storage_dir = os.path.join(storage_dir, 'mcp_oauth_storage')
|
||||
os.makedirs(self.storage_dir, exist_ok=True)
|
||||
|
||||
self.sessions_file = os.path.join(self.storage_dir, 'oauth_sessions.json')
|
||||
self.tokens_file = os.path.join(self.storage_dir, 'oauth_tokens.json')
|
||||
|
||||
logger.info(f"Persistent OAuth storage initialized at: {self.storage_dir}")
|
||||
|
||||
def _load_json(self, filepath: str) -> Dict:
|
||||
"""Load JSON data from file"""
|
||||
try:
|
||||
if os.path.exists(filepath):
|
||||
with open(filepath, 'r', encoding='utf-8') as f:
|
||||
return json.load(f)
|
||||
except Exception as e:
|
||||
logger.error(f"Error loading {filepath}: {e}")
|
||||
return {}
|
||||
|
||||
def _save_json(self, filepath: str, data: Dict):
|
||||
"""Save JSON data to file"""
|
||||
try:
|
||||
with open(filepath, 'w', encoding='utf-8') as f:
|
||||
json.dump(data, f, indent=2, default=str)
|
||||
except Exception as e:
|
||||
logger.error(f"Error saving {filepath}: {e}")
|
||||
|
||||
def get_sessions(self) -> Dict[str, Dict[str, Any]]:
|
||||
"""Get all OAuth sessions"""
|
||||
data = self._load_json(self.sessions_file)
|
||||
# Clean expired sessions
|
||||
now = datetime.utcnow().timestamp()
|
||||
valid_sessions = {k: v for k, v in data.items()
|
||||
if v.get('expires_at', 0) > now}
|
||||
if len(valid_sessions) != len(data):
|
||||
self._save_json(self.sessions_file, valid_sessions)
|
||||
return valid_sessions
|
||||
|
||||
def set_session(self, session_id: str, data: Dict[str, Any]):
|
||||
"""Set OAuth session data"""
|
||||
sessions = self.get_sessions()
|
||||
sessions[session_id] = data
|
||||
self._save_json(self.sessions_file, sessions)
|
||||
|
||||
def get_session(self, session_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Get specific OAuth session data"""
|
||||
sessions = self.get_sessions()
|
||||
return sessions.get(session_id)
|
||||
|
||||
def delete_session(self, session_id: str):
|
||||
"""Delete OAuth session"""
|
||||
sessions = self.get_sessions()
|
||||
if session_id in sessions:
|
||||
del sessions[session_id]
|
||||
self._save_json(self.sessions_file, sessions)
|
||||
|
||||
def get_tokens(self) -> Dict[str, Dict[str, Any]]:
|
||||
"""Get all OAuth tokens"""
|
||||
data = self._load_json(self.tokens_file)
|
||||
# Clean expired tokens
|
||||
now = datetime.utcnow().timestamp()
|
||||
valid_tokens = {k: v for k, v in data.items()
|
||||
if v.get('expires_at', 0) > now}
|
||||
if len(valid_tokens) != len(data):
|
||||
self._save_json(self.tokens_file, valid_tokens)
|
||||
return valid_tokens
|
||||
|
||||
def set_token(self, token_id: str, token_data: Dict[str, Any]):
|
||||
"""Set OAuth token data"""
|
||||
tokens = self.get_tokens()
|
||||
tokens[token_id] = token_data
|
||||
self._save_json(self.tokens_file, tokens)
|
||||
|
||||
def get_token(self, token_id: str) -> Optional[Dict[str, Any]]:
|
||||
"""Get specific OAuth token data"""
|
||||
tokens = self.get_tokens()
|
||||
return tokens.get(token_id)
|
||||
|
||||
def delete_token(self, token_id: str):
|
||||
"""Delete OAuth token"""
|
||||
tokens = self.get_tokens()
|
||||
if token_id in tokens:
|
||||
del tokens[token_id]
|
||||
self._save_json(self.tokens_file, tokens)
|
||||
|
||||
def cleanup_expired_sessions(self):
|
||||
"""Clean up expired sessions and tokens"""
|
||||
# This is handled automatically in get_sessions() and get_tokens()
|
||||
sessions = self.get_sessions()
|
||||
tokens = self.get_tokens()
|
||||
logger.debug(f"Cleanup: {len(sessions)} active sessions, {len(tokens)} active tokens")
|
||||
@@ -1,193 +0,0 @@
|
||||
"""
|
||||
Factory for creating FastMCP app with MCP Auth Toolkit integration
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Optional
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
try:
|
||||
from fastmcp import FastMCP
|
||||
FASTMCP_AVAILABLE = True
|
||||
except ImportError:
|
||||
FASTMCP_AVAILABLE = False
|
||||
FastMCP = None
|
||||
|
||||
from mcp_auth import (
|
||||
OAuthProvider,
|
||||
PolicyEngine,
|
||||
FastMCPAuthWrapper,
|
||||
create_default_policies
|
||||
)
|
||||
from mcp_auth.clerk_config import create_mcp_server_config
|
||||
|
||||
|
||||
def create_auth_enabled_app(app_name: str = "Yargı MCP Server") -> FastMCP:
|
||||
"""Create FastMCP app with authentication enabled"""
|
||||
|
||||
if not FASTMCP_AVAILABLE:
|
||||
raise ImportError("FastMCP is required for authenticated MCP server")
|
||||
|
||||
logger.info("Creating FastMCP app with MCP Auth Toolkit integration")
|
||||
|
||||
# Create base FastMCP app
|
||||
app = FastMCP(app_name)
|
||||
|
||||
# Check if authentication is enabled
|
||||
auth_enabled = os.getenv("ENABLE_AUTH", "true").lower() == "true"
|
||||
|
||||
if not auth_enabled:
|
||||
logger.info("Authentication disabled, returning basic FastMCP app")
|
||||
return app
|
||||
|
||||
try:
|
||||
# Get configuration
|
||||
logger.info("Getting MCP server configuration...")
|
||||
config = create_mcp_server_config()
|
||||
logger.info("Configuration loaded successfully")
|
||||
|
||||
# Create OAuth provider with Clerk config
|
||||
logger.info("Creating OAuth provider...")
|
||||
oauth_provider = OAuthProvider(
|
||||
config=config["oauth_config"],
|
||||
jwt_secret=config["jwt_secret"]
|
||||
)
|
||||
logger.info("OAuth provider created successfully")
|
||||
|
||||
# Create policy engine for Turkish legal database
|
||||
policy_engine = create_default_policies()
|
||||
|
||||
# Store auth components for later wrapping (after tools are defined)
|
||||
app._oauth_provider = oauth_provider
|
||||
app._policy_engine = policy_engine
|
||||
app._auth_config = config
|
||||
|
||||
# Add OAuth endpoints immediately
|
||||
@app.tool(
|
||||
description="Initiate OAuth 2.1 authorization flow with PKCE",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_authorize(redirect_uri: str, scopes: str = None):
|
||||
"""OAuth authorization endpoint"""
|
||||
scope_list = scopes.split(" ") if scopes else ["mcp:tools:read", "mcp:tools:write"]
|
||||
auth_url, pkce = oauth_provider.generate_authorization_url(
|
||||
redirect_uri=redirect_uri, scopes=scope_list
|
||||
)
|
||||
logger.info(f"Generated authorization URL for redirect_uri: {redirect_uri}")
|
||||
return {
|
||||
"authorization_url": auth_url,
|
||||
"code_verifier": pkce.verifier,
|
||||
"code_challenge": pkce.challenge,
|
||||
"instructions": "Use the authorization_url to complete OAuth flow, then exchange the returned code using oauth_token tool"
|
||||
}
|
||||
|
||||
@app.tool(
|
||||
description="Exchange OAuth authorization code for access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_token(code: str, state: str, redirect_uri: str):
|
||||
"""OAuth token exchange endpoint"""
|
||||
try:
|
||||
result = await oauth_provider.exchange_code_for_token(
|
||||
code=code, state=state, redirect_uri=redirect_uri
|
||||
)
|
||||
logger.info("Successfully exchanged authorization code for token")
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.error(f"Token exchange failed: {e}")
|
||||
raise
|
||||
|
||||
@app.tool(
|
||||
description="Validate and introspect OAuth access token",
|
||||
annotations={"readOnlyHint": True, "idempotentHint": True}
|
||||
)
|
||||
async def oauth_introspect(token: str):
|
||||
"""Token introspection endpoint"""
|
||||
result = oauth_provider.introspect_token(token)
|
||||
logger.debug(f"Token introspection: active={result.get('active', False)}")
|
||||
return result
|
||||
|
||||
@app.tool(
|
||||
description="Revoke OAuth access token",
|
||||
annotations={"readOnlyHint": False, "idempotentHint": False}
|
||||
)
|
||||
async def oauth_revoke(token: str):
|
||||
"""Token revocation endpoint"""
|
||||
success = oauth_provider.revoke_token(token)
|
||||
logger.info(f"Token revocation: success={success}")
|
||||
return {"revoked": success}
|
||||
|
||||
logger.info("Successfully created authenticated FastMCP app")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to create authenticated app: {e}")
|
||||
logger.info("Falling back to non-authenticated FastMCP app")
|
||||
# Return basic app if auth setup fails
|
||||
return app
|
||||
|
||||
return app
|
||||
|
||||
|
||||
def create_app() -> FastMCP:
|
||||
"""Create FastMCP app (backwards compatible with mcp_factory.py)"""
|
||||
return create_auth_enabled_app()
|
||||
|
||||
|
||||
def get_auth_wrapper(app: FastMCP) -> Optional[FastMCPAuthWrapper]:
|
||||
"""Get auth wrapper from app if available"""
|
||||
return getattr(app, '_auth_wrapper', None)
|
||||
|
||||
|
||||
def get_oauth_provider(app: FastMCP) -> Optional[OAuthProvider]:
|
||||
"""Get OAuth provider from app if available"""
|
||||
return getattr(app, '_oauth_provider', None)
|
||||
|
||||
|
||||
def get_policy_engine(app: FastMCP) -> Optional[PolicyEngine]:
|
||||
"""Get policy engine from app if available"""
|
||||
return getattr(app, '_policy_engine', None)
|
||||
|
||||
|
||||
def is_auth_enabled(app: FastMCP) -> bool:
|
||||
"""Check if authentication is enabled for the app"""
|
||||
return hasattr(app, '_oauth_provider') or hasattr(app, '_auth_wrapper')
|
||||
|
||||
|
||||
def enable_tool_authentication(app: FastMCP):
|
||||
"""Enable authentication on all existing tools (call after tools are defined)"""
|
||||
if not is_auth_enabled(app):
|
||||
logger.debug("Authentication not enabled, skipping tool authentication")
|
||||
return
|
||||
|
||||
oauth_provider = get_oauth_provider(app)
|
||||
policy_engine = get_policy_engine(app)
|
||||
|
||||
if not oauth_provider or not policy_engine:
|
||||
logger.warning("OAuth provider or policy engine not available")
|
||||
return
|
||||
|
||||
try:
|
||||
# Create auth wrapper and wrap tools
|
||||
auth_wrapper = FastMCPAuthWrapper(
|
||||
mcp_server=app,
|
||||
oauth_provider=oauth_provider,
|
||||
policy_engine=policy_engine
|
||||
)
|
||||
|
||||
# Store wrapper for reference
|
||||
app._auth_wrapper = auth_wrapper
|
||||
|
||||
logger.info("Tool authentication enabled successfully")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to enable tool authentication: {e}")
|
||||
|
||||
|
||||
def cleanup_auth_sessions(app: FastMCP):
|
||||
"""Clean up expired auth sessions and tokens"""
|
||||
oauth_provider = get_oauth_provider(app)
|
||||
if oauth_provider:
|
||||
oauth_provider.cleanup_expired_sessions()
|
||||
logger.debug("Cleaned up expired OAuth sessions")
|
||||
@@ -1,383 +0,0 @@
|
||||
"""
|
||||
HTTP adapter for MCP Auth Toolkit OAuth endpoints
|
||||
Exposes MCP OAuth tools as HTTP endpoints for Claude.ai integration
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
import secrets
|
||||
import time
|
||||
from typing import Optional
|
||||
from urllib.parse import urlencode, quote
|
||||
from datetime import datetime, timedelta
|
||||
|
||||
from fastapi import APIRouter, Request, Query, HTTPException
|
||||
from fastapi.responses import RedirectResponse, JSONResponse
|
||||
|
||||
# Try to import Clerk SDK
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError as e:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
# OAuth configuration
|
||||
BASE_URL = os.getenv("BASE_URL", "https://yargimcp.com")
|
||||
|
||||
|
||||
@router.get("/.well-known/oauth-authorization-server")
|
||||
async def get_oauth_metadata():
|
||||
"""OAuth 2.0 Authorization Server Metadata (RFC 8414)"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": f"{BASE_URL}/authorize",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code", "refresh_token"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"token_endpoint_auth_methods_supported": ["none"],
|
||||
"scopes_supported": ["mcp:tools:read", "mcp:tools:write", "openid", "profile", "email"],
|
||||
"service_documentation": f"{BASE_URL}/mcp/"
|
||||
})
|
||||
|
||||
|
||||
@router.get("/.well-known/oauth-protected-resource")
|
||||
async def get_protected_resource_metadata():
|
||||
"""OAuth Protected Resource Metadata (RFC 9728)"""
|
||||
return JSONResponse({
|
||||
"resource": BASE_URL,
|
||||
"authorization_servers": [BASE_URL],
|
||||
"bearer_methods_supported": ["header"],
|
||||
"scopes_supported": ["mcp:tools:read", "mcp:tools:write"],
|
||||
"resource_documentation": f"{BASE_URL}/docs"
|
||||
})
|
||||
|
||||
|
||||
@router.get("/authorize")
|
||||
async def authorize_endpoint(
|
||||
response_type: str = Query(...),
|
||||
client_id: str = Query(...),
|
||||
redirect_uri: str = Query(...),
|
||||
code_challenge: str = Query(...),
|
||||
code_challenge_method: str = Query("S256"),
|
||||
state: Optional[str] = Query(None),
|
||||
scope: Optional[str] = Query(None)
|
||||
):
|
||||
"""OAuth 2.1 Authorization Endpoint - Uses Clerk SDK for custom domains"""
|
||||
|
||||
logger.info(f"OAuth authorize request - client_id: {client_id}, redirect_uri: {redirect_uri}")
|
||||
|
||||
if not CLERK_AVAILABLE:
|
||||
logger.error("Clerk SDK not available")
|
||||
raise HTTPException(status_code=500, detail="Clerk SDK not available")
|
||||
|
||||
# Store OAuth session for later validation
|
||||
try:
|
||||
from mcp_server_main import app as mcp_app
|
||||
from mcp_auth_factory import get_oauth_provider
|
||||
|
||||
oauth_provider = get_oauth_provider(mcp_app)
|
||||
if not oauth_provider:
|
||||
raise HTTPException(status_code=500, detail="OAuth provider not configured")
|
||||
|
||||
# Generate session and store PKCE
|
||||
session_id = secrets.token_urlsafe(32)
|
||||
if state is None:
|
||||
state = secrets.token_urlsafe(16)
|
||||
|
||||
# Create PKCE challenge
|
||||
from mcp_auth.oauth import PKCEChallenge
|
||||
pkce = PKCEChallenge()
|
||||
|
||||
# Store session data
|
||||
session_data = {
|
||||
"pkce_verifier": pkce.verifier,
|
||||
"pkce_challenge": code_challenge, # Store the client's challenge
|
||||
"state": state,
|
||||
"redirect_uri": redirect_uri,
|
||||
"client_id": client_id,
|
||||
"scopes": scope.split(" ") if scope else ["mcp:tools:read", "mcp:tools:write"],
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=10)).timestamp(),
|
||||
}
|
||||
oauth_provider.storage.set_session(session_id, session_data)
|
||||
|
||||
# For Clerk with custom domains, we need to use their hosted sign-in page
|
||||
# We'll pass our callback URL and session info in the state
|
||||
callback_url = f"{BASE_URL}/auth/callback"
|
||||
|
||||
# Encode session info in state for retrieval after Clerk auth
|
||||
combined_state = f"{state}:{session_id}"
|
||||
|
||||
# Use Clerk's sign-in URL with proper parameters
|
||||
clerk_domain = os.getenv("CLERK_DOMAIN", "accounts.yargimcp.com")
|
||||
sign_in_params = {
|
||||
"redirect_url": f"{callback_url}?state={quote(combined_state)}",
|
||||
}
|
||||
|
||||
sign_in_url = f"https://{clerk_domain}/sign-in?{urlencode(sign_in_params)}"
|
||||
|
||||
logger.info(f"Redirecting to Clerk sign-in: {sign_in_url}")
|
||||
|
||||
return RedirectResponse(url=sign_in_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Authorization failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
|
||||
@router.get("/auth/callback")
|
||||
async def oauth_callback(
|
||||
request: Request,
|
||||
state: Optional[str] = Query(None),
|
||||
clerk_token: Optional[str] = Query(None)
|
||||
):
|
||||
"""Handle OAuth callback from Clerk - supports both JWT token and cookie auth"""
|
||||
|
||||
logger.info(f"OAuth callback received - state: {state}")
|
||||
logger.info(f"Query params: {dict(request.query_params)}")
|
||||
logger.info(f"Cookies: {dict(request.cookies)}")
|
||||
logger.info(f"Clerk JWT token provided: {bool(clerk_token)}")
|
||||
|
||||
# Support both JWT token (for cross-domain) and cookie auth (for subdomain)
|
||||
|
||||
try:
|
||||
if not state:
|
||||
logger.error("No state parameter provided")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Missing state parameter"}
|
||||
)
|
||||
|
||||
# Parse state to get original state and session ID
|
||||
try:
|
||||
if ":" in state:
|
||||
original_state, session_id = state.rsplit(":", 1)
|
||||
else:
|
||||
original_state = state
|
||||
session_id = state # Fallback
|
||||
except ValueError:
|
||||
logger.error(f"Invalid state format: {state}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Invalid state format"}
|
||||
)
|
||||
|
||||
# Get OAuth provider
|
||||
from mcp_server_main import app as mcp_app
|
||||
from mcp_auth_factory import get_oauth_provider
|
||||
|
||||
oauth_provider = get_oauth_provider(mcp_app)
|
||||
if not oauth_provider:
|
||||
raise HTTPException(status_code=500, detail="OAuth provider not configured")
|
||||
|
||||
# Get stored session
|
||||
oauth_session = oauth_provider.storage.get_session(session_id)
|
||||
|
||||
if not oauth_session:
|
||||
logger.error(f"OAuth session not found for ID: {session_id}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "OAuth session expired or not found"}
|
||||
)
|
||||
|
||||
# Check if we have a JWT token (for cross-domain auth)
|
||||
user_authenticated = False
|
||||
auth_method = "none"
|
||||
|
||||
if clerk_token:
|
||||
logger.info("Attempting JWT token validation")
|
||||
try:
|
||||
# Validate JWT token with Clerk
|
||||
from clerk_backend_api import Clerk
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
|
||||
# Extract session_id from JWT token and verify with Clerk
|
||||
import jwt
|
||||
decoded_token = jwt.decode(clerk_token, options={"verify_signature": False})
|
||||
session_id = decoded_token.get("sid") or decoded_token.get("session_id")
|
||||
|
||||
if session_id:
|
||||
# Verify with Clerk using session_id
|
||||
session = clerk.sessions.verify(session_id=session_id, token=clerk_token)
|
||||
user_id = session.user_id if session else None
|
||||
else:
|
||||
user_id = None
|
||||
|
||||
if user_id:
|
||||
logger.info(f"JWT token validation successful - user_id: {user_id}")
|
||||
user_authenticated = True
|
||||
auth_method = "jwt_token"
|
||||
# Store user info in session for token exchange
|
||||
oauth_session["user_id"] = user_id
|
||||
oauth_session["auth_method"] = "jwt_token"
|
||||
else:
|
||||
logger.error("JWT token validation failed - no user_id in claims")
|
||||
except Exception as e:
|
||||
logger.error(f"JWT token validation failed: {str(e)}")
|
||||
# Fall through to cookie validation
|
||||
|
||||
# If no JWT token or validation failed, check cookies
|
||||
if not user_authenticated:
|
||||
logger.info("Checking for Clerk session cookies")
|
||||
# Check for Clerk session cookies (for subdomain auth)
|
||||
clerk_session_cookie = request.cookies.get("__session")
|
||||
if clerk_session_cookie:
|
||||
logger.info("Found Clerk session cookie, assuming authenticated")
|
||||
user_authenticated = True
|
||||
auth_method = "cookie"
|
||||
oauth_session["auth_method"] = "cookie"
|
||||
else:
|
||||
logger.info("No Clerk session cookie found")
|
||||
|
||||
# For custom domains, we'll also trust that Clerk redirected here
|
||||
if not user_authenticated:
|
||||
logger.info("Trusting Clerk redirect for custom domain flow")
|
||||
user_authenticated = True
|
||||
auth_method = "trusted_redirect"
|
||||
oauth_session["auth_method"] = "trusted_redirect"
|
||||
|
||||
logger.info(f"User authenticated: {user_authenticated}, method: {auth_method}")
|
||||
|
||||
# Generate simple authorization code for custom domain flow
|
||||
auth_code = f"clerk_custom_{session_id}_{int(time.time())}"
|
||||
|
||||
# Store the code mapping for token exchange
|
||||
code_data = {
|
||||
"session_id": session_id,
|
||||
"clerk_authenticated": user_authenticated,
|
||||
"auth_method": auth_method,
|
||||
"custom_domain_flow": True,
|
||||
"created_at": time.time(),
|
||||
"expires_at": (datetime.utcnow() + timedelta(minutes=5)).timestamp(),
|
||||
}
|
||||
if "user_id" in oauth_session:
|
||||
code_data["user_id"] = oauth_session["user_id"]
|
||||
|
||||
oauth_provider.storage.set_session(f"code_{auth_code}", code_data)
|
||||
|
||||
# Build redirect URL back to Claude
|
||||
redirect_params = {
|
||||
"code": auth_code,
|
||||
"state": original_state
|
||||
}
|
||||
|
||||
redirect_url = f"{oauth_session['redirect_uri']}?{urlencode(redirect_params)}"
|
||||
logger.info(f"Redirecting back to Claude: {redirect_url}")
|
||||
|
||||
return RedirectResponse(url=redirect_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Callback processing failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
|
||||
|
||||
@router.post("/register")
|
||||
async def register_client(request: Request):
|
||||
"""Dynamic Client Registration (RFC 7591)"""
|
||||
|
||||
data = await request.json()
|
||||
logger.info(f"Client registration request: {data}")
|
||||
|
||||
# Simple dynamic registration - accept any client
|
||||
client_id = f"mcp-client-{os.urandom(8).hex()}"
|
||||
|
||||
return JSONResponse({
|
||||
"client_id": client_id,
|
||||
"client_secret": None, # Public client
|
||||
"redirect_uris": data.get("redirect_uris", []),
|
||||
"grant_types": ["authorization_code", "refresh_token"],
|
||||
"response_types": ["code"],
|
||||
"client_name": data.get("client_name", "MCP Client"),
|
||||
"token_endpoint_auth_method": "none",
|
||||
"client_id_issued_at": int(datetime.now().timestamp())
|
||||
})
|
||||
|
||||
|
||||
@router.post("/token")
|
||||
async def token_endpoint(request: Request):
|
||||
"""OAuth 2.1 Token Endpoint"""
|
||||
|
||||
# Parse form data
|
||||
form_data = await request.form()
|
||||
grant_type = form_data.get("grant_type")
|
||||
code = form_data.get("code")
|
||||
redirect_uri = form_data.get("redirect_uri")
|
||||
client_id = form_data.get("client_id")
|
||||
code_verifier = form_data.get("code_verifier")
|
||||
|
||||
logger.info(f"Token exchange - grant_type: {grant_type}, code: {code[:20] if code else 'None'}...")
|
||||
|
||||
if grant_type != "authorization_code":
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "unsupported_grant_type"}
|
||||
)
|
||||
|
||||
try:
|
||||
# OAuth token exchange - validate code and return Clerk JWT
|
||||
# This supports proper OAuth flow while using Clerk JWT tokens
|
||||
|
||||
if not code or not redirect_uri:
|
||||
logger.error("Missing required parameters: code or redirect_uri")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Missing code or redirect_uri"}
|
||||
)
|
||||
|
||||
# Validate OAuth code with Clerk
|
||||
if CLERK_AVAILABLE:
|
||||
try:
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
|
||||
# In a real implementation, you'd validate the code with Clerk
|
||||
# For now, we'll assume the code is valid if it looks like a Clerk code
|
||||
if len(code) > 10: # Basic validation
|
||||
# Create a mock session with the code
|
||||
# In practice, this would be validated with Clerk's OAuth flow
|
||||
|
||||
# Return Clerk JWT token format
|
||||
# This should be the actual Clerk JWT token from the OAuth flow
|
||||
return JSONResponse({
|
||||
"access_token": f"mock_clerk_jwt_{code}",
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "yargi.read yargi.search"
|
||||
})
|
||||
else:
|
||||
logger.error(f"Invalid code format: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid authorization code"}
|
||||
)
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Clerk validation failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Authorization code validation failed"}
|
||||
)
|
||||
else:
|
||||
logger.warning("Clerk SDK not available, using mock response")
|
||||
return JSONResponse({
|
||||
"access_token": "mock_jwt_token_for_development",
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "yargi.read yargi.search"
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Token exchange failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
@@ -1,522 +0,0 @@
|
||||
"""
|
||||
Simplified MCP OAuth HTTP adapter - only Clerk JWT based authentication
|
||||
Uses Redis for authorization code storage to support multi-machine deployment
|
||||
"""
|
||||
|
||||
import os
|
||||
import logging
|
||||
from typing import Optional
|
||||
from urllib.parse import urlencode, quote
|
||||
|
||||
from fastapi import APIRouter, Request, Query, HTTPException
|
||||
from fastapi.responses import RedirectResponse, JSONResponse
|
||||
|
||||
# Import Redis session store
|
||||
from redis_session_store import get_redis_store
|
||||
|
||||
# Try to import Clerk SDK
|
||||
try:
|
||||
from clerk_backend_api import Clerk
|
||||
CLERK_AVAILABLE = True
|
||||
except ImportError:
|
||||
CLERK_AVAILABLE = False
|
||||
Clerk = None
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
router = APIRouter()
|
||||
|
||||
# OAuth configuration
|
||||
BASE_URL = os.getenv("BASE_URL", "https://api.yargimcp.com")
|
||||
CLERK_DOMAIN = os.getenv("CLERK_DOMAIN", "accounts.yargimcp.com")
|
||||
|
||||
# Initialize Redis store
|
||||
redis_store = None
|
||||
|
||||
def get_redis_session_store():
|
||||
"""Get Redis store instance with lazy initialization."""
|
||||
global redis_store
|
||||
if redis_store is None:
|
||||
try:
|
||||
import concurrent.futures
|
||||
import functools
|
||||
|
||||
# Use thread pool with timeout to prevent hanging
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as executor:
|
||||
future = executor.submit(get_redis_store)
|
||||
try:
|
||||
# 5 second timeout for Redis initialization
|
||||
redis_store = future.result(timeout=5.0)
|
||||
if redis_store:
|
||||
logger.info("Redis session store initialized for OAuth handler")
|
||||
else:
|
||||
logger.warning("Redis store initialization returned None")
|
||||
except concurrent.futures.TimeoutError:
|
||||
logger.error("Redis initialization timed out after 5 seconds")
|
||||
redis_store = None
|
||||
future.cancel() # Try to cancel the hanging operation
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to initialize Redis store: {e}")
|
||||
redis_store = None
|
||||
|
||||
if redis_store is None:
|
||||
# Fall back to in-memory storage with warning
|
||||
logger.warning("Falling back to in-memory storage - multi-machine deployment will not work")
|
||||
|
||||
return redis_store
|
||||
|
||||
@router.get("/.well-known/oauth-authorization-server")
|
||||
async def get_oauth_metadata():
|
||||
"""OAuth 2.0 Authorization Server Metadata (RFC 8414)"""
|
||||
return JSONResponse({
|
||||
"issuer": BASE_URL,
|
||||
"authorization_endpoint": "https://yargimcp.com/mcp-callback",
|
||||
"token_endpoint": f"{BASE_URL}/token",
|
||||
"registration_endpoint": f"{BASE_URL}/register",
|
||||
"response_types_supported": ["code"],
|
||||
"grant_types_supported": ["authorization_code"],
|
||||
"code_challenge_methods_supported": ["S256"],
|
||||
"token_endpoint_auth_methods_supported": ["none"],
|
||||
"scopes_supported": ["read", "search", "openid", "profile", "email"],
|
||||
"service_documentation": f"{BASE_URL}/mcp/"
|
||||
})
|
||||
|
||||
@router.get("/auth/login")
|
||||
async def oauth_authorize(
|
||||
request: Request,
|
||||
client_id: str = Query(...),
|
||||
redirect_uri: str = Query(...),
|
||||
response_type: str = Query("code"),
|
||||
scope: Optional[str] = Query("read search"),
|
||||
state: Optional[str] = Query(None),
|
||||
code_challenge: Optional[str] = Query(None),
|
||||
code_challenge_method: Optional[str] = Query(None)
|
||||
):
|
||||
"""OAuth 2.1 Authorization Endpoint - redirects to Clerk"""
|
||||
|
||||
logger.info(f"OAuth authorize request - client_id: {client_id}")
|
||||
logger.info(f"Redirect URI: {redirect_uri}")
|
||||
logger.info(f"State: {state}")
|
||||
logger.info(f"PKCE Challenge: {bool(code_challenge)}")
|
||||
|
||||
try:
|
||||
# Build callback URL with all necessary parameters
|
||||
callback_url = f"{BASE_URL}/auth/callback"
|
||||
callback_params = {
|
||||
"client_id": client_id,
|
||||
"redirect_uri": redirect_uri,
|
||||
"state": state or "",
|
||||
"scope": scope or "read search"
|
||||
}
|
||||
|
||||
# Add PKCE parameters if present
|
||||
if code_challenge:
|
||||
callback_params["code_challenge"] = code_challenge
|
||||
callback_params["code_challenge_method"] = code_challenge_method or "S256"
|
||||
|
||||
# Encode callback URL as redirect_url for Clerk
|
||||
callback_with_params = f"{callback_url}?{urlencode(callback_params)}"
|
||||
|
||||
# Build Clerk sign-in URL - use yargimcp.com frontend for JWT token generation
|
||||
clerk_params = {
|
||||
"redirect_url": callback_with_params
|
||||
}
|
||||
|
||||
# Use frontend sign-in page that handles JWT token generation
|
||||
clerk_signin_url = f"https://yargimcp.com/sign-in?{urlencode(clerk_params)}"
|
||||
|
||||
logger.info(f"Redirecting to Clerk: {clerk_signin_url}")
|
||||
|
||||
return RedirectResponse(url=clerk_signin_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Authorization failed: {e}")
|
||||
raise HTTPException(status_code=500, detail=str(e))
|
||||
|
||||
@router.get("/auth/callback")
|
||||
async def oauth_callback(
|
||||
request: Request,
|
||||
client_id: str = Query(...),
|
||||
redirect_uri: str = Query(...),
|
||||
state: Optional[str] = Query(None),
|
||||
scope: Optional[str] = Query("read search"),
|
||||
code_challenge: Optional[str] = Query(None),
|
||||
code_challenge_method: Optional[str] = Query(None),
|
||||
clerk_token: Optional[str] = Query(None)
|
||||
):
|
||||
"""OAuth callback from Clerk - generates authorization code"""
|
||||
|
||||
logger.info(f"OAuth callback - client_id: {client_id}")
|
||||
logger.info(f"Clerk token provided: {bool(clerk_token)}")
|
||||
|
||||
try:
|
||||
# Validate user with Clerk and generate real JWT token
|
||||
user_authenticated = False
|
||||
user_id = None
|
||||
session_id = None
|
||||
real_jwt_token = None
|
||||
|
||||
if clerk_token and CLERK_AVAILABLE:
|
||||
try:
|
||||
# Extract user info from JWT token (no Clerk session verification needed)
|
||||
import jwt
|
||||
decoded_token = jwt.decode(clerk_token, options={"verify_signature": False})
|
||||
user_id = decoded_token.get("user_id") or decoded_token.get("sub")
|
||||
user_email = decoded_token.get("email")
|
||||
token_scopes = decoded_token.get("scopes", ["read", "search"])
|
||||
|
||||
logger.info(f"JWT token claims - user_id: {user_id}, email: {user_email}, scopes: {token_scopes}")
|
||||
|
||||
if user_id and user_email:
|
||||
# JWT token is already signed by Clerk and contains valid user info
|
||||
user_authenticated = True
|
||||
logger.info(f"User authenticated via JWT token - user_id: {user_id}")
|
||||
|
||||
# Use the JWT token directly as the real token (it's already from Clerk template)
|
||||
real_jwt_token = clerk_token
|
||||
logger.info("Using Clerk JWT token directly (already real token)")
|
||||
|
||||
else:
|
||||
logger.error(f"Missing required fields in JWT token - user_id: {bool(user_id)}, email: {bool(user_email)}")
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"JWT validation failed: {e}")
|
||||
|
||||
# Fallback to cookie validation
|
||||
if not user_authenticated:
|
||||
clerk_session = request.cookies.get("__session")
|
||||
if clerk_session:
|
||||
user_authenticated = True
|
||||
logger.info("User authenticated via cookie")
|
||||
|
||||
# Try to get session from cookie and generate JWT
|
||||
if CLERK_AVAILABLE:
|
||||
try:
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
# Note: sessions.verify_session is deprecated, but we'll try
|
||||
# In practice, you'd need to extract session_id from cookie
|
||||
logger.info("Cookie authentication - JWT generation not implemented yet")
|
||||
except Exception as e:
|
||||
logger.warning(f"Failed to generate JWT from cookie: {e}")
|
||||
|
||||
# Only generate authorization code if we have a real JWT token
|
||||
if user_authenticated and real_jwt_token:
|
||||
# Generate authorization code
|
||||
auth_code = f"clerk_auth_{os.urandom(16).hex()}"
|
||||
|
||||
# Prepare code data
|
||||
import time
|
||||
code_data = {
|
||||
"user_id": user_id,
|
||||
"session_id": session_id,
|
||||
"real_jwt_token": real_jwt_token,
|
||||
"user_authenticated": user_authenticated,
|
||||
"client_id": client_id,
|
||||
"redirect_uri": redirect_uri,
|
||||
"scope": scope or "read search"
|
||||
}
|
||||
|
||||
# Try to store in Redis, fall back to in-memory if Redis unavailable
|
||||
store = get_redis_session_store()
|
||||
if store:
|
||||
# Store in Redis with automatic expiration
|
||||
success = store.set_oauth_code(auth_code, code_data)
|
||||
if success:
|
||||
logger.info(f"Stored authorization code {auth_code[:10]}... in Redis with real JWT token")
|
||||
else:
|
||||
logger.error(f"Failed to store authorization code in Redis, falling back to in-memory")
|
||||
# Fall back to in-memory storage
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
oauth_callback._code_storage[auth_code] = code_data
|
||||
else:
|
||||
# Fall back to in-memory storage
|
||||
logger.warning("Redis not available, using in-memory storage")
|
||||
if not hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage = {}
|
||||
oauth_callback._code_storage[auth_code] = code_data
|
||||
logger.info(f"Stored authorization code in memory (fallback)")
|
||||
|
||||
# Redirect back to client with authorization code
|
||||
redirect_params = {
|
||||
"code": auth_code,
|
||||
"state": state or ""
|
||||
}
|
||||
|
||||
final_redirect_url = f"{redirect_uri}?{urlencode(redirect_params)}"
|
||||
logger.info(f"Redirecting back to client: {final_redirect_url}")
|
||||
|
||||
return RedirectResponse(url=final_redirect_url)
|
||||
else:
|
||||
# No JWT token yet - redirect back to sign-in page to wait for authentication
|
||||
logger.info("No JWT token provided - redirecting back to sign-in to complete authentication")
|
||||
|
||||
# Keep the same redirect URL so the flow continues
|
||||
sign_in_params = {
|
||||
"redirect_url": f"{request.url._url}" # Current callback URL with all params
|
||||
}
|
||||
|
||||
sign_in_url = f"https://yargimcp.com/sign-in?{urlencode(sign_in_params)}"
|
||||
logger.info(f"Redirecting back to sign-in: {sign_in_url}")
|
||||
|
||||
return RedirectResponse(url=sign_in_url)
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Callback processing failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
|
||||
@router.post("/auth/register")
|
||||
async def register_client(request: Request):
|
||||
"""Dynamic Client Registration (RFC 7591)"""
|
||||
|
||||
data = await request.json()
|
||||
logger.info(f"Client registration request: {data}")
|
||||
|
||||
# Simple dynamic registration - accept any client
|
||||
client_id = f"mcp-client-{os.urandom(8).hex()}"
|
||||
|
||||
return JSONResponse({
|
||||
"client_id": client_id,
|
||||
"client_secret": None, # Public client
|
||||
"redirect_uris": data.get("redirect_uris", []),
|
||||
"grant_types": ["authorization_code"],
|
||||
"response_types": ["code"],
|
||||
"client_name": data.get("client_name", "MCP Client"),
|
||||
"token_endpoint_auth_method": "none"
|
||||
})
|
||||
|
||||
@router.post("/auth/callback")
|
||||
async def oauth_callback_post(request: Request):
|
||||
"""OAuth callback POST endpoint for token exchange"""
|
||||
|
||||
# Parse form data (standard OAuth token exchange format)
|
||||
form_data = await request.form()
|
||||
grant_type = form_data.get("grant_type")
|
||||
code = form_data.get("code")
|
||||
redirect_uri = form_data.get("redirect_uri")
|
||||
client_id = form_data.get("client_id")
|
||||
code_verifier = form_data.get("code_verifier")
|
||||
|
||||
logger.info(f"OAuth callback POST - grant_type: {grant_type}")
|
||||
logger.info(f"Code: {code[:20] if code else 'None'}...")
|
||||
logger.info(f"Client ID: {client_id}")
|
||||
logger.info(f"PKCE verifier: {bool(code_verifier)}")
|
||||
|
||||
if grant_type != "authorization_code":
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "unsupported_grant_type"}
|
||||
)
|
||||
|
||||
if not code or not redirect_uri:
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Missing code or redirect_uri"}
|
||||
)
|
||||
|
||||
try:
|
||||
# Validate authorization code
|
||||
if not code.startswith("clerk_auth_"):
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid authorization code"}
|
||||
)
|
||||
|
||||
# Retrieve stored JWT token using authorization code from Redis or in-memory fallback
|
||||
stored_code_data = None
|
||||
|
||||
# Try to get from Redis first, then fall back to in-memory
|
||||
store = get_redis_session_store()
|
||||
if store:
|
||||
stored_code_data = store.get_oauth_code(code, delete_after_use=True)
|
||||
if stored_code_data:
|
||||
logger.info(f"Retrieved authorization code {code[:10]}... from Redis")
|
||||
else:
|
||||
logger.warning(f"Authorization code {code[:10]}... not found in Redis")
|
||||
|
||||
# Fall back to in-memory storage if Redis unavailable or code not found
|
||||
if not stored_code_data and hasattr(oauth_callback, '_code_storage'):
|
||||
stored_code_data = oauth_callback._code_storage.get(code)
|
||||
if stored_code_data:
|
||||
# Clean up in-memory storage
|
||||
oauth_callback._code_storage.pop(code, None)
|
||||
logger.info(f"Retrieved authorization code {code[:10]}... from in-memory storage")
|
||||
|
||||
if not stored_code_data:
|
||||
logger.error(f"No stored data found for authorization code: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Authorization code not found or expired"}
|
||||
)
|
||||
|
||||
# Note: Redis TTL handles expiration automatically, but check for manual expiration for in-memory fallback
|
||||
import time
|
||||
expires_at = stored_code_data.get("expires_at", 0)
|
||||
if expires_at and time.time() > expires_at:
|
||||
logger.error(f"Authorization code expired: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Authorization code expired"}
|
||||
)
|
||||
|
||||
# Get the real JWT token
|
||||
real_jwt_token = stored_code_data.get("real_jwt_token")
|
||||
|
||||
if real_jwt_token:
|
||||
logger.info("Returning real Clerk JWT token")
|
||||
# Note: Code already deleted from Redis, clean up in-memory fallback if used
|
||||
if hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage.pop(code, None)
|
||||
|
||||
return JSONResponse({
|
||||
"access_token": real_jwt_token,
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "read search"
|
||||
})
|
||||
else:
|
||||
logger.warning("No real JWT token found, generating mock token")
|
||||
# Fallback to mock token for testing
|
||||
mock_token = f"mock_clerk_jwt_{code}"
|
||||
return JSONResponse({
|
||||
"access_token": mock_token,
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "read search"
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"OAuth callback POST failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
|
||||
@router.post("/register")
|
||||
async def register_client(request: Request):
|
||||
"""Dynamic Client Registration (RFC 7591)"""
|
||||
|
||||
data = await request.json()
|
||||
logger.info(f"Client registration request: {data}")
|
||||
|
||||
# Simple dynamic registration - accept any client
|
||||
client_id = f"mcp-client-{os.urandom(8).hex()}"
|
||||
|
||||
return JSONResponse({
|
||||
"client_id": client_id,
|
||||
"client_secret": None, # Public client
|
||||
"redirect_uris": data.get("redirect_uris", []),
|
||||
"grant_types": ["authorization_code"],
|
||||
"response_types": ["code"],
|
||||
"client_name": data.get("client_name", "MCP Client"),
|
||||
"token_endpoint_auth_method": "none"
|
||||
})
|
||||
|
||||
@router.post("/token")
|
||||
async def token_endpoint(request: Request):
|
||||
"""OAuth 2.1 Token Endpoint - exchanges code for Clerk JWT"""
|
||||
|
||||
# Parse form data
|
||||
form_data = await request.form()
|
||||
grant_type = form_data.get("grant_type")
|
||||
code = form_data.get("code")
|
||||
redirect_uri = form_data.get("redirect_uri")
|
||||
client_id = form_data.get("client_id")
|
||||
code_verifier = form_data.get("code_verifier")
|
||||
|
||||
logger.info(f"Token exchange - grant_type: {grant_type}")
|
||||
logger.info(f"Code: {code[:20] if code else 'None'}...")
|
||||
|
||||
if grant_type != "authorization_code":
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "unsupported_grant_type"}
|
||||
)
|
||||
|
||||
if not code or not redirect_uri:
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_request", "error_description": "Missing code or redirect_uri"}
|
||||
)
|
||||
|
||||
try:
|
||||
# Validate authorization code
|
||||
if not code.startswith("clerk_auth_"):
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Invalid authorization code"}
|
||||
)
|
||||
|
||||
# Retrieve stored JWT token using authorization code from Redis or in-memory fallback
|
||||
stored_code_data = None
|
||||
|
||||
# Try to get from Redis first, then fall back to in-memory
|
||||
store = get_redis_session_store()
|
||||
if store:
|
||||
stored_code_data = store.get_oauth_code(code, delete_after_use=True)
|
||||
if stored_code_data:
|
||||
logger.info(f"Retrieved authorization code {code[:10]}... from Redis (/token endpoint)")
|
||||
else:
|
||||
logger.warning(f"Authorization code {code[:10]}... not found in Redis (/token endpoint)")
|
||||
|
||||
# Fall back to in-memory storage if Redis unavailable or code not found
|
||||
if not stored_code_data and hasattr(oauth_callback, '_code_storage'):
|
||||
stored_code_data = oauth_callback._code_storage.get(code)
|
||||
if stored_code_data:
|
||||
# Clean up in-memory storage
|
||||
oauth_callback._code_storage.pop(code, None)
|
||||
logger.info(f"Retrieved authorization code {code[:10]}... from in-memory storage (/token endpoint)")
|
||||
|
||||
if not stored_code_data:
|
||||
logger.error(f"No stored data found for authorization code: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Authorization code not found or expired"}
|
||||
)
|
||||
|
||||
# Note: Redis TTL handles expiration automatically, but check for manual expiration for in-memory fallback
|
||||
import time
|
||||
expires_at = stored_code_data.get("expires_at", 0)
|
||||
if expires_at and time.time() > expires_at:
|
||||
logger.error(f"Authorization code expired: {code}")
|
||||
return JSONResponse(
|
||||
status_code=400,
|
||||
content={"error": "invalid_grant", "error_description": "Authorization code expired"}
|
||||
)
|
||||
|
||||
# Get the real JWT token
|
||||
real_jwt_token = stored_code_data.get("real_jwt_token")
|
||||
|
||||
if real_jwt_token:
|
||||
logger.info("Returning real Clerk JWT token from /token endpoint")
|
||||
# Note: Code already deleted from Redis, clean up in-memory fallback if used
|
||||
if hasattr(oauth_callback, '_code_storage'):
|
||||
oauth_callback._code_storage.pop(code, None)
|
||||
|
||||
return JSONResponse({
|
||||
"access_token": real_jwt_token,
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "read search"
|
||||
})
|
||||
else:
|
||||
logger.warning("No real JWT token found in /token endpoint, generating mock token")
|
||||
# Fallback to mock token for testing
|
||||
mock_token = f"mock_clerk_jwt_{code}"
|
||||
return JSONResponse({
|
||||
"access_token": mock_token,
|
||||
"token_type": "Bearer",
|
||||
"expires_in": 3600,
|
||||
"scope": "read search"
|
||||
})
|
||||
|
||||
except Exception as e:
|
||||
logger.exception(f"Token exchange failed: {e}")
|
||||
return JSONResponse(
|
||||
status_code=500,
|
||||
content={"error": "server_error", "error_description": str(e)}
|
||||
)
|
||||
+984
-1144
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,57 @@
|
||||
"""
|
||||
Migration stub for the deprecated Yargı MCP endpoint.
|
||||
|
||||
Exposes a single tool that informs the MCP client the server has moved
|
||||
and the user must update their configuration.
|
||||
|
||||
Entrypoint variable `app` is a FastMCP instance so it works with
|
||||
Dokploy's FastMCP build pipeline (`fastmcp inspect`, `fastmcp run`).
|
||||
|
||||
Run with:
|
||||
fastmcp run migration_app.py:app --transport http --port 8000
|
||||
"""
|
||||
|
||||
from starlette.responses import JSONResponse
|
||||
from fastmcp import FastMCP
|
||||
|
||||
NEW_URL = "https://yargimcp.surucu.dev/mcp"
|
||||
|
||||
app = FastMCP(
|
||||
name="Yargı MCP (taşındı / moved)",
|
||||
instructions=(
|
||||
f"Bu Yargı MCP endpoint'i kullanımdan kaldırıldı. "
|
||||
f"Sunucu yeni adrese taşındı: {NEW_URL}. "
|
||||
f"Kullanıcıya MCP istemci ayarlarındaki sunucu URL'sini güncellemesini söyle."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@app.tool(
|
||||
description=(
|
||||
"DEPRECATED ENDPOINT — Yargı MCP sunucusu yeni adrese taşındı. "
|
||||
"Bu endpoint'teki eski araçlar (Yargıtay, Danıştay, Anayasa Mahkemesi, "
|
||||
"KVKK, BDDK, Rekabet Kurumu, Sayıştay, KİK, Emsal, Uyuşmazlık, GİB, "
|
||||
"Sigorta Tahkim, Bedesten vb.) artık burada çalışmıyor. "
|
||||
"Bu aracı çağırarak yeni sunucu adresini ve migrasyon talimatını al, "
|
||||
"ardından kullanıcıya MCP istemci (Claude Desktop, 5ire, ChatGPT vb.) "
|
||||
"ayarlarındaki sunucu URL'sini güncellemesi gerektiğini açıkça söyle."
|
||||
)
|
||||
)
|
||||
def migration_notice() -> dict:
|
||||
"""Yargı MCP sunucusunun yeni adresini ve migrasyon talimatını döner."""
|
||||
return {
|
||||
"status": "moved",
|
||||
"new_url": NEW_URL,
|
||||
"message": (
|
||||
f"Yargı MCP sunucusu yeni adrese taşındı: {NEW_URL}\n\n"
|
||||
f"Lütfen MCP istemcinin (Claude Desktop, 5ire, ChatGPT vb.) "
|
||||
f"ayarlarındaki sunucu URL'sini yukarıdaki yeni adresle güncelleyin. "
|
||||
f"Mevcut endpoint artık kullanım dışıdır ve sadece bu uyarıyı döner."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@app.custom_route("/health", methods=["GET"])
|
||||
async def health(request):
|
||||
"""Health check endpoint for monitoring services."""
|
||||
return JSONResponse({"status": "deprecated", "new_url": NEW_URL})
|
||||
-94
@@ -1,94 +0,0 @@
|
||||
events {
|
||||
worker_connections 1024;
|
||||
}
|
||||
|
||||
http {
|
||||
upstream yargi_mcp {
|
||||
server yargi-mcp:8000;
|
||||
}
|
||||
|
||||
# Rate limiting
|
||||
limit_req_zone $binary_remote_addr zone=api_limit:10m rate=10r/s;
|
||||
limit_req_zone $binary_remote_addr zone=mcp_limit:10m rate=100r/s;
|
||||
|
||||
server {
|
||||
listen 80;
|
||||
server_name localhost;
|
||||
|
||||
# Redirect HTTP to HTTPS in production
|
||||
# return 301 https://$server_name$request_uri;
|
||||
|
||||
# Security headers
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header X-Frame-Options DENY;
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header Referrer-Policy "strict-origin-when-cross-origin";
|
||||
|
||||
# API endpoints
|
||||
location /api/ {
|
||||
limit_req zone=api_limit burst=20 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# Timeouts
|
||||
proxy_connect_timeout 60s;
|
||||
proxy_send_timeout 60s;
|
||||
proxy_read_timeout 60s;
|
||||
}
|
||||
|
||||
# MCP endpoint (higher rate limit)
|
||||
location /mcp-server/mcp/ {
|
||||
limit_req zone=mcp_limit burst=50 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
|
||||
# WebSocket support
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
|
||||
# Longer timeouts for MCP operations
|
||||
proxy_connect_timeout 300s;
|
||||
proxy_send_timeout 300s;
|
||||
proxy_read_timeout 300s;
|
||||
}
|
||||
|
||||
# Health check (no rate limit)
|
||||
location /health {
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
}
|
||||
|
||||
# Root and other paths
|
||||
location / {
|
||||
limit_req zone=api_limit burst=10 nodelay;
|
||||
|
||||
proxy_pass http://yargi_mcp;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
}
|
||||
}
|
||||
|
||||
# SSL configuration (uncomment for production)
|
||||
# server {
|
||||
# listen 443 ssl http2;
|
||||
# server_name your-domain.com;
|
||||
#
|
||||
# ssl_certificate /etc/nginx/ssl/cert.pem;
|
||||
# ssl_certificate_key /etc/nginx/ssl/key.pem;
|
||||
# ssl_protocols TLSv1.2 TLSv1.3;
|
||||
# ssl_ciphers HIGH:!aNULL:!MD5;
|
||||
#
|
||||
# # Include all location blocks from above
|
||||
# }
|
||||
}
|
||||
+6
-11
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "yargi-mcp"
|
||||
version = "0.1.6"
|
||||
version = "0.2.1"
|
||||
description = "MCP Server For Turkish Legal Databases"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
@@ -25,12 +25,12 @@ dependencies = [
|
||||
"markitdown[pdf]>=0.1.1",
|
||||
"pydantic>=2.11.4",
|
||||
"aiohttp>=3.11.18",
|
||||
"playwright>=1.52.0",
|
||||
"fastmcp>=2.10.5",
|
||||
"pypdf>=5.5.0",
|
||||
"fastapi>=0.115.14",
|
||||
"PyJWT>=2.8.0",
|
||||
"tiktoken>=0.5.0",
|
||||
"cryptography>=44.0.0",
|
||||
"openai>=1.0.0",
|
||||
"numpy>=1.24.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
@@ -46,20 +46,15 @@ production = [
|
||||
"gunicorn>=22.0.0",
|
||||
"uvicorn[standard]>=0.30.0",
|
||||
]
|
||||
saas = [
|
||||
"clerk-backend-api>=3.0.0",
|
||||
"stripe>=9.1.0",
|
||||
"upstash-redis>=1.1.0",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
yargi-mcp = "mcp_server_main:main"
|
||||
|
||||
[tool.setuptools]
|
||||
py-modules = ["mcp_server_main", "mcp_auth_factory", "mcp_auth_http_adapter", "asgi_app", "fastapi_app", "starlette_app", "run_asgi", "stripe_webhook"]
|
||||
py-modules = ["mcp_server_main", "asgi_app"]
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["*_mcp_module", "mcp_auth"]
|
||||
include = ["*_mcp_module", "semantic_search"]
|
||||
|
||||
[build-system]
|
||||
requires = ["setuptools>=65.0", "wheel"]
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
# rekabet_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import List, Optional, Tuple, Dict, Any
|
||||
@@ -141,12 +142,12 @@ class RekabetKurumuApiClient:
|
||||
|
||||
# Row 1: Publication Date, Decision Number, Related Cases Link
|
||||
td_elements_r1 = rows[0].find_all("td")
|
||||
pub_date = td_elements_r1[0].get_text(strip=True) if len(td_elements_r1) > 0 else None
|
||||
dec_num = td_elements_r1[1].get_text(strip=True) if len(td_elements_r1) > 1 else None
|
||||
pub_date = td_elements_r1[0].get_text(strip=True) if len(td_elements_r1) > 0 else ""
|
||||
dec_num = td_elements_r1[1].get_text(strip=True) if len(td_elements_r1) > 1 else ""
|
||||
|
||||
related_cases_link_tag = td_elements_r1[2].find("a", href=True) if len(td_elements_r1) > 2 else None
|
||||
related_cases_url_str: Optional[str] = None
|
||||
karar_id_from_related: Optional[str] = None
|
||||
related_cases_url_str: str = ""
|
||||
karar_id_from_related: str = ""
|
||||
if related_cases_link_tag and related_cases_link_tag.has_attr('href'):
|
||||
related_cases_url_str = urljoin(self.BASE_URL, related_cases_link_tag['href'])
|
||||
qs_related = parse_qs(urlparse(related_cases_link_tag['href']).query)
|
||||
@@ -155,16 +156,16 @@ class RekabetKurumuApiClient:
|
||||
|
||||
# Row 2: Decision Date, Decision Type
|
||||
td_elements_r2 = rows[1].find_all("td")
|
||||
dec_date = td_elements_r2[0].get_text(strip=True) if len(td_elements_r2) > 0 else None
|
||||
dec_type_text = td_elements_r2[1].get_text(strip=True) if len(td_elements_r2) > 1 else None
|
||||
dec_date = td_elements_r2[0].get_text(strip=True) if len(td_elements_r2) > 0 else ""
|
||||
dec_type_text = td_elements_r2[1].get_text(strip=True) if len(td_elements_r2) > 1 else ""
|
||||
|
||||
# Row 3: Title and Main Decision Link
|
||||
title_cell = rows[2].find("td", colspan="5")
|
||||
decision_link_tag = title_cell.find("a", href=True) if title_cell else None
|
||||
|
||||
title_text: Optional[str] = None
|
||||
decision_landing_url_str: Optional[str] = None
|
||||
karar_id_from_main_link: Optional[str] = None
|
||||
title_text: str = ""
|
||||
decision_landing_url_str: str = ""
|
||||
karar_id_from_main_link: str = ""
|
||||
|
||||
if decision_link_tag and decision_link_tag.has_attr('href'):
|
||||
title_text = decision_link_tag.get_text(strip=True)
|
||||
@@ -185,16 +186,12 @@ class RekabetKurumuApiClient:
|
||||
logger.warning(f"Table {idx+1} Karar ID not found. Skipping. Title (if any): {title_text}")
|
||||
continue
|
||||
|
||||
# Convert string URLs to HttpUrl for the model
|
||||
final_decision_url = HttpUrl(decision_landing_url_str) if decision_landing_url_str else None
|
||||
final_related_cases_url = HttpUrl(related_cases_url_str) if related_cases_url_str else None
|
||||
|
||||
processed_decisions.append(RekabetDecisionSummary(
|
||||
publication_date=pub_date, decision_number=dec_num, decision_date=dec_date,
|
||||
decision_type_text=dec_type_text, title=title_text,
|
||||
decision_url=final_decision_url,
|
||||
decision_url=decision_landing_url_str,
|
||||
karar_id=current_karar_id,
|
||||
related_cases_url=final_related_cases_url
|
||||
related_cases_url=related_cases_url_str
|
||||
))
|
||||
logger.debug(f"Table {idx+1} parsed successfully: Karar ID '{current_karar_id}', Title '{title_text[:50] if title_text else 'N/A'}...'")
|
||||
|
||||
@@ -357,7 +354,7 @@ class RekabetKurumuApiClient:
|
||||
total_pdf_pages = total_pdf_pages_from_extraction
|
||||
|
||||
if single_page_pdf_bytes:
|
||||
markdown_for_requested_page = self._convert_pdf_bytes_to_markdown(single_page_pdf_bytes, str(pdf_url_to_report or full_landing_page_url))
|
||||
markdown_for_requested_page = await asyncio.to_thread(self._convert_pdf_bytes_to_markdown, single_page_pdf_bytes, str(pdf_url_to_report or full_landing_page_url))
|
||||
if not markdown_for_requested_page:
|
||||
error_message = (error_message or "") + f"; Could not convert page {page_number} of PDF to Markdown."
|
||||
elif total_pdf_pages > 0 :
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
fastmcp
|
||||
httpx
|
||||
beautifulsoup4
|
||||
markitdown[pdf]
|
||||
pydantic
|
||||
aiohttp
|
||||
playwright
|
||||
pypdf
|
||||
fastapi>=0.115.14
|
||||
uvicorn[standard]>=0.30.0
|
||||
starlette>=0.37.0
|
||||
-119
@@ -1,119 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Standalone ASGI server runner for Yargı MCP
|
||||
|
||||
This script provides a simple way to run the Yargı MCP server
|
||||
as a web service using uvicorn.
|
||||
|
||||
Usage:
|
||||
python run_asgi.py
|
||||
python run_asgi.py --host 0.0.0.0 --port 8080
|
||||
python run_asgi.py --reload # For development
|
||||
"""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import argparse
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
# Add project root to Python path
|
||||
sys.path.insert(0, str(Path(__file__).parent))
|
||||
|
||||
try:
|
||||
import uvicorn
|
||||
except ImportError:
|
||||
print("Error: uvicorn is not installed.")
|
||||
print("Please install it with: pip install uvicorn")
|
||||
sys.exit(1)
|
||||
|
||||
# Configure logging
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Run Yargı MCP server as an ASGI web service"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--host",
|
||||
type=str,
|
||||
default=os.getenv("HOST", "127.0.0.1"),
|
||||
help="Host to bind to (default: 127.0.0.1)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--port",
|
||||
type=int,
|
||||
default=int(os.getenv("PORT", "8000")),
|
||||
help="Port to bind to (default: 8000)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--reload",
|
||||
action="store_true",
|
||||
help="Enable auto-reload for development"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--transport",
|
||||
choices=["http", "sse"],
|
||||
default="http",
|
||||
help="Transport type (default: http)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--log-level",
|
||||
choices=["debug", "info", "warning", "error"],
|
||||
default=os.getenv("LOG_LEVEL", "info").lower(),
|
||||
help="Log level (default: info)"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--workers",
|
||||
type=int,
|
||||
default=1,
|
||||
help="Number of worker processes (default: 1)"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
# Select app based on transport
|
||||
app_name = "asgi_app:app" if args.transport == "http" else "asgi_app:sse_app"
|
||||
|
||||
# Configure uvicorn
|
||||
config = {
|
||||
"app": app_name,
|
||||
"host": args.host,
|
||||
"port": args.port,
|
||||
"log_level": args.log_level,
|
||||
"reload": args.reload,
|
||||
"access_log": True,
|
||||
}
|
||||
|
||||
# Add workers only if not in reload mode
|
||||
if not args.reload and args.workers > 1:
|
||||
config["workers"] = args.workers
|
||||
|
||||
# Print startup information
|
||||
print(f"Starting Yargı MCP server...")
|
||||
print(f"Host: {args.host}")
|
||||
print(f"Port: {args.port}")
|
||||
print(f"Transport: {args.transport}")
|
||||
print(f"Log level: {args.log_level}")
|
||||
if args.reload:
|
||||
print("Auto-reload: enabled")
|
||||
else:
|
||||
print(f"Workers: {args.workers}")
|
||||
print(f"\nServer will be available at: http://{args.host}:{args.port}")
|
||||
print(f"MCP endpoint: http://{args.host}:{args.port}/mcp/")
|
||||
print(f"Health check: http://{args.host}:{args.port}/health")
|
||||
print(f"API status: http://{args.host}:{args.port}/status")
|
||||
print("\nPress CTRL+C to stop the server\n")
|
||||
|
||||
# Run uvicorn
|
||||
try:
|
||||
uvicorn.run(**config)
|
||||
except KeyboardInterrupt:
|
||||
print("\nShutting down server...")
|
||||
sys.exit(0)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,5 +1,6 @@
|
||||
# sayistay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
import re
|
||||
from bs4 import BeautifulSoup
|
||||
@@ -48,6 +49,12 @@ class SayistayApiClient:
|
||||
TEMYIZ_KURULU_ENDPOINT = "/KararlarTemyiz/DataTablesList"
|
||||
DAIRE_ENDPOINT = "/KararlarDaire/DataTablesList"
|
||||
|
||||
# Marker present in the upstream WAF block page (also returns HTTP 418).
|
||||
# Verified 2026-05-03 against real Chrome — the block targets POSTs to
|
||||
# the DataTablesList endpoints regardless of headers/cookies/CSRF, so
|
||||
# we surface a specific error instead of the generic "I'm a teapot".
|
||||
_WAF_BLOCK_MARKER = "Bilgi Güvenliği Politikaları Gereği Kısıtlanmıştır"
|
||||
|
||||
# Page endpoints for session initialization and document access
|
||||
GENEL_KURUL_PAGE = "/KararlarGenelKurul"
|
||||
TEMYIZ_KURULU_PAGE = "/KararlarTemyiz"
|
||||
@@ -141,6 +148,23 @@ class SayistayApiClient:
|
||||
|
||||
return enum_value
|
||||
|
||||
def _raise_if_waf_blocked(self, response: httpx.Response, endpoint_label: str) -> None:
|
||||
"""
|
||||
Sayıştay's upstream WAF returns HTTP 418 with a Turkish HTML block
|
||||
page for POSTs to the DataTablesList endpoints. This affects every
|
||||
client (verified with real Chrome on 2026-05-03), so there is no
|
||||
client-side workaround. Detect it and raise a clear error.
|
||||
"""
|
||||
if response.status_code == 418 or self._WAF_BLOCK_MARKER in response.text:
|
||||
raise RuntimeError(
|
||||
f"Sayıştay upstream WAF blocked the {endpoint_label} request "
|
||||
f"(HTTP {response.status_code} from {response.request.url}). "
|
||||
"This is a server-side restriction at sayistay.gov.tr — affects "
|
||||
"all clients including a real browser — and cannot be worked "
|
||||
"around from yargi-mcp. Try again later or contact Sayıştay if "
|
||||
"the block persists."
|
||||
)
|
||||
|
||||
def _build_datatables_params(self, start: int, length: int, draw: int = 1) -> List[Tuple[str, str]]:
|
||||
"""Build standard DataTables parameters for all endpoints."""
|
||||
params = [
|
||||
@@ -384,6 +408,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Genel Kurul")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -443,6 +468,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Temyiz Kurulu")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -502,6 +528,7 @@ class SayistayApiClient:
|
||||
data=encoded_data,
|
||||
headers=headers
|
||||
)
|
||||
self._raise_if_waf_blocked(response, "Daire")
|
||||
response.raise_for_status()
|
||||
response_json = response.json()
|
||||
|
||||
@@ -631,7 +658,7 @@ class SayistayApiClient:
|
||||
)
|
||||
|
||||
# Convert HTML to Markdown using existing method
|
||||
markdown_content = self._convert_html_to_markdown(html_content)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, html_content)
|
||||
|
||||
if markdown_content and "Error converting HTML content" not in markdown_content:
|
||||
logger.info(f"Successfully retrieved and converted document {decision_id} to Markdown")
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
# semantic_search/__init__.py
|
||||
|
||||
from .embedder import (
|
||||
OpenRouterEmbedder,
|
||||
LocalEmbedder,
|
||||
get_embedder,
|
||||
is_openrouter_available,
|
||||
is_local_embedding_configured,
|
||||
is_semantic_search_available,
|
||||
)
|
||||
from .vector_store import VectorStore
|
||||
from .processor import DocumentProcessor
|
||||
|
||||
__all__ = [
|
||||
'OpenRouterEmbedder',
|
||||
'LocalEmbedder',
|
||||
'get_embedder',
|
||||
'is_openrouter_available',
|
||||
'is_local_embedding_configured',
|
||||
'is_semantic_search_available',
|
||||
'VectorStore',
|
||||
'DocumentProcessor',
|
||||
]
|
||||
@@ -0,0 +1,348 @@
|
||||
# semantic_search/embedder.py
|
||||
|
||||
import logging
|
||||
import os
|
||||
from typing import Dict, List, Optional
|
||||
import numpy as np
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# OpenRouter defaults (preserve backward compatibility)
|
||||
DEFAULT_MODEL = "google/gemini-embedding-001"
|
||||
DEFAULT_DIMENSION = 3072
|
||||
|
||||
# Local provider defaults — Ollama with nomic-embed-text out of the box.
|
||||
# Override via LOCAL_EMBEDDING_BASE_URL / LOCAL_EMBEDDING_MODEL /
|
||||
# LOCAL_EMBEDDING_DIMENSION when using a different server or model.
|
||||
# For Turkish, intfloat/multilingual-e5-large (1024 dims, prompt_style=e5)
|
||||
# served via HuggingFace TEI is the recommended setup — see README.
|
||||
LOCAL_DEFAULT_BASE_URL = "http://localhost:11434/v1"
|
||||
LOCAL_DEFAULT_MODEL = "nomic-embed-text"
|
||||
LOCAL_DEFAULT_DIMENSION = 768
|
||||
|
||||
# Prompt-template styles. Embedding models are trained with specific
|
||||
# prefixes — using the wrong style silently degrades retrieval quality.
|
||||
# - "gemini": "task: {task} | query: {text}" / "title: {title} | text: {text}"
|
||||
# (matches google/gemini-embedding-001, the OpenRouter default)
|
||||
# - "e5": "query: {text}" / "passage: {text}"
|
||||
# (matches intfloat/multilingual-e5-* models — best for Turkish)
|
||||
# - "raw": no prefix; pass text through as-is
|
||||
PROMPT_STYLES = ("gemini", "e5", "raw")
|
||||
DEFAULT_PROMPT_STYLE = "gemini"
|
||||
|
||||
|
||||
def _format_query(prompt_style: str, query: str, task: str) -> str:
|
||||
if prompt_style == "e5":
|
||||
return f"query: {query}"
|
||||
if prompt_style == "raw":
|
||||
return query
|
||||
# gemini (default)
|
||||
return f"task: {task} | query: {query}"
|
||||
|
||||
|
||||
def _format_document(prompt_style: str, doc: str, title: str) -> str:
|
||||
if prompt_style == "e5":
|
||||
return f"passage: {doc}"
|
||||
if prompt_style == "raw":
|
||||
return doc
|
||||
# gemini (default)
|
||||
return f"title: {title} | text: {doc}"
|
||||
|
||||
|
||||
def _resolve_prompt_style(explicit: Optional[str], default: str) -> str:
|
||||
style = (explicit or os.getenv("EMBEDDING_PROMPT_STYLE") or default).strip().lower()
|
||||
if style not in PROMPT_STYLES:
|
||||
raise ValueError(
|
||||
f"Unknown EMBEDDING_PROMPT_STYLE {style!r}; expected one of {PROMPT_STYLES}"
|
||||
)
|
||||
return style
|
||||
|
||||
|
||||
def is_openrouter_available() -> bool:
|
||||
"""Check if OpenRouter API key is available."""
|
||||
return bool(os.getenv("OPENROUTER_API_KEY"))
|
||||
|
||||
|
||||
def is_local_embedding_configured() -> bool:
|
||||
"""Check if the user opted into a local embedding endpoint."""
|
||||
return os.getenv("EMBEDDING_PROVIDER", "").strip().lower() == "local"
|
||||
|
||||
|
||||
def is_semantic_search_available() -> bool:
|
||||
"""Returns True if any embedding provider is configured."""
|
||||
return is_local_embedding_configured() or is_openrouter_available()
|
||||
|
||||
|
||||
def _coerce_dimension(value, env_name: str, default: int) -> int:
|
||||
"""Parse a dimension value (int or str) with clear error messages."""
|
||||
if value is None:
|
||||
return default
|
||||
try:
|
||||
parsed = int(value)
|
||||
except (TypeError, ValueError) as e:
|
||||
raise ValueError(
|
||||
f"{env_name} must be an integer, got {value!r}"
|
||||
) from e
|
||||
if parsed <= 0:
|
||||
raise ValueError(f"Embedding dimension must be positive, got {parsed}")
|
||||
return parsed
|
||||
|
||||
|
||||
class _BaseOpenAICompatibleEmbedder:
|
||||
"""
|
||||
Shared encode/similarity logic for embedders backed by the OpenAI Python
|
||||
SDK. Subclasses configure ``client``, ``model``, ``dimension``, and
|
||||
optionally ``_extra_headers`` (e.g. OpenRouter ranking headers).
|
||||
"""
|
||||
|
||||
# Subclasses may override; sent on every embeddings.create call when set.
|
||||
_extra_headers: Dict[str, str] = {}
|
||||
|
||||
# Set by subclasses
|
||||
client = None
|
||||
model: str = ""
|
||||
dimension: int = 0
|
||||
prompt_style: str = DEFAULT_PROMPT_STYLE
|
||||
|
||||
def encode_query(self, query: str, task: str = "search result") -> np.ndarray:
|
||||
"""
|
||||
Encode a search query. Prefix is selected by ``self.prompt_style``.
|
||||
|
||||
Args:
|
||||
query: The search query text
|
||||
task: Task hint used by the gemini-style prefix; ignored for
|
||||
e5/raw styles.
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (``self.dimension`` elements).
|
||||
"""
|
||||
text = _format_query(self.prompt_style, query, task)
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=text,
|
||||
encoding_format="float",
|
||||
extra_headers=self._extra_headers or None,
|
||||
)
|
||||
|
||||
embedding = np.array(response.data[0].embedding, dtype=np.float32)
|
||||
|
||||
# L2 normalize for cosine similarity
|
||||
norm = np.linalg.norm(embedding)
|
||||
if norm > 0:
|
||||
embedding = embedding / norm
|
||||
|
||||
logger.debug(f"Encoded query: {query[:50]}... -> shape: {embedding.shape}")
|
||||
return embedding
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode query: {e}")
|
||||
raise
|
||||
|
||||
def encode_documents(self, documents: List[str], titles: Optional[List[str]] = None) -> np.ndarray:
|
||||
"""
|
||||
Encode multiple documents with a batch API call.
|
||||
|
||||
Args:
|
||||
documents: List of document texts
|
||||
titles: Optional list of document titles
|
||||
|
||||
Returns:
|
||||
Numpy array of embeddings (N x ``self.dimension``).
|
||||
"""
|
||||
if not documents:
|
||||
return np.array([])
|
||||
|
||||
texts = []
|
||||
for i, doc in enumerate(documents):
|
||||
title = titles[i] if titles and i < len(titles) else "none"
|
||||
texts.append(_format_document(self.prompt_style, doc, title))
|
||||
|
||||
try:
|
||||
response = self.client.embeddings.create(
|
||||
model=self.model,
|
||||
input=texts,
|
||||
encoding_format="float",
|
||||
extra_headers=self._extra_headers or None,
|
||||
)
|
||||
|
||||
embeddings = np.array(
|
||||
[d.embedding for d in sorted(response.data, key=lambda x: x.index)],
|
||||
dtype=np.float32,
|
||||
)
|
||||
|
||||
# L2 normalize each embedding for cosine similarity
|
||||
norms = np.linalg.norm(embeddings, axis=1, keepdims=True)
|
||||
embeddings = embeddings / (norms + 1e-8)
|
||||
|
||||
logger.info(f"Encoded {len(documents)} documents -> shape: {embeddings.shape}")
|
||||
return embeddings
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to encode documents: {e}")
|
||||
raise
|
||||
|
||||
def compute_similarity(self, query_embedding: np.ndarray, document_embeddings: np.ndarray) -> np.ndarray:
|
||||
"""
|
||||
Compute cosine similarity between query and documents.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding (``self.dimension``,)
|
||||
document_embeddings: Document embeddings (N x ``self.dimension``)
|
||||
|
||||
Returns:
|
||||
Similarity scores (N,)
|
||||
"""
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Embeddings are already L2-normalized.
|
||||
similarities = np.dot(document_embeddings, query_embedding.T).squeeze()
|
||||
return similarities
|
||||
|
||||
|
||||
class OpenRouterEmbedder(_BaseOpenAICompatibleEmbedder):
|
||||
"""
|
||||
Embedder using OpenRouter's embedding API.
|
||||
|
||||
The model and dimension are configurable so users can pick any OpenRouter
|
||||
embedding model (e.g. when one becomes paid). Configuration precedence:
|
||||
explicit constructor args > environment variables > defaults.
|
||||
|
||||
Environment variables:
|
||||
OPENROUTER_API_KEY (required): OpenRouter credential
|
||||
OPENROUTER_EMBEDDING_MODEL (optional): override the embedding model id
|
||||
OPENROUTER_EMBEDDING_DIMENSION (optional): override the vector size
|
||||
|
||||
Defaults preserve backward compatibility: ``google/gemini-embedding-001``
|
||||
at 3072 dimensions.
|
||||
"""
|
||||
|
||||
_extra_headers = {
|
||||
"HTTP-Referer": "https://yargimcp.com",
|
||||
"X-Title": "Yargi MCP Server",
|
||||
}
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
model: Optional[str] = None,
|
||||
dimension: Optional[int] = None,
|
||||
prompt_style: Optional[str] = None,
|
||||
):
|
||||
api_key = os.getenv("OPENROUTER_API_KEY")
|
||||
if not api_key:
|
||||
raise ValueError("OPENROUTER_API_KEY environment variable is not set")
|
||||
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError:
|
||||
raise ImportError("openai package is required. Install with: pip install openai")
|
||||
|
||||
self.client = OpenAI(
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
api_key=api_key,
|
||||
)
|
||||
self.model = model or os.getenv("OPENROUTER_EMBEDDING_MODEL") or DEFAULT_MODEL
|
||||
self.dimension = _coerce_dimension(
|
||||
dimension if dimension is not None else os.getenv("OPENROUTER_EMBEDDING_DIMENSION"),
|
||||
"OPENROUTER_EMBEDDING_DIMENSION",
|
||||
DEFAULT_DIMENSION,
|
||||
)
|
||||
# Default to gemini-style prefix for OpenRouter — matches the default
|
||||
# google/gemini-embedding-001 model. Override via constructor or
|
||||
# EMBEDDING_PROMPT_STYLE env var when picking a different model.
|
||||
self.prompt_style = _resolve_prompt_style(prompt_style, "gemini")
|
||||
|
||||
logger.info(
|
||||
f"OpenRouter Embedder initialized with model: {self.model} "
|
||||
f"(dimension={self.dimension}, prompt_style={self.prompt_style})"
|
||||
)
|
||||
|
||||
|
||||
class LocalEmbedder(_BaseOpenAICompatibleEmbedder):
|
||||
"""
|
||||
Embedder for a local OpenAI-compatible embedding server — Ollama,
|
||||
llama.cpp, vLLM, LM Studio, etc. Zero new Python dependencies; just
|
||||
point the existing OpenAI SDK at a local base URL.
|
||||
|
||||
Environment variables:
|
||||
EMBEDDING_PROVIDER=local (selects this provider)
|
||||
LOCAL_EMBEDDING_BASE_URL (default: http://localhost:11434/v1)
|
||||
LOCAL_EMBEDDING_MODEL (default: nomic-embed-text)
|
||||
LOCAL_EMBEDDING_DIMENSION (default: 768)
|
||||
LOCAL_EMBEDDING_API_KEY (optional; ignored by most local servers)
|
||||
|
||||
Setup (Ollama):
|
||||
$ ollama serve
|
||||
$ ollama pull nomic-embed-text # or bge-m3 for better Turkish
|
||||
|
||||
The dimension MUST match the model's actual output size (e.g. 768 for
|
||||
nomic-embed-text, 1024 for bge-m3, 1024 for mxbai-embed-large).
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
base_url: Optional[str] = None,
|
||||
model: Optional[str] = None,
|
||||
dimension: Optional[int] = None,
|
||||
api_key: Optional[str] = None,
|
||||
prompt_style: Optional[str] = None,
|
||||
):
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except ImportError:
|
||||
raise ImportError("openai package is required. Install with: pip install openai")
|
||||
|
||||
self.base_url = (
|
||||
base_url
|
||||
or os.getenv("LOCAL_EMBEDDING_BASE_URL")
|
||||
or LOCAL_DEFAULT_BASE_URL
|
||||
)
|
||||
# Most local servers don't validate the key — use a placeholder so
|
||||
# the OpenAI SDK doesn't error on the missing-key check.
|
||||
effective_key = (
|
||||
api_key
|
||||
or os.getenv("LOCAL_EMBEDDING_API_KEY")
|
||||
or "no-key-needed"
|
||||
)
|
||||
|
||||
self.client = OpenAI(base_url=self.base_url, api_key=effective_key)
|
||||
self.model = model or os.getenv("LOCAL_EMBEDDING_MODEL") or LOCAL_DEFAULT_MODEL
|
||||
self.dimension = _coerce_dimension(
|
||||
dimension if dimension is not None else os.getenv("LOCAL_EMBEDDING_DIMENSION"),
|
||||
"LOCAL_EMBEDDING_DIMENSION",
|
||||
LOCAL_DEFAULT_DIMENSION,
|
||||
)
|
||||
# Default to e5 prefix for local — the recommended Turkish setup
|
||||
# (multilingual-e5-large). Override via EMBEDDING_PROMPT_STYLE when
|
||||
# using a different model family (e.g. nomic, bge).
|
||||
self.prompt_style = _resolve_prompt_style(prompt_style, "e5")
|
||||
|
||||
logger.info(
|
||||
f"Local Embedder initialized: model={self.model} "
|
||||
f"base_url={self.base_url} dimension={self.dimension} "
|
||||
f"prompt_style={self.prompt_style}"
|
||||
)
|
||||
|
||||
|
||||
def get_embedder():
|
||||
"""
|
||||
Factory that picks the embedder based on EMBEDDING_PROVIDER.
|
||||
|
||||
- ``EMBEDDING_PROVIDER=local`` -> ``LocalEmbedder``
|
||||
- otherwise -> ``OpenRouterEmbedder`` (requires OPENROUTER_API_KEY)
|
||||
|
||||
Raises:
|
||||
ValueError: If no provider is configured (neither local nor OpenRouter).
|
||||
"""
|
||||
if is_local_embedding_configured():
|
||||
return LocalEmbedder()
|
||||
if is_openrouter_available():
|
||||
return OpenRouterEmbedder()
|
||||
raise ValueError(
|
||||
"No embedding provider configured. Set OPENROUTER_API_KEY for hosted "
|
||||
"embeddings, or EMBEDDING_PROVIDER=local (with LOCAL_EMBEDDING_* "
|
||||
"env vars) for a local OpenAI-compatible server like Ollama."
|
||||
)
|
||||
@@ -0,0 +1,305 @@
|
||||
# semantic_search/processor.py
|
||||
|
||||
import logging
|
||||
import re
|
||||
from typing import List, Dict, Any, Optional
|
||||
from dataclasses import dataclass
|
||||
import hashlib
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class DocumentChunk:
|
||||
"""Represents a chunk of a document."""
|
||||
chunk_id: str
|
||||
document_id: str
|
||||
text: str
|
||||
metadata: Dict[str, Any]
|
||||
chunk_index: int
|
||||
total_chunks: int
|
||||
|
||||
class DocumentProcessor:
|
||||
"""
|
||||
Processes legal documents for semantic search.
|
||||
Handles chunking, cleaning, and metadata extraction.
|
||||
"""
|
||||
|
||||
def __init__(self,
|
||||
chunk_size: int = 1000,
|
||||
chunk_overlap: int = 200,
|
||||
min_chunk_size: int = 100):
|
||||
"""
|
||||
Initialize document processor.
|
||||
|
||||
Args:
|
||||
chunk_size: Target size for each chunk in characters
|
||||
chunk_overlap: Number of overlapping characters between chunks
|
||||
min_chunk_size: Minimum chunk size to keep
|
||||
"""
|
||||
self.chunk_size = chunk_size
|
||||
self.chunk_overlap = chunk_overlap
|
||||
self.min_chunk_size = min_chunk_size
|
||||
|
||||
logger.info(f"Initialized DocumentProcessor (chunk_size={chunk_size}, overlap={chunk_overlap})")
|
||||
|
||||
def process_document(self,
|
||||
document_id: str,
|
||||
text: str,
|
||||
metadata: Optional[Dict[str, Any]] = None) -> List[DocumentChunk]:
|
||||
"""
|
||||
Process a single document into chunks.
|
||||
|
||||
Args:
|
||||
document_id: Unique document identifier
|
||||
text: Document text content
|
||||
metadata: Optional document metadata
|
||||
|
||||
Returns:
|
||||
List of document chunks
|
||||
"""
|
||||
if not text or len(text.strip()) < self.min_chunk_size:
|
||||
logger.warning(f"Document {document_id} too short to process")
|
||||
return []
|
||||
|
||||
# Clean text
|
||||
cleaned_text = self._clean_text(text)
|
||||
|
||||
# Extract metadata from text if not provided
|
||||
if metadata is None:
|
||||
metadata = {}
|
||||
|
||||
# Add extracted metadata
|
||||
extracted_metadata = self._extract_metadata(cleaned_text)
|
||||
metadata.update(extracted_metadata)
|
||||
|
||||
# Create chunks
|
||||
chunks = self._create_chunks(cleaned_text)
|
||||
|
||||
# Create DocumentChunk objects
|
||||
document_chunks = []
|
||||
for i, chunk_text in enumerate(chunks):
|
||||
chunk_id = self._generate_chunk_id(document_id, i)
|
||||
|
||||
chunk = DocumentChunk(
|
||||
chunk_id=chunk_id,
|
||||
document_id=document_id,
|
||||
text=chunk_text,
|
||||
metadata={
|
||||
**metadata,
|
||||
'chunk_index': i,
|
||||
'total_chunks': len(chunks)
|
||||
},
|
||||
chunk_index=i,
|
||||
total_chunks=len(chunks)
|
||||
)
|
||||
document_chunks.append(chunk)
|
||||
|
||||
logger.info(f"Processed document {document_id} into {len(chunks)} chunks")
|
||||
return document_chunks
|
||||
|
||||
def _clean_text(self, text: str) -> str:
|
||||
"""
|
||||
Clean and normalize text for processing.
|
||||
|
||||
Args:
|
||||
text: Raw text
|
||||
|
||||
Returns:
|
||||
Cleaned text
|
||||
"""
|
||||
# Remove excessive whitespace
|
||||
text = re.sub(r'\s+', ' ', text)
|
||||
|
||||
# Remove special characters but keep Turkish characters
|
||||
# Keep: letters, numbers, spaces, and common punctuation
|
||||
text = re.sub(r'[^\w\s\.\,\;\:\!\?\-\(\)\"\'ÇĞIİÖŞÜçğıiöşü]', ' ', text)
|
||||
|
||||
# Remove multiple spaces
|
||||
text = re.sub(r' +', ' ', text)
|
||||
|
||||
# Trim
|
||||
text = text.strip()
|
||||
|
||||
return text
|
||||
|
||||
def _extract_metadata(self, text: str) -> Dict[str, Any]:
|
||||
"""
|
||||
Extract metadata from legal document text.
|
||||
|
||||
Args:
|
||||
text: Document text
|
||||
|
||||
Returns:
|
||||
Extracted metadata
|
||||
"""
|
||||
metadata = {}
|
||||
|
||||
# Extract case numbers (Esas/Karar)
|
||||
esas_pattern = r'E(?:sas)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
karar_pattern = r'K(?:arar)?[\s\.\:]*(\d{4})[\/\-](\d+)'
|
||||
|
||||
esas_match = re.search(esas_pattern, text[:500]) # Look in first 500 chars
|
||||
if esas_match:
|
||||
metadata['esas_no'] = f"E.{esas_match.group(1)}/{esas_match.group(2)}"
|
||||
|
||||
karar_match = re.search(karar_pattern, text[:500])
|
||||
if karar_match:
|
||||
metadata['karar_no'] = f"K.{karar_match.group(1)}/{karar_match.group(2)}"
|
||||
|
||||
# Extract dates (DD.MM.YYYY or DD/MM/YYYY format)
|
||||
date_pattern = r'(\d{1,2})[\.\/](\d{1,2})[\.\/](\d{4})'
|
||||
dates = re.findall(date_pattern, text[:1000]) # Look in first 1000 chars
|
||||
if dates:
|
||||
# Take the first date as decision date
|
||||
day, month, year = dates[0]
|
||||
metadata['karar_tarihi'] = f"{year}-{month.zfill(2)}-{day.zfill(2)}"
|
||||
|
||||
# Extract court/chamber name
|
||||
chamber_patterns = [
|
||||
r'(\d+)\.\s*Hukuk\s+Dairesi',
|
||||
r'(\d+)\.\s*Ceza\s+Dairesi',
|
||||
r'Hukuk\s+Genel\s+Kurulu',
|
||||
r'Ceza\s+Genel\s+Kurulu',
|
||||
r'(\d+)\.\s*Daire'
|
||||
]
|
||||
|
||||
for pattern in chamber_patterns:
|
||||
match = re.search(pattern, text[:500], re.IGNORECASE)
|
||||
if match:
|
||||
metadata['chamber'] = match.group(0)
|
||||
break
|
||||
|
||||
return metadata
|
||||
|
||||
def _create_chunks(self, text: str) -> List[str]:
|
||||
"""
|
||||
Create overlapping chunks from text.
|
||||
|
||||
Args:
|
||||
text: Cleaned document text
|
||||
|
||||
Returns:
|
||||
List of text chunks
|
||||
"""
|
||||
chunks = []
|
||||
|
||||
# Split by sentences for better semantic coherence
|
||||
sentences = self._split_sentences(text)
|
||||
|
||||
current_chunk = []
|
||||
current_size = 0
|
||||
|
||||
for sentence in sentences:
|
||||
sentence_size = len(sentence)
|
||||
|
||||
# If adding this sentence exceeds chunk size
|
||||
if current_size + sentence_size > self.chunk_size and current_chunk:
|
||||
# Save current chunk
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
chunks.append(chunk_text)
|
||||
|
||||
# Create overlap for next chunk
|
||||
overlap_size = 0
|
||||
overlap_sentences = []
|
||||
|
||||
# Add sentences from the end until we reach overlap size
|
||||
for sent in reversed(current_chunk):
|
||||
overlap_size += len(sent)
|
||||
overlap_sentences.insert(0, sent)
|
||||
if overlap_size >= self.chunk_overlap:
|
||||
break
|
||||
|
||||
# Start new chunk with overlap
|
||||
current_chunk = overlap_sentences
|
||||
current_size = sum(len(s) for s in current_chunk)
|
||||
|
||||
# Add sentence to current chunk
|
||||
current_chunk.append(sentence)
|
||||
current_size += sentence_size
|
||||
|
||||
# Add final chunk if not empty
|
||||
if current_chunk:
|
||||
chunk_text = ' '.join(current_chunk)
|
||||
if len(chunk_text) >= self.min_chunk_size:
|
||||
chunks.append(chunk_text)
|
||||
|
||||
return chunks
|
||||
|
||||
def _split_sentences(self, text: str) -> List[str]:
|
||||
"""
|
||||
Split text into sentences.
|
||||
|
||||
Args:
|
||||
text: Text to split
|
||||
|
||||
Returns:
|
||||
List of sentences
|
||||
"""
|
||||
# Simple sentence splitting for Turkish text
|
||||
# Split on period, question mark, exclamation, but not on abbreviations
|
||||
|
||||
# Common Turkish abbreviations to preserve
|
||||
abbreviations = ['Dr', 'Prof', 'Av', 'Md', 'Yrd', 'Doç', 'No', 'S', 'vs', 'vb', 'bkz']
|
||||
|
||||
# Replace abbreviations temporarily
|
||||
temp_text = text
|
||||
replacements = {}
|
||||
for i, abbr in enumerate(abbreviations):
|
||||
placeholder = f"__ABBR{i}__"
|
||||
temp_text = temp_text.replace(f"{abbr}.", placeholder)
|
||||
replacements[placeholder] = f"{abbr}."
|
||||
|
||||
# Split sentences
|
||||
sentence_endings = re.compile(r'[.!?]+')
|
||||
sentences = sentence_endings.split(temp_text)
|
||||
|
||||
# Restore abbreviations and clean
|
||||
cleaned_sentences = []
|
||||
for sentence in sentences:
|
||||
# Restore abbreviations
|
||||
for placeholder, original in replacements.items():
|
||||
sentence = sentence.replace(placeholder, original)
|
||||
|
||||
# Clean and add if not empty
|
||||
sentence = sentence.strip()
|
||||
if sentence and len(sentence) > 10: # Minimum sentence length
|
||||
cleaned_sentences.append(sentence)
|
||||
|
||||
return cleaned_sentences
|
||||
|
||||
def _generate_chunk_id(self, document_id: str, chunk_index: int) -> str:
|
||||
"""
|
||||
Generate unique chunk ID.
|
||||
|
||||
Args:
|
||||
document_id: Parent document ID
|
||||
chunk_index: Index of chunk in document
|
||||
|
||||
Returns:
|
||||
Unique chunk ID
|
||||
"""
|
||||
chunk_string = f"{document_id}_chunk_{chunk_index}"
|
||||
chunk_hash = hashlib.md5(chunk_string.encode()).hexdigest()[:8]
|
||||
return f"{document_id}_c{chunk_index}_{chunk_hash}"
|
||||
|
||||
def combine_chunks(self, chunks: List[DocumentChunk]) -> str:
|
||||
"""
|
||||
Combine chunks back into full document text.
|
||||
|
||||
Args:
|
||||
chunks: List of document chunks
|
||||
|
||||
Returns:
|
||||
Combined text
|
||||
"""
|
||||
if not chunks:
|
||||
return ""
|
||||
|
||||
# Sort by chunk index
|
||||
sorted_chunks = sorted(chunks, key=lambda x: x.chunk_index)
|
||||
|
||||
# For overlapping chunks, we need to be careful about duplication
|
||||
# Simple approach: just concatenate with space
|
||||
combined = " ".join([chunk.text for chunk in sorted_chunks])
|
||||
|
||||
return combined
|
||||
@@ -0,0 +1,235 @@
|
||||
# semantic_search/vector_store.py
|
||||
|
||||
import logging
|
||||
import numpy as np
|
||||
from typing import List, Dict, Any, Tuple, Optional
|
||||
from dataclasses import dataclass
|
||||
import json
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@dataclass
|
||||
class Document:
|
||||
"""Represents a document with its embedding and metadata."""
|
||||
id: str
|
||||
text: str
|
||||
embedding: np.ndarray
|
||||
metadata: Dict[str, Any]
|
||||
|
||||
def to_dict(self) -> Dict[str, Any]:
|
||||
"""Convert to dictionary (excluding embedding for serialization)."""
|
||||
return {
|
||||
'id': self.id,
|
||||
'text': self.text,
|
||||
'metadata': self.metadata
|
||||
}
|
||||
|
||||
class VectorStore:
|
||||
"""
|
||||
In-memory vector storage with similarity search capabilities.
|
||||
Future versions can use Faiss, ChromaDB, or other vector databases.
|
||||
"""
|
||||
|
||||
def __init__(self, dimension: int = 768):
|
||||
"""
|
||||
Initialize vector store.
|
||||
|
||||
Args:
|
||||
dimension: Embedding dimension size
|
||||
"""
|
||||
self.dimension = dimension
|
||||
self.documents: List[Document] = []
|
||||
self.embeddings: Optional[np.ndarray] = None
|
||||
self.index_built = False
|
||||
|
||||
logger.info(f"Initialized VectorStore with dimension: {dimension}")
|
||||
|
||||
def add_documents(self,
|
||||
ids: List[str],
|
||||
texts: List[str],
|
||||
embeddings: np.ndarray,
|
||||
metadata: Optional[List[Dict[str, Any]]] = None) -> int:
|
||||
"""
|
||||
Add documents to the vector store.
|
||||
|
||||
Args:
|
||||
ids: Document IDs
|
||||
texts: Document texts
|
||||
embeddings: Document embeddings (N x dimension)
|
||||
metadata: Optional metadata for each document
|
||||
|
||||
Returns:
|
||||
Number of documents added
|
||||
"""
|
||||
if len(ids) != len(texts) or len(ids) != embeddings.shape[0]:
|
||||
raise ValueError("Mismatched lengths for ids, texts, and embeddings")
|
||||
|
||||
if metadata and len(metadata) != len(ids):
|
||||
raise ValueError("Metadata length doesn't match document count")
|
||||
|
||||
# Add documents
|
||||
for i in range(len(ids)):
|
||||
doc = Document(
|
||||
id=ids[i],
|
||||
text=texts[i],
|
||||
embedding=embeddings[i],
|
||||
metadata=metadata[i] if metadata else {}
|
||||
)
|
||||
self.documents.append(doc)
|
||||
|
||||
# Rebuild index
|
||||
self._build_index()
|
||||
|
||||
logger.info(f"Added {len(ids)} documents to vector store. Total: {len(self.documents)}")
|
||||
return len(ids)
|
||||
|
||||
def _build_index(self):
|
||||
"""Build or rebuild the embedding index."""
|
||||
if not self.documents:
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
return
|
||||
|
||||
# Stack all embeddings into a single array
|
||||
self.embeddings = np.vstack([doc.embedding for doc in self.documents])
|
||||
self.index_built = True
|
||||
|
||||
logger.debug(f"Built index with shape: {self.embeddings.shape}")
|
||||
|
||||
def search(self,
|
||||
query_embedding: np.ndarray,
|
||||
top_k: int = 10,
|
||||
threshold: Optional[float] = None) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Search for similar documents using cosine similarity.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
top_k: Number of results to return
|
||||
threshold: Optional similarity threshold (0-1)
|
||||
|
||||
Returns:
|
||||
List of (Document, similarity_score) tuples
|
||||
"""
|
||||
if not self.index_built or self.embeddings is None:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Ensure query is 2D
|
||||
if len(query_embedding.shape) == 1:
|
||||
query_embedding = query_embedding.reshape(1, -1)
|
||||
|
||||
# Compute cosine similarities (assuming normalized embeddings)
|
||||
similarities = np.dot(self.embeddings, query_embedding.T).squeeze()
|
||||
|
||||
# Apply threshold if specified
|
||||
if threshold is not None:
|
||||
valid_indices = np.where(similarities >= threshold)[0]
|
||||
if len(valid_indices) == 0:
|
||||
logger.info(f"No documents above threshold {threshold}")
|
||||
return []
|
||||
similarities = similarities[valid_indices]
|
||||
valid_docs = [self.documents[i] for i in valid_indices]
|
||||
else:
|
||||
valid_docs = self.documents
|
||||
|
||||
# Get top-k indices
|
||||
top_k = min(top_k, len(valid_docs))
|
||||
if top_k == 0:
|
||||
return []
|
||||
|
||||
# Use argpartition for efficiency with large arrays
|
||||
if len(similarities) > top_k:
|
||||
top_indices = np.argpartition(similarities, -top_k)[-top_k:]
|
||||
top_indices = top_indices[np.argsort(similarities[top_indices])[::-1]]
|
||||
else:
|
||||
top_indices = np.argsort(similarities)[::-1]
|
||||
|
||||
# Create results
|
||||
results = []
|
||||
for idx in top_indices:
|
||||
doc = valid_docs[idx] if threshold else self.documents[idx]
|
||||
score = float(similarities[idx])
|
||||
results.append((doc, score))
|
||||
|
||||
logger.info(f"Search returned {len(results)} results (top_k={top_k})")
|
||||
return results
|
||||
|
||||
def hybrid_search(self,
|
||||
query_embedding: np.ndarray,
|
||||
keyword_scores: Dict[str, float],
|
||||
top_k: int = 10,
|
||||
alpha: float = 0.5) -> List[Tuple[Document, float]]:
|
||||
"""
|
||||
Hybrid search combining vector similarity and keyword scores.
|
||||
|
||||
Args:
|
||||
query_embedding: Query embedding vector
|
||||
keyword_scores: Document ID to keyword relevance score mapping
|
||||
top_k: Number of results to return
|
||||
alpha: Weight for vector similarity (1-alpha for keyword score)
|
||||
|
||||
Returns:
|
||||
List of (Document, combined_score) tuples
|
||||
"""
|
||||
if not self.index_built:
|
||||
logger.warning("No documents in vector store")
|
||||
return []
|
||||
|
||||
# Get vector similarities
|
||||
vector_results = self.search(query_embedding, top_k=len(self.documents))
|
||||
|
||||
# Combine scores
|
||||
combined_scores = []
|
||||
for doc, vector_score in vector_results:
|
||||
keyword_score = keyword_scores.get(doc.id, 0.0)
|
||||
# Normalize keyword score to 0-1 range if needed
|
||||
if keyword_score > 1.0:
|
||||
keyword_score = keyword_score / max(keyword_scores.values())
|
||||
|
||||
combined_score = alpha * vector_score + (1 - alpha) * keyword_score
|
||||
combined_scores.append((doc, combined_score))
|
||||
|
||||
# Sort by combined score and return top-k
|
||||
combined_scores.sort(key=lambda x: x[1], reverse=True)
|
||||
results = combined_scores[:top_k]
|
||||
|
||||
logger.info(f"Hybrid search returned {len(results)} results")
|
||||
return results
|
||||
|
||||
def clear(self):
|
||||
"""Clear all documents from the store."""
|
||||
self.documents = []
|
||||
self.embeddings = None
|
||||
self.index_built = False
|
||||
logger.info("Cleared vector store")
|
||||
|
||||
def size(self) -> int:
|
||||
"""Get number of documents in store."""
|
||||
return len(self.documents)
|
||||
|
||||
def get_by_id(self, doc_id: str) -> Optional[Document]:
|
||||
"""Get document by ID."""
|
||||
for doc in self.documents:
|
||||
if doc.id == doc_id:
|
||||
return doc
|
||||
return None
|
||||
|
||||
def get_stats(self) -> Dict[str, Any]:
|
||||
"""Get statistics about the vector store."""
|
||||
stats = {
|
||||
'num_documents': len(self.documents),
|
||||
'dimension': self.dimension,
|
||||
'index_built': self.index_built,
|
||||
'memory_usage_mb': 0
|
||||
}
|
||||
|
||||
if self.embeddings is not None:
|
||||
# Estimate memory usage
|
||||
memory_bytes = self.embeddings.nbytes
|
||||
for doc in self.documents:
|
||||
memory_bytes += len(doc.text.encode('utf-8'))
|
||||
memory_bytes += len(json.dumps(doc.metadata).encode('utf-8'))
|
||||
stats['memory_usage_mb'] = memory_bytes / (1024 * 1024)
|
||||
|
||||
return stats
|
||||
@@ -0,0 +1,21 @@
|
||||
# sigorta_tahkim_mcp_module/__init__.py
|
||||
|
||||
from .client import SigortaTahkimApiClient
|
||||
from .models import (
|
||||
SigortaTahkimSearchRequest,
|
||||
SigortaTahkimDecisionSummary,
|
||||
SigortaTahkimSearchResult,
|
||||
SigortaTahkimDocumentMarkdown,
|
||||
SigortaTahkimSearchWithinMatch,
|
||||
SigortaTahkimSearchWithinResult
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"SigortaTahkimApiClient",
|
||||
"SigortaTahkimSearchRequest",
|
||||
"SigortaTahkimDecisionSummary",
|
||||
"SigortaTahkimSearchResult",
|
||||
"SigortaTahkimDocumentMarkdown",
|
||||
"SigortaTahkimSearchWithinMatch",
|
||||
"SigortaTahkimSearchWithinResult"
|
||||
]
|
||||
@@ -0,0 +1,345 @@
|
||||
# sigorta_tahkim_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from typing import Optional
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import io
|
||||
import math
|
||||
from markitdown import MarkItDown
|
||||
|
||||
from .models import (
|
||||
SigortaTahkimSearchRequest,
|
||||
SigortaTahkimDecisionSummary,
|
||||
SigortaTahkimSearchResult,
|
||||
SigortaTahkimDocumentMarkdown,
|
||||
SigortaTahkimSearchWithinMatch,
|
||||
SigortaTahkimSearchWithinResult
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
if not logger.hasHandlers():
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s - %(name)s - %(levelname)s - %(message)s'
|
||||
)
|
||||
|
||||
|
||||
# Turkish-specific lowercase: İ→i, I→ı (Python's str.lower() doesn't handle these)
|
||||
_TR_UPPER = str.maketrans("İIÇĞÖŞÜ", "iıçğöşü")
|
||||
|
||||
|
||||
def _turkish_lower(text: str) -> str:
|
||||
"""Lowercase with Turkish İ/I handling."""
|
||||
return text.translate(_TR_UPPER).lower()
|
||||
|
||||
|
||||
class SigortaTahkimApiClient:
|
||||
"""
|
||||
API client for searching and retrieving Sigorta Tahkim Komisyonu
|
||||
(Insurance Arbitration Commission) decisions using Tavily Search API
|
||||
for discovery and direct PDF download for content retrieval.
|
||||
|
||||
The commission publishes quarterly PDF journals ("Hakem Karar Dergisi")
|
||||
containing arbitration decisions. There are 64 issues spanning 2010-2025.
|
||||
"""
|
||||
|
||||
TAVILY_API_URL = "https://api.tavily.com/search"
|
||||
BASE_URL = "https://www.sigortatahkim.org"
|
||||
PDF_BASE_URL = "https://www.sigortatahkim.org/content/CmsFiles/"
|
||||
DOCUMENT_MARKDOWN_CHUNK_SIZE = 5000
|
||||
|
||||
def __init__(self, request_timeout: float = 60.0):
|
||||
"""Initialize the Sigorta Tahkim API client."""
|
||||
self.tavily_api_key = os.getenv("TAVILY_API_KEY")
|
||||
if not self.tavily_api_key:
|
||||
self.tavily_api_key = "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"
|
||||
logger.info("Using fallback Tavily API token (development token)")
|
||||
else:
|
||||
logger.info("Using Tavily API key from environment variable")
|
||||
|
||||
self.http_client = httpx.AsyncClient(
|
||||
headers={
|
||||
"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
|
||||
},
|
||||
timeout=httpx.Timeout(request_timeout)
|
||||
)
|
||||
self.markitdown = MarkItDown()
|
||||
|
||||
async def close_client_session(self):
|
||||
"""Close the HTTP client session."""
|
||||
await self.http_client.aclose()
|
||||
logger.info("SigortaTahkimApiClient: HTTP client session closed.")
|
||||
|
||||
def _get_pdf_filename(self, issue_number: int) -> str:
|
||||
"""Get the PDF filename for a given journal issue number."""
|
||||
if issue_number == 4:
|
||||
return "karardergisisayi4.pdf"
|
||||
elif 57 <= issue_number <= 61:
|
||||
return f"revizekd{issue_number}.pdf"
|
||||
else:
|
||||
return f"karardrgs{issue_number}.pdf"
|
||||
|
||||
def _extract_issue_number(self, url: str) -> Optional[str]:
|
||||
"""Extract journal issue number from a sigortatahkim.org URL."""
|
||||
# Pattern: karardrgs{N}.pdf
|
||||
match = re.search(r'karardrgs(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: revizekd{N}.pdf
|
||||
match = re.search(r'revizekd(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: karardergisisayi{N}.pdf
|
||||
match = re.search(r'karardergisisayi(\d+)\.pdf', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
# Pattern: sayı or sayi in URL path with number
|
||||
match = re.search(r'say[ıi]\s*[-:]?\s*(\d+)', url, re.IGNORECASE)
|
||||
if match:
|
||||
return match.group(1)
|
||||
|
||||
return None
|
||||
|
||||
async def search_decisions(
|
||||
self,
|
||||
request: SigortaTahkimSearchRequest
|
||||
) -> SigortaTahkimSearchResult:
|
||||
"""
|
||||
Search for Sigorta Tahkim Komisyonu decisions using Tavily API.
|
||||
|
||||
Args:
|
||||
request: Search request parameters
|
||||
|
||||
Returns:
|
||||
SigortaTahkimSearchResult with matching decisions
|
||||
"""
|
||||
try:
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Authorization": f"Bearer {self.tavily_api_key}"
|
||||
}
|
||||
|
||||
payload = {
|
||||
"query": request.keywords,
|
||||
"country": "turkey",
|
||||
"include_domains": ["sigortatahkim.org"],
|
||||
"max_results": request.pageSize,
|
||||
"search_depth": "advanced"
|
||||
}
|
||||
|
||||
if request.page > 1:
|
||||
logger.warning(f"Tavily API doesn't support pagination. Page {request.page} requested.")
|
||||
|
||||
response = await self.http_client.post(
|
||||
self.TAVILY_API_URL,
|
||||
json=payload,
|
||||
headers=headers
|
||||
)
|
||||
response.raise_for_status()
|
||||
|
||||
data = response.json()
|
||||
logger.info(f"Tavily returned {len(data.get('results', []))} results for Sigorta Tahkim")
|
||||
|
||||
decisions = []
|
||||
for result in data.get("results", []):
|
||||
url = result.get("url", "")
|
||||
title = result.get("title", "").strip()
|
||||
content = result.get("content", "")[:500]
|
||||
|
||||
issue_num = self._extract_issue_number(url)
|
||||
doc_id = issue_num if issue_num else url
|
||||
|
||||
decision = SigortaTahkimDecisionSummary(
|
||||
title=title,
|
||||
document_id=doc_id,
|
||||
content=content,
|
||||
url=url
|
||||
)
|
||||
decisions.append(decision)
|
||||
|
||||
return SigortaTahkimSearchResult(
|
||||
decisions=decisions,
|
||||
total_results=len(data.get("results", [])),
|
||||
page=request.page,
|
||||
pageSize=request.pageSize
|
||||
)
|
||||
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error searching Sigorta Tahkim decisions: {e}")
|
||||
if e.response.status_code == 401:
|
||||
raise Exception("Tavily API authentication failed. Check API key.")
|
||||
raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error searching Sigorta Tahkim decisions: {e}")
|
||||
raise Exception(f"Failed to search Sigorta Tahkim decisions: {str(e)}")
|
||||
|
||||
# Regex pattern to split decisions within a journal issue
|
||||
DECISION_HEADER_PATTERN = re.compile(
|
||||
r'(\d{2}\.\d{2}\.\d{4}\s+Tarih\s+ve\s+K-\d{4}/\d+\s+Sayılı\s+Hakem\s+Kararı)'
|
||||
)
|
||||
# Minimum body length to distinguish real decisions from TOC entries
|
||||
MIN_DECISION_BODY_LENGTH = 1000
|
||||
|
||||
async def _download_and_convert_pdf(self, issue_number: str) -> tuple[str, str]:
|
||||
"""
|
||||
Download a journal issue PDF and convert to markdown.
|
||||
|
||||
Returns:
|
||||
Tuple of (markdown_content, pdf_url)
|
||||
"""
|
||||
issue_num = int(issue_number)
|
||||
filename = self._get_pdf_filename(issue_num)
|
||||
pdf_url = f"{self.PDF_BASE_URL}{filename}"
|
||||
|
||||
logger.info(f"Downloading Sigorta Tahkim PDF: {pdf_url}")
|
||||
|
||||
response = await self.http_client.get(pdf_url, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
|
||||
pdf_stream = io.BytesIO(response.content)
|
||||
# markitdown is sync; offload to thread so PDF parsing doesn't block
|
||||
# the event-loop / other in-flight MCP requests.
|
||||
result = await asyncio.to_thread(
|
||||
self.markitdown.convert_stream, pdf_stream, file_extension=".pdf"
|
||||
)
|
||||
return result.text_content.strip(), pdf_url
|
||||
|
||||
def _split_into_decisions(self, markdown_content: str) -> list[tuple[str, str]]:
|
||||
"""
|
||||
Split markdown content into individual decisions.
|
||||
|
||||
Returns:
|
||||
List of (header, body) tuples for decisions with substantial content.
|
||||
"""
|
||||
parts = self.DECISION_HEADER_PATTERN.split(markdown_content)
|
||||
decisions = []
|
||||
for i in range(1, len(parts) - 1, 2):
|
||||
header = parts[i].strip()
|
||||
body = parts[i + 1].strip() if i + 1 < len(parts) else ""
|
||||
if len(body) >= self.MIN_DECISION_BODY_LENGTH:
|
||||
decisions.append((header, body))
|
||||
return decisions
|
||||
|
||||
async def get_document_markdown(
|
||||
self,
|
||||
issue_number: str,
|
||||
page_number: int = 1
|
||||
) -> SigortaTahkimDocumentMarkdown:
|
||||
"""
|
||||
Retrieve a Sigorta Tahkim journal issue PDF and convert to Markdown.
|
||||
|
||||
Args:
|
||||
issue_number: Journal issue number (e.g., '64')
|
||||
page_number: Page number for paginated content (1-indexed)
|
||||
|
||||
Returns:
|
||||
SigortaTahkimDocumentMarkdown with paginated content
|
||||
"""
|
||||
try:
|
||||
markdown_content, pdf_url = await self._download_and_convert_pdf(issue_number)
|
||||
|
||||
total_length = len(markdown_content)
|
||||
total_pages = max(1, math.ceil(total_length / self.DOCUMENT_MARKDOWN_CHUNK_SIZE))
|
||||
|
||||
start_idx = (page_number - 1) * self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
end_idx = start_idx + self.DOCUMENT_MARKDOWN_CHUNK_SIZE
|
||||
page_content = markdown_content[start_idx:end_idx]
|
||||
|
||||
return SigortaTahkimDocumentMarkdown(
|
||||
document_id=issue_number,
|
||||
markdown_content=page_content,
|
||||
page_number=page_number,
|
||||
total_pages=total_pages,
|
||||
source_url=pdf_url
|
||||
)
|
||||
|
||||
except ValueError:
|
||||
raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').")
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error fetching Sigorta Tahkim issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error processing Sigorta Tahkim issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to process journal issue {issue_number}: {str(e)}")
|
||||
|
||||
async def search_within_issue(
|
||||
self,
|
||||
issue_number: str,
|
||||
keyword: str,
|
||||
max_results: int = 10
|
||||
) -> SigortaTahkimSearchWithinResult:
|
||||
"""
|
||||
Search for a keyword within a specific journal issue's decisions.
|
||||
|
||||
Downloads the PDF, splits into individual decisions, and returns
|
||||
matching decisions sorted by relevance (match count).
|
||||
|
||||
Args:
|
||||
issue_number: Journal issue number (e.g., '64')
|
||||
keyword: Search keyword or phrase in Turkish
|
||||
max_results: Maximum matching decisions to return
|
||||
|
||||
Returns:
|
||||
SigortaTahkimSearchWithinResult with matching decisions
|
||||
"""
|
||||
try:
|
||||
markdown_content, _ = await self._download_and_convert_pdf(issue_number)
|
||||
decisions = self._split_into_decisions(markdown_content)
|
||||
|
||||
logger.info(
|
||||
f"Searching '{keyword}' within issue {issue_number}: "
|
||||
f"{len(decisions)} decisions found"
|
||||
)
|
||||
|
||||
keyword_lower = _turkish_lower(keyword)
|
||||
matches = []
|
||||
|
||||
for header, body in decisions:
|
||||
body_lower = _turkish_lower(body)
|
||||
count = body_lower.count(keyword_lower)
|
||||
if count == 0:
|
||||
continue
|
||||
|
||||
# Extract excerpt around the first match
|
||||
first_pos = body_lower.find(keyword_lower)
|
||||
excerpt_start = max(0, first_pos - 200)
|
||||
excerpt_end = min(len(body), first_pos + len(keyword) + 200)
|
||||
excerpt = body[excerpt_start:excerpt_end].strip()
|
||||
if excerpt_start > 0:
|
||||
excerpt = "..." + excerpt
|
||||
if excerpt_end < len(body):
|
||||
excerpt = excerpt + "..."
|
||||
|
||||
matches.append(SigortaTahkimSearchWithinMatch(
|
||||
decision_header=header,
|
||||
relevance_score=count,
|
||||
excerpt=excerpt,
|
||||
body_length=len(body)
|
||||
))
|
||||
|
||||
# Sort by relevance (highest match count first)
|
||||
matches.sort(key=lambda m: m.relevance_score, reverse=True)
|
||||
matches = matches[:max_results]
|
||||
|
||||
return SigortaTahkimSearchWithinResult(
|
||||
issue_number=issue_number,
|
||||
keyword=keyword,
|
||||
total_decisions=len(decisions),
|
||||
matching_decisions=len(matches),
|
||||
matches=matches
|
||||
)
|
||||
|
||||
except ValueError:
|
||||
raise Exception(f"Invalid issue number: {issue_number}. Must be a number (e.g., '64').")
|
||||
except httpx.HTTPStatusError as e:
|
||||
logger.error(f"HTTP error in search_within issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to fetch journal issue {issue_number}: {str(e)}")
|
||||
except Exception as e:
|
||||
logger.error(f"Error in search_within issue {issue_number}: {e}")
|
||||
raise Exception(f"Failed to search within issue {issue_number}: {str(e)}")
|
||||
@@ -0,0 +1,59 @@
|
||||
# sigorta_tahkim_mcp_module/models.py
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
from typing import List
|
||||
|
||||
|
||||
class SigortaTahkimSearchRequest(BaseModel):
|
||||
"""Request model for searching Sigorta Tahkim Komisyonu decisions via Tavily API."""
|
||||
keywords: str = Field(..., description="Search keywords in Turkish")
|
||||
page: int = Field(1, ge=1, description="Page number (1-indexed)")
|
||||
pageSize: int = Field(10, ge=1, le=50, description="Results per page (1-50)")
|
||||
|
||||
|
||||
class SigortaTahkimDecisionSummary(BaseModel):
|
||||
"""Summary of a Sigorta Tahkim decision from search results."""
|
||||
title: str = Field(..., description="Decision title or journal issue info")
|
||||
document_id: str = Field(..., description="Journal issue number (e.g., '64')")
|
||||
content: str = Field(..., description="Decision summary/excerpt")
|
||||
url: str = Field("", description="Source URL")
|
||||
|
||||
|
||||
class SigortaTahkimSearchResult(BaseModel):
|
||||
"""Response model for Sigorta Tahkim decision search results."""
|
||||
decisions: List[SigortaTahkimDecisionSummary] = Field(
|
||||
default_factory=list,
|
||||
description="List of matching decisions"
|
||||
)
|
||||
total_results: int = Field(0, description="Total number of results")
|
||||
page: int = Field(1, description="Current page number")
|
||||
pageSize: int = Field(10, description="Results per page")
|
||||
|
||||
|
||||
class SigortaTahkimDocumentMarkdown(BaseModel):
|
||||
"""Sigorta Tahkim journal issue converted to Markdown format."""
|
||||
document_id: str = Field(..., description="Journal issue number")
|
||||
markdown_content: str = Field("", description="Document content in Markdown")
|
||||
page_number: int = Field(1, description="Current page number")
|
||||
total_pages: int = Field(1, description="Total number of pages")
|
||||
source_url: str = Field("", description="PDF source URL")
|
||||
|
||||
|
||||
class SigortaTahkimSearchWithinMatch(BaseModel):
|
||||
"""A single matching decision from search within a journal issue."""
|
||||
decision_header: str = Field(..., description="Decision header (date and K-number)")
|
||||
relevance_score: int = Field(0, description="Number of keyword matches")
|
||||
excerpt: str = Field("", description="Matching excerpt with context")
|
||||
body_length: int = Field(0, description="Full decision body length in chars")
|
||||
|
||||
|
||||
class SigortaTahkimSearchWithinResult(BaseModel):
|
||||
"""Response model for search within a journal issue."""
|
||||
issue_number: str = Field(..., description="Journal issue number searched")
|
||||
keyword: str = Field("", description="Search keyword used")
|
||||
total_decisions: int = Field(0, description="Total decisions in issue")
|
||||
matching_decisions: int = Field(0, description="Number of matching decisions")
|
||||
matches: List[SigortaTahkimSearchWithinMatch] = Field(
|
||||
default_factory=list,
|
||||
description="List of matching decisions sorted by relevance"
|
||||
)
|
||||
@@ -1,159 +0,0 @@
|
||||
"""
|
||||
Starlette integration example for Yargı MCP Server
|
||||
|
||||
This module demonstrates how to integrate the Yargı MCP server
|
||||
with a Starlette application, including authentication middleware
|
||||
and custom routing.
|
||||
|
||||
Usage:
|
||||
uvicorn starlette_app:app --host 0.0.0.0 --port 8000
|
||||
"""
|
||||
|
||||
import os
|
||||
from starlette.applications import Starlette
|
||||
from starlette.routing import Mount, Route
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import JSONResponse, PlainTextResponse, RedirectResponse
|
||||
from starlette.middleware import Middleware
|
||||
from starlette.middleware.cors import CORSMiddleware
|
||||
from starlette.middleware.authentication import AuthenticationMiddleware
|
||||
from starlette.authentication import (
|
||||
AuthenticationBackend, AuthCredentials, SimpleUser, AuthenticationError
|
||||
)
|
||||
|
||||
# Import the main MCP app
|
||||
from mcp_server_main import app as mcp_server
|
||||
|
||||
# Simple token authentication backend
|
||||
class TokenAuthBackend(AuthenticationBackend):
|
||||
async def authenticate(self, request):
|
||||
auth_header = request.headers.get("Authorization")
|
||||
expected_token = os.getenv("API_TOKEN")
|
||||
|
||||
# Skip auth for health check and public endpoints
|
||||
if request.url.path in ["/health", "/", "/login"]:
|
||||
return None
|
||||
|
||||
if not expected_token:
|
||||
# No token configured, allow all
|
||||
return AuthCredentials(["authenticated"]), SimpleUser("anonymous")
|
||||
|
||||
if not auth_header:
|
||||
raise AuthenticationError("Authorization header required")
|
||||
|
||||
try:
|
||||
scheme, token = auth_header.split()
|
||||
if scheme.lower() != "bearer":
|
||||
raise AuthenticationError("Invalid authentication scheme")
|
||||
|
||||
if token != expected_token:
|
||||
raise AuthenticationError("Invalid token")
|
||||
|
||||
return AuthCredentials(["authenticated"]), SimpleUser("user")
|
||||
except ValueError:
|
||||
raise AuthenticationError("Invalid authorization header format")
|
||||
|
||||
# Homepage
|
||||
async def homepage(request: Request):
|
||||
return JSONResponse({
|
||||
"service": "Yargı MCP Server",
|
||||
"version": "0.1.0",
|
||||
"endpoints": {
|
||||
"mcp": "/mcp-server/mcp/",
|
||||
"api": "/api/",
|
||||
"health": "/health"
|
||||
}
|
||||
})
|
||||
|
||||
# API info endpoint
|
||||
async def api_info(request: Request):
|
||||
if not request.user.is_authenticated:
|
||||
return JSONResponse({"error": "Authentication required"}, status_code=401)
|
||||
|
||||
return JSONResponse({
|
||||
"authenticated_as": request.user.display_name,
|
||||
"available_tools": len(mcp_server._tool_manager._tools),
|
||||
"databases": [
|
||||
"Yargıtay", "Danıştay", "Emsal", "Uyuşmazlık",
|
||||
"Anayasa", "KIK", "Rekabet", "Bedesten"
|
||||
]
|
||||
})
|
||||
|
||||
# Health check
|
||||
async def health_check(request: Request):
|
||||
return JSONResponse({
|
||||
"status": "healthy",
|
||||
"service": "Yargı MCP Server"
|
||||
})
|
||||
|
||||
# Login example (returns token for demo)
|
||||
async def login(request: Request):
|
||||
token = os.getenv("API_TOKEN", "demo-token")
|
||||
return JSONResponse({
|
||||
"message": "Use this token in Authorization header",
|
||||
"example": f"Authorization: Bearer {token}",
|
||||
"note": "Set API_TOKEN environment variable to change token"
|
||||
})
|
||||
|
||||
# Create MCP ASGI app
|
||||
mcp_app = mcp_server.http_app(path='/mcp')
|
||||
|
||||
# Configure middleware
|
||||
middleware = [
|
||||
Middleware(
|
||||
CORSMiddleware,
|
||||
allow_origins=os.getenv("ALLOWED_ORIGINS", "*").split(","),
|
||||
allow_credentials=True,
|
||||
allow_methods=["*"],
|
||||
allow_headers=["*"],
|
||||
),
|
||||
Middleware(AuthenticationMiddleware, backend=TokenAuthBackend()),
|
||||
]
|
||||
|
||||
# Create routes
|
||||
routes = [
|
||||
Route("/", homepage),
|
||||
Route("/health", health_check),
|
||||
Route("/login", login),
|
||||
Route("/api/info", api_info),
|
||||
Mount("/mcp-server", app=mcp_app),
|
||||
]
|
||||
|
||||
# Create Starlette app
|
||||
app = Starlette(
|
||||
routes=routes,
|
||||
middleware=middleware,
|
||||
lifespan=mcp_app.lifespan
|
||||
)
|
||||
|
||||
# Nested mount example
|
||||
def create_nested_app():
|
||||
"""Example of nested mounting for complex routing structures"""
|
||||
|
||||
# Create inner app with MCP
|
||||
inner_app = Starlette(
|
||||
routes=[Mount("/services", app=mcp_app)],
|
||||
middleware=middleware
|
||||
)
|
||||
|
||||
# Create outer app
|
||||
outer_app = Starlette(
|
||||
routes=[
|
||||
Route("/", homepage),
|
||||
Mount("/v1", app=inner_app),
|
||||
],
|
||||
lifespan=mcp_app.lifespan
|
||||
)
|
||||
|
||||
# MCP would be available at /v1/services/mcp/
|
||||
return outer_app
|
||||
|
||||
# Export both apps
|
||||
nested_app = create_nested_app()
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
print("Starting Starlette app with authentication...")
|
||||
print("Set API_TOKEN environment variable to enable authentication")
|
||||
print("Example: API_TOKEN=secret-token python starlette_app.py")
|
||||
uvicorn.run(app, host="0.0.0.0", port=8000)
|
||||
@@ -1,25 +0,0 @@
|
||||
import os, stripe
|
||||
from clerk_backend_api import Clerk # Clerk backend SDK
|
||||
from fastapi import APIRouter, Request, HTTPException
|
||||
|
||||
router = APIRouter()
|
||||
stripe.api_key = os.getenv("STRIPE_SECRET")
|
||||
clerk = Clerk(bearer_auth=os.getenv("CLERK_SECRET_KEY"))
|
||||
|
||||
@router.post("/stripe/webhook")
|
||||
async def stripe_hook(req: Request):
|
||||
payload, sig = await req.body(), req.headers["stripe-signature"]
|
||||
try:
|
||||
event = stripe.Webhook.construct_event( # Stripe-recommended verify
|
||||
payload, sig, os.getenv("STRIPE_WEBHOOK_SECRET"))
|
||||
except stripe.error.SignatureVerificationError:
|
||||
raise HTTPException(400, "Bad sig")
|
||||
|
||||
if event["type"] == "customer.subscription.updated":
|
||||
item = event["data"]["object"]["items"]["data"][0]
|
||||
plan = item["price"]["nickname"] # "Pro", "Enterprise"…
|
||||
userID = event["data"]["object"]["metadata"]["clerk_user_id"]
|
||||
clerk.users.update_user_metadata( # merge into unsafe_metadata
|
||||
userID, unsafe_metadata={"plan": plan})
|
||||
return {"ok": True}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# uyusmazlik_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
import aiohttp
|
||||
from bs4 import BeautifulSoup
|
||||
from typing import Dict, Any, List, Optional, Union, Tuple
|
||||
import logging
|
||||
@@ -9,7 +9,7 @@ import html
|
||||
import re
|
||||
import io
|
||||
from markitdown import MarkItDown
|
||||
from urllib.parse import urljoin, urlencode # urlencode for aiohttp form data
|
||||
from urllib.parse import urljoin
|
||||
|
||||
from .models import (
|
||||
UyusmazlikSearchRequest,
|
||||
@@ -56,17 +56,21 @@ class UyusmazlikApiClient:
|
||||
# Individual documents are fetched by their full URLs obtained from search results.
|
||||
|
||||
def __init__(self, request_timeout: float = 30.0):
|
||||
self.request_timeout = request_timeout # Store timeout for aiohttp and httpx
|
||||
# Headers for aiohttp search. httpx for docs will create its own.
|
||||
self.default_aiohttp_search_headers = {
|
||||
"Accept": "*/*", # Mimicking browser headers provided by user
|
||||
self.request_timeout = request_timeout
|
||||
# Create shared httpx client for all requests
|
||||
self.http_client = httpx.AsyncClient(
|
||||
base_url=self.BASE_URL,
|
||||
headers={
|
||||
"Accept": "*/*",
|
||||
"Accept-Encoding": "gzip, deflate, br, zstd",
|
||||
"Accept-Language": "tr-TR,tr;q=0.9,en-US;q=0.8,en;q=0.7",
|
||||
"X-Requested-With": "XMLHttpRequest",
|
||||
"Origin": self.BASE_URL,
|
||||
"Referer": self.BASE_URL + "/",
|
||||
|
||||
}
|
||||
},
|
||||
timeout=request_timeout,
|
||||
verify=False
|
||||
)
|
||||
|
||||
|
||||
async def search_decisions(
|
||||
@@ -107,32 +111,36 @@ class UyusmazlikApiClient:
|
||||
add_to_form_data("Hepsi", params.hepsi)
|
||||
add_to_form_data("Herhangibirisi", params.herhangi_birisi)
|
||||
add_to_form_data("NotHepsi", params.not_hepsi)
|
||||
# X-Requested-With is handled by default_aiohttp_search_headers
|
||||
|
||||
search_url = urljoin(self.BASE_URL, self.SEARCH_ENDPOINT)
|
||||
# For aiohttp, data for application/x-www-form-urlencoded should be a dict or str.
|
||||
# Using urlencode for list of tuples.
|
||||
encoded_form_payload = urlencode(form_data_list, encoding='UTF-8')
|
||||
# Convert form data to dict for httpx
|
||||
form_data_dict = {}
|
||||
for key, value in form_data_list:
|
||||
if key in form_data_dict:
|
||||
# Handle multiple values (like KararSonucuList)
|
||||
if not isinstance(form_data_dict[key], list):
|
||||
form_data_dict[key] = [form_data_dict[key]]
|
||||
form_data_dict[key].append(value)
|
||||
else:
|
||||
form_data_dict[key] = value
|
||||
|
||||
logger.info(f"UyusmazlikApiClient (aiohttp): Performing search to {search_url} with form_data: {encoded_form_payload}")
|
||||
|
||||
html_content = ""
|
||||
aiohttp_headers = self.default_aiohttp_search_headers.copy()
|
||||
aiohttp_headers["Content-Type"] = "application/x-www-form-urlencoded; charset=UTF-8"
|
||||
logger.info(f"UyusmazlikApiClient (httpx): Performing search to {self.SEARCH_ENDPOINT} with form_data: {form_data_dict}")
|
||||
|
||||
try:
|
||||
# Create a new session for each call for simplicity with aiohttp here
|
||||
async with aiohttp.ClientSession(headers=aiohttp_headers) as session:
|
||||
async with session.post(search_url, data=encoded_form_payload, timeout=self.request_timeout) as response:
|
||||
response.raise_for_status() # Raises ClientResponseError for 400-599
|
||||
html_content = await response.text(encoding='utf-8') # Ensure correct encoding
|
||||
logger.debug("UyusmazlikApiClient (aiohttp): Received HTML response for search.")
|
||||
# Use shared httpx client
|
||||
response = await self.http_client.post(
|
||||
self.SEARCH_ENDPOINT,
|
||||
data=form_data_dict,
|
||||
headers={"Content-Type": "application/x-www-form-urlencoded; charset=UTF-8"}
|
||||
)
|
||||
response.raise_for_status()
|
||||
html_content = response.text
|
||||
logger.debug("UyusmazlikApiClient (httpx): Received HTML response for search.")
|
||||
|
||||
except aiohttp.ClientError as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): HTTP client error during search: {e}")
|
||||
except httpx.HTTPError as e:
|
||||
logger.error(f"UyusmazlikApiClient (httpx): HTTP client error during search: {e}")
|
||||
raise # Re-raise to be handled by the MCP tool
|
||||
except Exception as e:
|
||||
logger.error(f"UyusmazlikApiClient (aiohttp): Error processing search request: {e}")
|
||||
logger.error(f"UyusmazlikApiClient (httpx): Error processing search request: {e}")
|
||||
raise
|
||||
|
||||
# --- HTML Parsing (remains the same as previous version) ---
|
||||
@@ -217,7 +225,6 @@ class UyusmazlikApiClient:
|
||||
try:
|
||||
# Using a new httpx.AsyncClient instance for this GET request for simplicity
|
||||
async with httpx.AsyncClient(verify=False, timeout=self.request_timeout) as doc_fetch_client:
|
||||
|
||||
get_response = await doc_fetch_client.get(document_url, headers={"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"})
|
||||
get_response.raise_for_status()
|
||||
html_content_from_api = get_response.text
|
||||
@@ -226,7 +233,7 @@ class UyusmazlikApiClient:
|
||||
logger.warning(f"UyusmazlikApiClient: Received empty or non-string HTML from URL {document_url}.")
|
||||
return UyusmazlikDocumentMarkdown(source_url=document_url, markdown_content=None)
|
||||
|
||||
markdown_content = self._convert_html_to_markdown_uyusmazlik(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown_uyusmazlik, html_content_from_api)
|
||||
return UyusmazlikDocumentMarkdown(source_url=document_url, markdown_content=markdown_content)
|
||||
except httpx.RequestError as e:
|
||||
logger.error(f"UyusmazlikApiClient (httpx for docs): HTTP error fetching Uyuşmazlık document from {document_url}: {e}")
|
||||
@@ -236,5 +243,9 @@ class UyusmazlikApiClient:
|
||||
raise
|
||||
|
||||
async def close_client_session(self):
|
||||
|
||||
"""Close the shared httpx client session."""
|
||||
if hasattr(self, 'http_client') and self.http_client:
|
||||
await self.http_client.aclose()
|
||||
logger.info("UyusmazlikApiClient: HTTP client session closed.")
|
||||
else:
|
||||
logger.info("UyusmazlikApiClient: No persistent client session from __init__ to close.")
|
||||
@@ -1,5 +1,6 @@
|
||||
# yargitay_mcp_module/client.py
|
||||
|
||||
import asyncio
|
||||
import httpx
|
||||
from bs4 import BeautifulSoup # Still needed for pre-processing HTML before markitdown
|
||||
from typing import Dict, Any, List, Optional
|
||||
@@ -159,7 +160,7 @@ class YargitayOfficialApiClient:
|
||||
logger.error(f"YargitayOfficialApiClient: 'data' field in API response is not a string or not found (ID: {id}).")
|
||||
raise ValueError("Expected HTML content not found in API response's 'data' field.")
|
||||
|
||||
markdown_content = self._convert_html_to_markdown(html_content_from_api)
|
||||
markdown_content = await asyncio.to_thread(self._convert_html_to_markdown, html_content_from_api)
|
||||
|
||||
return YargitayDocumentMarkdown(
|
||||
id=id,
|
||||
|
||||
Reference in New Issue
Block a user