diff --git a/services/core-api/src/clients/ai_client.py b/services/core-api/src/clients/ai_client.py new file mode 100644 index 0000000..9ab377e --- /dev/null +++ b/services/core-api/src/clients/ai_client.py @@ -0,0 +1,197 @@ +""" +Core-AI HTTP Client + +Provides interface to Core-AI service for AI performance metrics. +""" +import httpx +from typing import Optional, Dict, List, Any +from src.logging_config import get_logger +from src.config import get_settings + +logger = get_logger(__name__) +settings = get_settings() + + +class CoreAIClient: + """ + HTTP client for Core-AI service + + Provides access to AI performance metrics, tool execution stats, + and memory system monitoring. + """ + + def __init__( + self, + base_url: Optional[str] = None, + timeout: int = 10 + ): + """ + Initialize Core-AI client + + Args: + base_url: Core-AI base URL (default from settings) + timeout: Request timeout in seconds + """ + self.base_url = (base_url or getattr(settings, 'core_ai_base_url', 'http://core-ai:8086')).rstrip("/") + self.timeout = timeout + self.client = httpx.AsyncClient(timeout=self.timeout) + + async def close(self): + """Close the HTTP client""" + await self.client.aclose() + + async def health_check(self) -> bool: + """ + Check if Core-AI service is accessible + + Returns: + True if accessible, False otherwise + """ + try: + response = await self.client.get(f"{self.base_url}/health") + return response.status_code == 200 + except Exception as e: + logger.error(f"Core-AI health check failed: {e}") + return False + + async def get_metrics(self) -> Dict[str, Any]: + """ + Get comprehensive AI performance metrics + + Returns: + Dict with agent performance, tool execution, memory stats + + Example: + { + "uptime_seconds": 3600, + "timestamp": "2025-12-03T20:00:00Z", + "agent": { + "total_requests": 100, + "avg_response_time_ms": 1250.5, + "p95_response_time_ms": 3200.0, + ... + }, + "tools": { + "total_calls": 250, + "success_rate": 0.98, + "top_tools": {...} + }, + "memory": { + "tier1_hit_rate": 0.85, + ... + }, + ... + } + """ + try: + response = await self.client.get(f"{self.base_url}/metrics") + response.raise_for_status() + return response.json() + except httpx.HTTPStatusError as e: + logger.error(f"Failed to get metrics: HTTP {e.response.status_code}") + raise + except Exception as e: + logger.error(f"Failed to get metrics: {e}") + raise + + async def get_recent_errors(self, limit: int = 20) -> List[Dict[str, Any]]: + """ + Get recent request errors + + Args: + limit: Maximum number of errors to return + + Returns: + List of error records with timestamps + + Example: + [ + { + "timestamp": "2025-12-03T19:45:12Z", + "agent_type": "pydantic", + "error": "Connection timeout", + "duration_ms": 5000 + }, + ... + ] + """ + try: + response = await self.client.get( + f"{self.base_url}/metrics/errors", + params={"limit": limit} + ) + response.raise_for_status() + data = response.json() + return data.get("errors", []) + except Exception as e: + logger.error(f"Failed to get recent errors: {e}") + raise + + async def get_tool_failures(self, limit: int = 20) -> List[Dict[str, Any]]: + """ + Get recent tool execution failures + + Args: + limit: Maximum number of failures to return + + Returns: + List of tool failure records + + Example: + [ + { + "timestamp": "2025-12-03T19:50:30Z", + "tool_name": "list_containers", + "error": "Connection refused", + "duration_ms": 150 + }, + ... + ] + """ + try: + response = await self.client.get( + f"{self.base_url}/metrics/tool-failures", + params={"limit": limit} + ) + response.raise_for_status() + data = response.json() + return data.get("failures", []) + except Exception as e: + logger.error(f"Failed to get tool failures: {e}") + raise + + async def reset_metrics(self) -> bool: + """ + Reset all metrics (admin operation) + + Returns: + True if successful + """ + try: + response = await self.client.post(f"{self.base_url}/metrics/reset") + response.raise_for_status() + logger.info("Successfully reset Core-AI metrics") + return True + except Exception as e: + logger.error(f"Failed to reset metrics: {e}") + raise + + async def __aenter__(self): + """Async context manager entry""" + return self + + async def __aexit__(self, exc_type, exc_val, exc_tb): + """Async context manager exit""" + await self.close() + + +# Singleton instance +_ai_client: Optional[CoreAIClient] = None + + +def get_ai_client() -> CoreAIClient: + """Get singleton Core-AI client instance""" + global _ai_client + if _ai_client is None: + _ai_client = CoreAIClient() + return _ai_client diff --git a/services/core-api/src/config.py b/services/core-api/src/config.py index 29cf463..88f6b33 100644 --- a/services/core-api/src/config.py +++ b/services/core-api/src/config.py @@ -116,6 +116,9 @@ class Settings(BaseSettings): kuma_password: str = KUMA_PASSWORD kuma_api_key: str = KUMA_API_KEY + # Core-AI Service (AI performance metrics) + core_ai_base_url: str = "http://core-ai:8086" + # OIDC Authentication (Authentik) oidc_enabled: bool = False # Set to True to require authentication oidc_issuer: str = "https://auth.schweitz.net/application/o/core-api/" diff --git a/services/core-api/src/controllers/ai_controller.py b/services/core-api/src/controllers/ai_controller.py new file mode 100644 index 0000000..2a52b9f --- /dev/null +++ b/services/core-api/src/controllers/ai_controller.py @@ -0,0 +1,219 @@ +""" +AI Metrics Proxy Controller + +Provides proxy endpoints to Core-AI service metrics. +Allows external access to AI performance stats via core-api. +""" +from fastapi import APIRouter, HTTPException +from typing import Dict, List, Any +from src.clients.ai_client import get_ai_client +from src.logging_config import get_logger + +logger = get_logger(__name__) + +# Create router +router = APIRouter( + prefix="/ai", + tags=["AI Metrics"] +) + + +@router.get( + "/health", + summary="Check Core-AI service health", + description="Verify that the Core-AI service is accessible and responding" +) +async def ai_health_check(): + """ + Check if Core-AI service is healthy + + Returns: + Health status and availability + """ + try: + ai_client = get_ai_client() + is_healthy = await ai_client.health_check() + + return { + "service": "core-ai", + "status": "healthy" if is_healthy else "unhealthy", + "accessible": is_healthy + } + + except Exception as e: + logger.error(f"AI health check failed: {e}") + return { + "service": "core-ai", + "status": "error", + "accessible": False, + "error": str(e) + } + + +@router.get( + "/metrics", + response_model=Dict[str, Any], + summary="Get comprehensive AI performance metrics", + description="Returns detailed metrics including agent performance, tool execution stats, memory system metrics, and user activity" +) +async def get_ai_metrics(): + """ + Proxy endpoint for Core-AI metrics + + Returns comprehensive AI performance data: + - Agent request statistics (total, by type, response times) + - Response time percentiles (p50, p95, p99) + - Tool execution metrics (calls, success rates, durations) + - Memory system statistics (cache hits, consolidations) + - User activity tracking + - Concurrency metrics + + Returns: + Dict with all collected metrics + + Raises: + HTTPException: If Core-AI is unreachable or returns error + """ + try: + ai_client = get_ai_client() + metrics = await ai_client.get_metrics() + return metrics + + except Exception as e: + logger.error(f"Failed to fetch AI metrics: {e}") + raise HTTPException( + status_code=503, + detail=f"Core-AI service unavailable: {str(e)}" + ) + + +@router.get( + "/metrics/errors", + response_model=Dict[str, Any], + summary="Get recent request errors", + description="Returns recent AI agent request errors with timestamps and details" +) +async def get_ai_errors(limit: int = 20): + """ + Get recent AI request errors + + Args: + limit: Maximum number of errors to return (default: 20) + + Returns: + Dict with error list and total count + + Example response: + { + "errors": [ + { + "timestamp": "2025-12-03T19:45:12Z", + "agent_type": "pydantic", + "error": "Connection timeout", + "duration_ms": 5000 + } + ], + "total": 1 + } + """ + try: + ai_client = get_ai_client() + errors = await ai_client.get_recent_errors(limit=limit) + + return { + "errors": errors, + "total": len(errors) + } + + except Exception as e: + logger.error(f"Failed to fetch AI errors: {e}") + raise HTTPException( + status_code=503, + detail=f"Core-AI service unavailable: {str(e)}" + ) + + +@router.get( + "/metrics/tool-failures", + response_model=Dict[str, Any], + summary="Get recent tool execution failures", + description="Returns recent tool execution failures with error details" +) +async def get_ai_tool_failures(limit: int = 20): + """ + Get recent tool execution failures + + Args: + limit: Maximum number of failures to return (default: 20) + + Returns: + Dict with failure list and total count + + Example response: + { + "failures": [ + { + "timestamp": "2025-12-03T19:50:30Z", + "tool_name": "list_containers", + "error": "Connection refused", + "duration_ms": 150 + } + ], + "total": 1 + } + """ + try: + ai_client = get_ai_client() + failures = await ai_client.get_tool_failures(limit=limit) + + return { + "failures": failures, + "total": len(failures) + } + + except Exception as e: + logger.error(f"Failed to fetch tool failures: {e}") + raise HTTPException( + status_code=503, + detail=f"Core-AI service unavailable: {str(e)}" + ) + + +@router.post( + "/metrics/reset", + summary="Reset all AI metrics (admin)", + description="Clear all collected metrics. This is an administrative operation that resets all counters and history." +) +async def reset_ai_metrics(): + """ + Reset all AI metrics (admin operation) + + Clears all collected metrics including: + - Request history + - Tool execution stats + - Memory system metrics + - Error logs + + Returns: + Success confirmation + + Note: + This is an administrative operation that should be used carefully. + All historical data will be lost. + """ + try: + ai_client = get_ai_client() + await ai_client.reset_metrics() + + logger.info("AI metrics reset successfully") + return { + "success": True, + "message": "AI metrics reset successfully" + } + + except Exception as e: + logger.error(f"Failed to reset AI metrics: {e}") + raise HTTPException( + status_code=503, + detail=f"Core-AI service unavailable: {str(e)}" + ) diff --git a/services/core-api/src/main.py b/services/core-api/src/main.py index 84946fe..9421594 100644 --- a/services/core-api/src/main.py +++ b/services/core-api/src/main.py @@ -13,6 +13,7 @@ from src.controllers.infrastructure_controller import infrastructure_controller from src.controllers.tools_controller import tools_controller from src.controllers.health_controller import health_controller from src.controllers.static_controller import static_controller +from src.controllers.ai_controller import router as ai_router from src.security import initialize_oidc # Initialize settings @@ -150,6 +151,7 @@ app.include_router(health_controller.router) # / and /health app.include_router(tools_controller.router) # /web-scraper/scrape app.include_router(infrastructure_controller.router) # /infrastructure/* app.include_router(static_controller.router) # /static/* +app.include_router(ai_router) # /ai/* # Global exception handler diff --git a/services/core-api/static/widgets/ai-stats.html b/services/core-api/static/widgets/ai-stats.html new file mode 100644 index 0000000..6ab68c3 --- /dev/null +++ b/services/core-api/static/widgets/ai-stats.html @@ -0,0 +1,491 @@ + + + + + + AI Performance Stats + + + +
+
+
+
Loading AI performance metrics... +
+
+ + + +