#!/usr/bin/env python3 """ SamCloud Advanced Home Lab Dashboard ==================================== A next-generation, fully dynamic home server monitoring solution Built for macOS 15.3.2 Sequoia with comprehensive service discovery Key Features: - Real-time multi-source service discovery (Docker, native apps, Homebrew, K8s) - Apple HIG 2025 compliant responsive grid layout (zero scroll) - Live health monitoring with API endpoint validation - Interactive service management with detailed metrics - Enterprise-grade monitoring with WebSocket real-time updates - Prometheus/Grafana integration ready - Extensible plugin architecture for custom services Author: GitHub Copilot for SamCloud Date: June 23, 2025 """ import asyncio import json import os import platform import psutil import requests import socket import subprocess import time import logging import re import yaml from datetime import datetime, timedelta from pathlib import Path from typing import Dict, List, Optional, Any, Tuple, Union from collections import defaultdict, deque from dataclasses import dataclass, asdict from concurrent.futures import ThreadPoolExecutor, as_completed from flask import Flask, render_template_string, jsonify, request from flask_socketio import SocketIO, emit from flask_cors import CORS import threading # Configure logging logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(name)s - %(levelname)s - %(message)s') logger = logging.getLogger(__name__) app = Flask(__name__) app.config['SECRET_KEY'] = 'samcloud-advanced-secret-2025' CORS(app) socketio = SocketIO(app, cors_allowed_origins="*", async_mode='threading', ping_timeout=60, ping_interval=25) @dataclass class ServiceHealth: """Comprehensive service health and metrics""" name: str display_name: str category: str status: str # 'healthy', 'unhealthy', 'warning', 'unknown' url: Optional[str] = None external_url: Optional[str] = None # Add separate external URL field port: Optional[int] = None response_time: Optional[float] = None last_check: Optional[str] = None # Change to string for JSON serialization error_message: Optional[str] = None cpu_usage: Optional[float] = None memory_usage: Optional[float] = None uptime: Optional[str] = None version: Optional[str] = None api_health: Optional[bool] = None container_id: Optional[str] = None process_id: Optional[int] = None icon: str = '🔧' color: str = '#666666' priority: int = 1 # 1=critical, 2=important, 3=optional dependencies: Optional[List[str]] = None logs: Optional[List[str]] = None class ServiceDiscovery: """Advanced service discovery engine""" def __init__(self): self.services = {} self.service_definitions = self._load_service_definitions() self.metrics_history = defaultdict(lambda: deque(maxlen=100)) def _load_service_definitions(self) -> Dict[str, Dict]: """Load comprehensive service definitions with confirmed SamCloud domains only""" # Only domains confirmed to exist based on your Cloudflare DNS confirmed_domains = { 'jellyfin': 'https://jellyfin.samcloud.ca', 'jellyseerr': 'https://req.samcloud.ca', 'immich': 'https://immich.samcloud.ca', 'filebrowser': 'https://index.samcloud.ca', 'openwebui': 'https://ai.samcloud.ca', 'code-server': 'https://code.samcloud.ca', 'navidrome': 'https://music.samcloud.ca', 'homepage': 'https://home.samcloud.ca', 'uptime-kuma': 'wg.samcloud.ca', # This one might be different 'nginx-proxy-manager': 'wg.samcloud.ca' # This one might be different } return { # Media Management (*arr stack) 'sonarr': { 'display_name': 'Sonarr', 'category': '🎬 Media Management', 'icon': '📺', 'color': '#35c5f0', 'ports': [8989], 'api_path': '/api/v3/system/status', 'api_key_required': True, 'priority': 1, 'external_url': confirmed_domains.get('sonarr') }, 'radarr': { 'display_name': 'Radarr', 'category': '🎬 Media Management', 'icon': '🎬', 'color': '#ffc230', 'ports': [7878], 'api_path': '/api/v3/system/status', 'api_key_required': True, 'priority': 1, 'external_url': confirmed_domains.get('radarr') }, 'lidarr': { 'display_name': 'Lidarr', 'category': '🎵 Music Management', 'icon': '🎵', 'color': '#159552', 'ports': [8686], 'api_path': '/api/v1/system/status', 'api_key_required': True, 'priority': 2, 'external_url': confirmed_domains.get('lidarr') }, 'bazarr': { 'display_name': 'Bazarr', 'category': '📝 Subtitles', 'icon': '📝', 'color': '#ffd700', 'ports': [6767], 'api_path': '/api/system/status', 'api_key_required': True, 'priority': 2, 'external_url': confirmed_domains.get('bazarr') }, 'prowlarr': { 'display_name': 'Prowlarr', 'category': '🔍 Indexers', 'icon': '🔍', 'color': '#ff6b6b', 'ports': [9696], 'api_path': '/api/v1/system/status', 'api_key_required': True, 'priority': 1, 'external_url': confirmed_domains.get('prowlarr') }, 'jackett': { 'display_name': 'Jackett', 'category': '🔍 Indexers', 'icon': '🧥', 'color': '#9f4f96', 'ports': [9117], 'api_path': '/api/v2.0/server/config', 'priority': 2, 'external_url': confirmed_domains.get('jackett') }, 'jellyseerr': { 'display_name': 'Jellyseerr', 'category': '📋 Requests', 'icon': '📋', 'color': '#7c4dff', 'ports': [5055], 'api_path': '/api/v1/status', 'priority': 1, 'external_url': confirmed_domains.get('jellyseerr') }, # Download Clients 'qbittorrent': { 'display_name': 'qBittorrent', 'category': '⬇️ Downloads', 'icon': '🌊', 'color': '#2196f3', 'ports': [8080], 'api_path': '/api/v2/app/version', 'login_required': True, 'priority': 1, 'external_url': confirmed_domains.get('qbittorrent') }, # Media Servers 'jellyfin': { 'display_name': 'Jellyfin', 'category': '🎥 Media Servers', 'icon': '🐙', 'color': '#00a4dc', 'ports': [8096], 'api_path': '/System/Info', 'native_app': True, 'priority': 1, 'external_url': confirmed_domains.get('jellyfin') }, 'navidrome': { 'display_name': 'Navidrome', 'category': '🎵 Music Streaming', 'icon': '🎼', 'color': '#663399', 'ports': [4533], 'api_path': '/api/ping', 'priority': 1, 'external_url': confirmed_domains.get('navidrome') }, 'spotspot': { 'display_name': 'SpotSpot', 'category': '🎵 Music Streaming', 'icon': '🎧', 'color': '#1db954', 'ports': [6544], 'api_path': '/health', 'priority': 2, 'external_url': confirmed_domains.get('spotspot') }, # AI/ML Services 'openwebui': { 'display_name': 'Open WebUI', 'category': '🤖 AI/ML', 'icon': '🧠', 'color': '#9c27b0', 'ports': [3000], 'api_path': '/api/v1/models', 'priority': 2, 'external_url': confirmed_domains.get('openwebui') }, # Infrastructure & Monitoring 'uptime-kuma': { 'display_name': 'Uptime Kuma', 'category': '📊 Monitoring', 'icon': '📈', 'color': '#ff9800', 'ports': [3001], 'api_path': '/api/status-page', 'priority': 1, 'external_url': confirmed_domains.get('uptime-kuma') }, 'nginx-proxy-manager': { 'display_name': 'Nginx Proxy Manager', 'category': '🌐 Proxy & Load Balancer', 'icon': '🌐', 'color': '#009639', 'ports': [81], 'api_path': '/api', 'priority': 1, 'external_url': confirmed_domains.get('nginx-proxy-manager') }, 'caddy': { 'display_name': 'Caddy', 'category': '🌐 Web Server', 'icon': '⚡', 'color': '#1f88e5', 'ports': [443, 80], 'native_service': True, 'priority': 1, 'external_url': confirmed_domains.get('caddy') }, # File Management 'filebrowser': { 'display_name': 'File Browser', 'category': '📁 File Management', 'icon': '📂', 'color': '#607d8b', 'ports': [8833], 'api_path': '/api/public/settings', 'priority': 2, 'external_url': confirmed_domains.get('filebrowser') }, 'immich': { 'display_name': 'Immich', 'category': '📷 Photo Management', 'icon': '📸', 'color': '#4285f4', 'ports': [2283], 'api_path': '/api/server-info', 'priority': 1, 'external_url': confirmed_domains.get('immich') }, # Dashboards 'homepage': { 'display_name': 'Homepage', 'category': '📊 Dashboards', 'icon': '🏠', 'color': '#4caf50', 'ports': [10001], 'api_path': '/api/config', 'priority': 3, 'external_url': confirmed_domains.get('homepage') }, 'dashy': { 'display_name': 'Dashy', 'category': '📊 Dashboards', 'icon': '🎛️', 'color': '#00af87', 'ports': [8081], 'api_path': '/status', 'priority': 3, 'external_url': confirmed_domains.get('dashy') }, # System Services 'dnsmasq': { 'display_name': 'DNS (dnsmasq)', 'category': '🌐 Network Services', 'icon': '🔗', 'color': '#ff5722', 'ports': [53], 'homebrew_service': True, 'priority': 1 }, 'ddclient': { 'display_name': 'Dynamic DNS Client', 'category': '🌐 Network Services', 'icon': '🔄', 'color': '#795548', 'homebrew_service': True, 'priority': 2 } } async def discover_all_services(self) -> Dict[str, ServiceHealth]: """Comprehensive service discovery from all sources""" discovered_services = {} # Run discovery methods in parallel with ThreadPoolExecutor(max_workers=10) as executor: futures = { executor.submit(self._discover_docker_services): 'docker', executor.submit(self._discover_native_apps): 'native', executor.submit(self._discover_homebrew_services): 'homebrew', executor.submit(self._discover_network_services): 'network', executor.submit(self._discover_kubernetes_services): 'kubernetes' } for future in as_completed(futures): source = futures[future] try: services = future.result() discovered_services.update(services) logger.info(f"Discovered {len(services)} services from {source}") except Exception as e: logger.error(f"Error discovering {source} services: {e}") # Enrich services with health checks await self._enrich_services_health(discovered_services) return discovered_services def _discover_docker_services(self) -> Dict[str, ServiceHealth]: """Discover Docker/OrbStack containers""" services = {} try: result = subprocess.run(['docker', 'ps', '-a', '--format', 'table {{.Names}}\t{{.Image}}\t{{.Status}}\t{{.Ports}}'], capture_output=True, text=True, timeout=10) if result.returncode == 0: lines = result.stdout.strip().split('\n')[1:] # Skip header for line in lines: if not line.strip(): continue parts = line.split('\t') if len(parts) >= 3: name = parts[0].strip() image = parts[1].strip() status = parts[2].strip() ports = parts[3].strip() if len(parts) > 3 else "" # Skip k8s pause containers if 'pause' in image.lower() or 'POD_' in name: continue # Clean up container names clean_name = self._clean_container_name(name) service_def = self.service_definitions.get(clean_name, {}) # Extract port information extracted_ports = self._extract_ports_from_docker(ports) services[clean_name] = ServiceHealth( name=clean_name, display_name=service_def.get('display_name', clean_name.title()), category=service_def.get('category', '🐳 Containers'), status='healthy' if 'Up' in status else 'unhealthy' if 'Exited' in status else 'stopped', url=f"http://localhost:{extracted_ports[0]}" if extracted_ports else None, external_url=service_def.get('external_url'), # Store external URL separately port=extracted_ports[0] if extracted_ports else None, container_id=name, icon=service_def.get('icon', '🐳'), color=service_def.get('color', '#0db7ed'), priority=service_def.get('priority', 2), uptime=self._parse_uptime_from_status(status), last_check=datetime.now().isoformat() # Convert to string for JSON serialization ) except Exception as e: logger.error(f"Error discovering Docker services: {e}") return services def _discover_native_apps(self) -> Dict[str, ServiceHealth]: """Discover native macOS applications""" services = {} try: # Check for Jellyfin native app result = subprocess.run(['ps', 'aux'], capture_output=True, text=True, timeout=5) for line in result.stdout.split('\n'): if 'jellyfin' in line.lower() and '/Applications/Jellyfin.app' in line: # Extract process info parts = line.split() if len(parts) > 1: pid = parts[1] services['jellyfin'] = ServiceHealth( name='jellyfin', display_name='Jellyfin (Native)', category='🎥 Media Servers', status='healthy', url='http://localhost:8096', external_url=self.service_definitions.get('jellyfin', {}).get('external_url'), port=8096, process_id=int(pid), icon='🐙', color='#00a4dc', priority=1, last_check=datetime.now().isoformat() ) break except Exception as e: logger.error(f"Error discovering native apps: {e}") return services def _discover_homebrew_services(self) -> Dict[str, ServiceHealth]: """Discover Homebrew services""" services = {} try: result = subprocess.run(['brew', 'services', 'list'], capture_output=True, text=True, timeout=10) if result.returncode == 0: lines = result.stdout.strip().split('\n')[1:] # Skip header for line in lines: parts = line.split() if len(parts) >= 2: name = parts[0] status = parts[1] service_def = self.service_definitions.get(name, {}) # Determine status service_status = 'unknown' if status == 'started': service_status = 'healthy' elif status == 'none': service_status = 'stopped' elif 'error' in status: service_status = 'unhealthy' elif status == 'scheduled': service_status = 'warning' services[name] = ServiceHealth( name=name, display_name=service_def.get('display_name', name.title()), category=service_def.get('category', '🍺 Homebrew Services'), status=service_status, url=f"http://localhost:{service_def.get('ports', [None])[0]}" if service_def.get('ports') else None, external_url=service_def.get('external_url'), port=service_def.get('ports', [None])[0] if service_def.get('ports') else None, icon=service_def.get('icon', '🍺'), color=service_def.get('color', '#fbb040'), priority=service_def.get('priority', 2), last_check=datetime.now().isoformat() ) except Exception as e: logger.error(f"Error discovering Homebrew services: {e}") return services def _discover_network_services(self) -> Dict[str, ServiceHealth]: """Discover services by scanning network ports""" services = {} try: # Use lsof to find listening services result = subprocess.run(['lsof', '-i', '-P', '-n'], capture_output=True, text=True, timeout=10) port_services = {} for line in result.stdout.split('\n'): if 'LISTEN' in line: parts = line.split() if len(parts) >= 9: process = parts[0] pid = parts[1] address = parts[8] # Extract port if ':' in address: port_str = address.split(':')[-1] try: port = int(port_str) if port not in port_services: port_services[port] = { 'process': process, 'pid': pid } except ValueError: continue # Match ports to known services for service_name, service_def in self.service_definitions.items(): for port in service_def.get('ports', []): if port in port_services and service_name not in services: process_info = port_services[port] services[service_name] = ServiceHealth( name=service_name, display_name=service_def.get('display_name', service_name.title()), category=service_def.get('category', '🌐 Network Services'), status='healthy', url=f"http://localhost:{port}", external_url=service_def.get('external_url'), port=port, process_id=int(process_info['pid']), icon=service_def.get('icon', '🌐'), color=service_def.get('color', '#4caf50'), priority=service_def.get('priority', 2), last_check=datetime.now().isoformat() ) except Exception as e: logger.error(f"Error discovering network services: {e}") return services def _discover_kubernetes_services(self) -> Dict[str, ServiceHealth]: """Discover Kubernetes services (k3s pods, etc.)""" services = {} # K8s discovery would go here if needed return services async def _enrich_services_health(self, services: Dict[str, ServiceHealth]): """Enrich services with detailed health information""" async def check_service_health(service: ServiceHealth): """Check individual service health""" try: # API Health Check service_def = self.service_definitions.get(service.name, {}) api_path = service_def.get('api_path') external_url = service_def.get('external_url') # Always set localhost URL for local access if service.port: service.url = f"http://localhost:{service.port}" # Store external URL separately if available if external_url: service.external_url = external_url # Initialize to unhealthy if status is not already set if service.status not in ['healthy', 'unhealthy', 'warning', 'stopped']: service.status = 'unknown' # Check port availability first - this is most important for service status if service.port: # Check if port is reachable try: with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: sock.settimeout(1) # Short timeout for quick check result = sock.connect_ex(('localhost', service.port)) # Port is open if result is 0 if result == 0: # Service port is reachable, mark as likely healthy if service.status != 'unhealthy': # Don't override unhealthy from Docker/Homebrew service.status = 'healthy' else: # Port is unreachable, service is definitely down service.status = 'unhealthy' except Exception as port_err: logger.debug(f"Port check error for {service.name}: {port_err}") # Don't change status based just on a failed check # Try API health check if port is available and API path is defined if service.port and api_path and service.status != 'unhealthy': # Use localhost for API health checks url = f"http://localhost:{service.port}{api_path}" start_time = time.time() try: # Use a shorter timeout to prevent slow services from blocking response = requests.get(url, timeout=2) service.response_time = round((time.time() - start_time) * 1000, 2) service.api_health = response.status_code < 400 # Update service status based on API health if service.api_health: service.status = 'healthy' else: # API returned error code service.status = 'warning' # Try to extract version info only if the content is JSON if response.headers.get('content-type', '').startswith('application/json'): try: data = response.json() service.version = self._extract_version(data) except: pass except requests.RequestException: service.api_health = False # Don't mark as unhealthy just because API check failed # Service might still be starting up if service.status == 'healthy': service.status = 'warning' # Process metrics if service.process_id: try: process = psutil.Process(service.process_id) service.cpu_usage = round(process.cpu_percent(), 1) service.memory_usage = round(process.memory_percent(), 1) except psutil.NoSuchProcess: # Process no longer exists, service is definitely down service.status = 'unhealthy' service.last_check = datetime.now().isoformat() except Exception as e: logger.error(f"Error checking health for {service.name}: {e}") service.error_message = str(e) # Run health checks in parallel tasks = [check_service_health(service) for service in services.values()] if tasks: await asyncio.gather(*tasks, return_exceptions=True) def _clean_container_name(self, name: str) -> str: """Clean Docker container names to match service definitions""" # Remove k8s prefixes if name.startswith('k8s_'): parts = name.split('_') if len(parts) > 1: name = parts[1] # Remove common suffixes name = re.sub(r'[-_](server|app|service|container|1)$', '', name) return name.lower() def _extract_ports_from_docker(self, ports_str: str) -> List[int]: """Extract port numbers from Docker ports string""" ports = [] if ports_str: # Match patterns like "0.0.0.0:8080->8080/tcp" matches = re.findall(r'0\.0\.0\.0:(\d+)->', ports_str) ports = [int(match) for match in matches] return ports def _parse_uptime_from_status(self, status: str) -> Optional[str]: """Parse uptime from Docker status string""" if 'Up ' in status: uptime_part = status.split('Up ')[1].split(' (')[0] return uptime_part return None def _extract_version(self, data: dict) -> Optional[str]: """Extract version information from API response""" version_keys = ['version', 'Version', 'ver', 'build', 'Build'] for key in version_keys: if key in data: return str(data[key]) return None class SystemMetrics: """System resource monitoring with comprehensive storage tracking""" @staticmethod def get_system_info() -> dict: """Get comprehensive system information including all storage devices and GPU metrics""" # Get basic system info boot_time = datetime.fromtimestamp(psutil.boot_time()).isoformat() # Get GPU information (macOS specific) gpu_info = {} try: # Using system_profiler to get GPU information on macOS gpu_result = subprocess.run( ['system_profiler', 'SPDisplaysDataType', '-json'], capture_output=True, text=True, timeout=5 ) if gpu_result.returncode == 0: try: gpu_data = json.loads(gpu_result.stdout) # Extract GPU information from the SPDisplaysDataType section if 'SPDisplaysDataType' in gpu_data and gpu_data['SPDisplaysDataType']: for gpu in gpu_data['SPDisplaysDataType']: if 'sppci_model' in gpu: gpu_name = gpu.get('sppci_model', 'Unknown GPU') gpu_info['name'] = gpu_name gpu_info['vram'] = gpu.get('spdisplays_vram', 'Unknown') gpu_info['vendor'] = gpu.get('spdisplays_vendor', 'Unknown') gpu_info['cores'] = gpu.get('spdisplays_cores', 'Unknown') break except json.JSONDecodeError: logger.error("Failed to parse GPU JSON data") # Get GPU usage using top command top_result = subprocess.run( ['top', '-l', '1', '-n', '0', '-stats', 'pid,command,gpu'], capture_output=True, text=True, timeout=5 ) if top_result.returncode == 0: # Extract total GPU usage gpu_usage = 0.0 try: for line in top_result.stdout.split('\n'): if 'GPU' in line and '%' in line: parts = line.strip().split() for part in parts: if '%' in part and part != '0.0%': try: usage = float(part.replace('%', '')) gpu_usage += usage except ValueError: pass gpu_info['usage_percent'] = round(gpu_usage, 1) except Exception as e: logger.error(f"Failed to parse GPU usage: {e}") gpu_info['usage_percent'] = 0.0 except Exception as e: logger.error(f"Error getting GPU info: {e}") gpu_info = {"name": "Unknown", "usage_percent": 0.0} # Get comprehensive disk usage for all mounted volumes disk_usage = {} # Define important volumes to monitor important_volumes = { '/': 'System (macOS)', '/System/Volumes/Data': 'User Data', '/Volumes/SamMgmt': 'SamMgmt Drive', '/Volumes/Games': 'Games Drive', '/Volumes/Sam Storage': 'Sam Storage (4TB)', '/Volumes/Sam [8TB]': 'Sam 8TB Drive', '/Volumes/Jellyfin Media': 'Jellyfin Media (8TB)', '/Volumes/Prod Media': 'Production Media (8TB)', '/Volumes/Server Media': 'Server Media (8TB)', '/Volumes/Sam AirPort': 'Sam AirPort (Network)', '/Users/sammacmini/OrbStack': 'OrbStack Container Storage' } # Get disk usage for all partitions for partition in psutil.disk_partitions(): try: usage = psutil.disk_usage(partition.mountpoint) # Get friendly name or use mountpoint display_name = important_volumes.get(partition.mountpoint, partition.mountpoint) # Calculate percentages and format sizes percent_used = round((usage.used / usage.total) * 100, 1) disk_usage[partition.mountpoint] = { 'display_name': display_name, 'device': partition.device, 'filesystem': partition.fstype, 'total': usage.total, 'used': usage.used, 'free': usage.free, 'percent': percent_used, 'total_gb': round(usage.total / (1024**3), 1), 'used_gb': round(usage.used / (1024**3), 1), 'free_gb': round(usage.free / (1024**3), 1), 'total_tb': round(usage.total / (1024**4), 2), 'used_tb': round(usage.used / (1024**4), 2), 'free_tb': round(usage.free / (1024**4), 2) } except (PermissionError, OSError): # Skip volumes we can't access continue # Get network stats net_io = psutil.net_io_counters() network_stats = { 'bytes_sent': net_io.bytes_sent, 'bytes_recv': net_io.bytes_recv, 'packets_sent': net_io.packets_sent, 'packets_recv': net_io.packets_recv, 'bytes_sent_gb': round(net_io.bytes_sent / (1024**3), 2), 'bytes_recv_gb': round(net_io.bytes_recv / (1024**3), 2) } # Calculate total storage summary total_storage: Dict[str, Union[float, int]] = { 'total_capacity_tb': 0.0, 'total_used_tb': 0.0, 'total_free_tb': 0.0, 'average_usage_percent': 0.0, 'drive_count': 0 } # Sum up only the main storage drives (exclude system volumes) main_storage_volumes = [ '/Volumes/SamMgmt', '/Volumes/Games', '/Volumes/Sam Storage', '/Volumes/Sam [8TB]', '/Volumes/Jellyfin Media', '/Volumes/Prod Media', '/Volumes/Server Media' ] usage_percentages = [] for volume_path in main_storage_volumes: if volume_path in disk_usage: volume_data = disk_usage[volume_path] total_storage['total_capacity_tb'] += volume_data['total_tb'] total_storage['total_used_tb'] += volume_data['used_tb'] total_storage['total_free_tb'] += volume_data['free_tb'] usage_percentages.append(volume_data['percent']) total_storage['drive_count'] += 1 if usage_percentages: total_storage['average_usage_percent'] = round(sum(usage_percentages) / len(usage_percentages), 1) # Round the totals total_storage['total_capacity_tb'] = round(total_storage['total_capacity_tb'], 1) total_storage['total_used_tb'] = round(total_storage['total_used_tb'], 1) total_storage['total_free_tb'] = round(total_storage['total_free_tb'], 1) return { 'hostname': socket.gethostname(), 'platform': platform.system(), 'platform_version': platform.release(), 'architecture': platform.machine(), 'cpu_count': psutil.cpu_count(), 'cpu_percent': psutil.cpu_percent(interval=1), 'memory_total': psutil.virtual_memory().total, 'memory_used': psutil.virtual_memory().used, 'memory_percent': psutil.virtual_memory().percent, 'memory_total_gb': round(psutil.virtual_memory().total / (1024**3), 1), 'memory_used_gb': round(psutil.virtual_memory().used / (1024**3), 1), 'disk_usage': disk_usage, 'total_storage': total_storage, 'network_stats': network_stats, 'boot_time': boot_time, 'uptime_hours': round((datetime.now() - datetime.fromtimestamp(psutil.boot_time())).total_seconds() / 3600, 1), 'load_average': os.getloadavg() if hasattr(os, 'getloadavg') else [0, 0, 0], 'gpu_info': gpu_info } # Global service discovery instance service_discovery = ServiceDiscovery() system_metrics = SystemMetrics() @app.route('/') def index(): """Main dashboard page""" return render_template_string(DASHBOARD_TEMPLATE) @app.route('/api/services') def get_services(): """Get all discovered services""" loop = None try: loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) services = loop.run_until_complete(service_discovery.discover_all_services()) return jsonify({ 'services': {name: asdict(service) for name, service in services.items()}, 'last_updated': datetime.now().isoformat(), 'total_services': len(services) }) finally: if loop: loop.close() @app.route('/api/system') def get_system_metrics(): """Get system metrics""" return jsonify({ 'system': system_metrics.get_system_info(), 'timestamp': datetime.now().isoformat() }) @app.route('/api/service//restart', methods=['POST']) def restart_service(service_name): """Restart a service""" try: service = None # Find the service to get its type loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) services = loop.run_until_complete(service_discovery.discover_all_services()) loop.close() if service_name in services: service = services[service_name] else: return jsonify({'success': False, 'error': f'Service {service_name} not found'}), 404 # Determine service type and restart appropriately if service.container_id: # Docker container result = subprocess.run(['docker', 'restart', service.container_id], capture_output=True, text=True) if result.returncode != 0: return jsonify({'success': False, 'error': f'Failed to restart Docker container: {result.stderr}'}), 500 elif service_name in service_discovery.service_definitions and service_discovery.service_definitions[service_name].get('homebrew_service'): # Homebrew service result = subprocess.run(['brew', 'services', 'restart', service_name], capture_output=True, text=True) if result.returncode != 0: return jsonify({'success': False, 'error': f'Failed to restart Homebrew service: {result.stderr}'}), 500 elif service.process_id: # Native process - try graceful restart if possible try: process = psutil.Process(service.process_id) process.terminate() # Allow time for graceful shutdown gone, alive = psutil.wait_procs([process], timeout=3) if alive: process.kill() # For native processes, we don't auto-restart - return success with a note return jsonify({'success': True, 'message': f'Service {service_name} stopped. It may need to be manually restarted.'}) except psutil.NoSuchProcess: return jsonify({'success': False, 'error': 'Process not found, may already be stopped'}), 400 else: return jsonify({'success': False, 'error': 'Unable to determine how to restart this service'}), 400 return jsonify({'success': True, 'message': f'Service {service_name} restart initiated'}) except Exception as e: logger.error(f"Error restarting service {service_name}: {e}") return jsonify({'success': False, 'error': str(e)}), 500 @socketio.on('connect') def handle_connect(): """Handle WebSocket connection""" logger.info(f'🟢 Client connected via WebSocket: {request.sid}') emit('status', {'message': 'Connected to SamCloud Dashboard'}) @socketio.on('disconnect') def handle_disconnect(): """Handle WebSocket disconnection""" logger.info(f'🔴 Client disconnected: {request.sid}') @socketio.on('request_update') def handle_update_request(): """Handle real-time update requests""" logger.info(f'📊 Update request from client: {request.sid}') loop = None try: loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) services = loop.run_until_complete(service_discovery.discover_all_services()) system = system_metrics.get_system_info() emit('services_update', { 'services': {name: asdict(service) for name, service in services.items()}, 'system': system, 'timestamp': datetime.now().isoformat() }) logger.info(f'📤 Sent services update to client: {request.sid}') except Exception as e: logger.error(f"Error handling update request: {e}") emit('error', {'message': str(e)}) finally: if loop: loop.close() @socketio.on('request_metrics_update') def handle_metrics_update_request(): """Handle lightweight metrics update requests (CPU, RAM, GPU only)""" logger.debug(f'📡 Metrics update request from client: {request.sid}') try: # Only collect essential metrics without expensive operations gpu_usage = 0.0 try: # Get GPU usage using top command top_result = subprocess.run( ['top', '-l', '1', '-n', '0', '-stats', 'pid,command,gpu'], capture_output=True, text=True, timeout=3 ) if top_result.returncode == 0: for line in top_result.stdout.split('\n'): if 'GPU' in line and '%' in line: parts = line.strip().split() for part in parts: if '%' in part and part != '0.0%': try: usage = float(part.replace('%', '')) gpu_usage += usage except ValueError: pass except Exception as e: logger.error(f"Error getting GPU usage: {e}") system = { 'cpu_percent': psutil.cpu_percent(interval=0.1), 'memory_percent': psutil.virtual_memory().percent, 'uptime_hours': round((datetime.now() - datetime.fromtimestamp(psutil.boot_time())).total_seconds() / 3600, 1), 'gpu_info': { 'usage_percent': round(gpu_usage, 1) } } emit('metrics_update', { 'system': system, 'timestamp': datetime.now().isoformat() }) except Exception as e: logger.error(f"Error handling metrics update request: {e}") emit('error', {'message': str(e)}) def background_metrics_collector(): """Background thread for continuous metrics collection""" last_full_update = datetime.now() while True: loop = None try: current_time = datetime.now() time_since_full = (current_time - last_full_update).total_seconds() # Every 30 seconds, do a full update of all services and system if time_since_full >= 30: # Emit full updates to all connected clients loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) services = loop.run_until_complete(service_discovery.discover_all_services()) system = system_metrics.get_system_info() socketio.emit('live_update', { 'services': {name: asdict(service) for name, service in services.items()}, 'system': system, 'timestamp': datetime.now().isoformat() }) last_full_update = current_time else: # Every 5 seconds, just check basic service status without heavy API checks # This focuses on checking if services are running/stopped loop = asyncio.new_event_loop() asyncio.set_event_loop(loop) # Get minimal service status updates services_status = {} for service_name, service_def in service_discovery.service_definitions.items(): # Only check services with ports if service_def.get('ports'): port = service_def['ports'][0] try: with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock: sock.settimeout(0.5) # Very short timeout result = sock.connect_ex(('localhost', port)) status = 'healthy' if result == 0 else 'unhealthy' services_status[service_name] = {'status': status} except: # Skip on error pass # Get essential system metrics system = { 'cpu_percent': psutil.cpu_percent(interval=0.1), 'memory_percent': psutil.virtual_memory().percent, 'gpu_info': {'usage_percent': 0.0} # Initialize with default } # Get GPU usage try: top_result = subprocess.run( ['top', '-l', '1', '-n', '0', '-stats', 'gpu'], capture_output=True, text=True, timeout=2 ) if top_result.returncode == 0: gpu_usage = 0.0 for line in top_result.stdout.split('\n'): if 'GPU' in line and '%' in line: parts = line.strip().split() for part in parts: if '%' in part and part != '0.0%': try: usage = float(part.replace('%', '')) gpu_usage += usage except ValueError: pass system['gpu_info']['usage_percent'] = round(gpu_usage, 1) except Exception as e: logger.debug(f"GPU metrics error: {e}") # Emit lightweight update with just status changes and system metrics if services_status: socketio.emit('status_update', { 'services': services_status, 'system': system, 'timestamp': datetime.now().isoformat() }) except Exception as e: logger.error(f"Background metrics error: {e}") finally: if loop: loop.close() time.sleep(5) # Update every 5 seconds # Start background thread metrics_thread = threading.Thread(target=background_metrics_collector, daemon=True) metrics_thread.start() # Apple HIG 2025 compliant dashboard template DASHBOARD_TEMPLATE = ''' SamCloud Advanced Dashboard
🔴 Disconnected
--%
CPU
--%
GPU
--%
Memory
--%
Storage
--
Services
--
Updated
🔄 Initializing dashboard...
''' if __name__ == '__main__': logger.info("Starting SamCloud Advanced Home Lab Dashboard") logger.info("Dashboard will be available at: http://localhost:5001") try: socketio.run(app, host='0.0.0.0', port=5001, debug=False, allow_unsafe_werkzeug=True) except KeyboardInterrupt: logger.info("Dashboard stopped by user") except Exception as e: logger.error(f"Dashboard startup error: {e}")