""" Startup Validator Checks configuration, the cache directory, plugins and the installed systemd units when the display service starts, and reports what it finds. validate_all() never raises: it returns (is_valid, errors, warnings) and DisplayController logs them. Startup continues either way, so a problem found here shows up in the log rather than stopping the display. raise_on_errors() turns the errors into exceptions for a caller that does want to stop; the display service does not call it. """ import os from typing import Any, List, Optional, Tuple from pathlib import Path from src.core_config_keys import CORE_CONFIG_KEYS from src.exceptions import ConfigError, PluginError, CacheError from src.logging_config import get_logger class StartupValidator: """Validates system state on startup.""" def __init__(self, config_manager: Any, plugin_manager: Optional[Any] = None, cache_manager: Optional[Any] = None) -> None: """ Initialize the startup validator. Args: config_manager: ConfigManager instance plugin_manager: Optional PluginManager instance cache_manager: The CacheManager the application will actually use. Pass it. Without one this validator builds its own just to read a directory path, which reports on a cache the app does not use and leaves behind a cleanup thread that nothing stops -- validation runs twice per startup, so that was two of them. """ self.config_manager = config_manager self.plugin_manager = plugin_manager self.cache_manager = cache_manager self.logger = get_logger(__name__) self.errors: List[str] = [] self.warnings: List[str] = [] def validate_all(self) -> Tuple[bool, List[str], List[str]]: """ Run all validation checks. Returns: Tuple of (is_valid, errors, warnings) """ self.logger.info("Starting startup validation...") # Fresh lists each run — without this, calling validate_all() twice # duplicated every message. self.errors = [] self.warnings = [] # Validate configuration self._validate_config() # Validate cache directory self._validate_cache_directory() # Validate display configuration self._validate_display_config() # Validate plugins if plugin manager is available if self.plugin_manager: self._validate_plugins() # Warn when the running systemd unit no longer matches the repo's self._validate_systemd_units() is_valid = len(self.errors) == 0 if is_valid: self.logger.info("Startup validation passed") if self.warnings: self.logger.warning(f"Startup validation completed with {len(self.warnings)} warning(s)") else: self.logger.error(f"Startup validation failed with {len(self.errors)} error(s)") return (is_valid, self.errors.copy(), self.warnings.copy()) #: Units this project installs, and where each is installed to. _UNITS = ( ("systemd/ledmatrix.service", "/etc/systemd/system/ledmatrix.service"), ("systemd/ledmatrix-web.service", "/etc/systemd/system/ledmatrix-web.service"), ("systemd/ledmatrix-update-verify.service", "/etc/systemd/system/ledmatrix-update-verify.service"), ("systemd/ledmatrix-update-verify.path", "/etc/systemd/system/ledmatrix-update-verify.path"), ) def _validate_systemd_units(self) -> None: """Warn when an installed unit has drifted from the repo's template. Before updates refreshed units, nothing re-applied these after the first install: `git pull` brought a new template into the checkout, but nothing copied it to /etc/systemd/system, so the unit that actually ran was whatever first_time_install.sh wrote on day one. Updates now install changed units through the root helper ledmatrix-refresh-units (web_interface/unit_refresh.py) -- but only on a device whose installer granted it, so this still catches the rest. That makes every hardening added to a unit inert on existing installs. Measured on one rig: the installed unit was thirteen days older than the repo's and differed in content, so a MemoryMax the repo had specified was not being enforced at all -- `systemctl show` reported MemoryMax=infinity. A warning rather than an error, and certainly not a silent rewrite: editing files under /etc and restarting services is the installer's job, not something a display process should do to a machine while it boots. The remedy is to re-run scripts/install/install_service.sh. """ try: project_root = Path(__file__).resolve().parent.parent for template_rel, installed_path in self._UNITS: template = project_root / template_rel installed = Path(installed_path) if not template.is_file() or not installed.is_file(): continue try: actual = installed.read_text(encoding="utf-8") except PermissionError: continue # The template carries placeholders the installer substitutes, # so compare the substituted form rather than the raw file. expected = template.read_text(encoding="utf-8") expected = expected.replace("__PROJECT_ROOT_DIR__", str(project_root)) # User= is an install-time decision, not something the template # dictates: the installers write whoever ran them, which on a # non-root install is never "root". Substituting a fixed "root" # here reported drift on every such install, permanently -- and # re-running the installer, which is what the warning tells you # to do, could not clear it. Taking the installed unit's own # value keeps the comparison on the directives the template # actually controls. expected = expected.replace("__USER__", self._installed_user(actual)) if self._unit_body(expected) != self._unit_body(actual): self.warnings.append( f"{installed.name} differs from {template_rel}, so " "settings added to the template are not in effect. " "Updates apply them only once the installer has granted " "ledmatrix-refresh-units: re-run " "scripts/install/install_service.sh (or first_time_install.sh) to apply them." ) except OSError as e: self.logger.debug("Could not compare systemd units: %s", e) @staticmethod def _installed_user(unit_text: str) -> str: """The installed unit's ``User=``, or "root" when it does not set one. systemd itself defaults to root for a system unit with no User=, so that is the right fallback rather than an empty string. """ for line in unit_text.splitlines(): stripped = line.strip() if stripped.startswith("User="): return stripped.split("=", 1)[1].strip() return "root" @staticmethod def _unit_body(text: str) -> str: """A unit's meaningful lines, in order: no comments, no blanks. Order is preserved deliberately. This used to sort, which made the comparison insensitive to two changes that matter in a systemd unit: repeated directives such as ExecStartPre= and ExecStartPost= run in the order they appear, and a directive that moves between [Unit], [Service] and [Install] means something different -- or nothing -- where it lands. A drift check that normalises those away reports no drift for a unit that has genuinely changed. """ lines = [] for line in text.splitlines(): line = line.strip() if line and not line.startswith("#"): lines.append(line) return "\n".join(lines) def _validate_config(self) -> None: """Validate configuration files.""" try: config = self.config_manager.load_config() required_keys = ['display', 'timezone'] for key in required_keys: if key not in config: self.errors.append(f"Missing required configuration key: {key}") # A missing display section is reported once, above, and an empty # one here; _validate_display_config leaves both to this method. if 'display' in config and not config['display']: self.errors.append("Display configuration is empty") except ConfigError as e: self.errors.append(f"Configuration error: {e}") except Exception as e: self.errors.append(f"Unexpected error validating configuration: {e}") def _validate_cache_directory(self) -> None: """Validate cache directory permissions.""" try: cache_manager = self.cache_manager if cache_manager is None: # No caller supplied one (older embedders, direct use in a # script). Build one, but do not leave its cleanup thread # running behind us -- this instance is discarded on the next # line but the thread is a closure over it, so it would never # be collected. from src.cache_manager import CacheManager cache_manager = CacheManager() try: cache_dir = cache_manager.get_cache_dir() finally: cache_manager.stop_cleanup_thread() else: cache_dir = cache_manager.get_cache_dir() if not cache_dir: self.warnings.append("Cache directory not available - caching will be disabled") return # Check if directory exists and is writable if not os.path.exists(cache_dir): self.errors.append(f"Cache directory does not exist: {cache_dir}") return if not os.access(cache_dir, os.W_OK): self.errors.append(f"Cache directory is not writable: {cache_dir}") return # Test write access test_file = os.path.join(cache_dir, '.startup_test') try: with open(test_file, 'w') as f: f.write('test') os.remove(test_file) except (IOError, OSError) as e: self.errors.append(f"Cannot write to cache directory {cache_dir}: {e}") except Exception as e: self.warnings.append(f"Could not validate cache directory: {e}") def _validate_display_config(self) -> None: """Validate display configuration.""" try: config = self.config_manager.get_config() display_config = config.get('display', {}) if not display_config: return # reported by _validate_config hardware_config = display_config.get('hardware', {}) if not hardware_config: self.errors.append("Display hardware configuration is missing") return # Check required hardware settings required_hardware = ['rows', 'cols'] for key in required_hardware: if key not in hardware_config: self.warnings.append(f"Display hardware setting '{key}' not specified, using default") except Exception as e: self.warnings.append(f"Could not validate display configuration: {e}") def _validate_plugins(self, discovered_plugins=None) -> None: """Validate plugin configurations and dependencies. ``discovered_plugins`` is a list the caller already got from ``discover_plugins()``; passing it skips a second directory scan (and its duplicate log lines) at startup. """ if not self.plugin_manager: return try: # Get enabled plugins from config config = self.config_manager.get_config() if discovered_plugins is None: discovered_plugins = self.plugin_manager.discover_plugins() # Check for enabled plugins that don't exist for plugin_id, plugin_config in config.items(): # Skip core sections: auto_update and dim_schedule have an # 'enabled' key too, and are not plugins that went missing. if plugin_id in CORE_CONFIG_KEYS: continue if not isinstance(plugin_config, dict): continue if plugin_config.get('enabled', False): if plugin_id not in discovered_plugins: self.warnings.append(f"Plugin '{plugin_id}' is enabled but not found in plugins directory") # Validate plugin configurations for plugin_id in discovered_plugins: plugin_config = config.get(plugin_id) # A null block ("my-plugin": null) is not an enabled plugin; # .get() on it raised and abandoned every remaining check. if not isinstance(plugin_config, dict): continue if plugin_config.get('enabled', False): # Check if plugin can be loaded (without actually loading it) plugin_dir = self.plugin_manager.get_plugin_directory(plugin_id) if plugin_dir: manifest_path = Path(plugin_dir) / "manifest.json" if not manifest_path.exists(): self.errors.append(f"Plugin '{plugin_id}' manifest.json not found") except Exception as e: self.warnings.append(f"Could not validate plugins: {e}") def raise_on_errors(self) -> None: """ Raise one exception if validation errors exist; return None if not. Nothing in core calls this (see the module docstring). Errors are grouped by a keyword in their message, not by which check produced them, and only the first non-empty group is raised, in the order config > cache > plugin: a "plugin ... config" message counts as a config error, and cache/plugin errors are not reported while a config error exists. The raised exception's ``context['errors']`` holds that group's messages only. Raises: ConfigError: If any message mentions config/configuration, or if none matches any group CacheError: If a message mentions cache (and none config) PluginError: If a message mentions plugin (and none of the above) """ if not self.errors: return # Group errors by type config_errors = [e for e in self.errors if 'configuration' in e.lower() or 'config' in e.lower()] cache_errors = [e for e in self.errors if 'cache' in e.lower()] plugin_errors = [e for e in self.errors if 'plugin' in e.lower()] other_errors = [e for e in self.errors if e not in config_errors + cache_errors + plugin_errors] # Raise appropriate exceptions if config_errors: raise ConfigError("Configuration validation failed", context={'errors': config_errors}) if cache_errors: raise CacheError("Cache validation failed", context={'errors': cache_errors}) if plugin_errors: raise PluginError("Plugin validation failed", context={'errors': plugin_errors}) if other_errors: raise ConfigError("Startup validation failed", context={'errors': other_errors})