Files
LEDMatrix/web_interface/blueprints/api_v3/misc.py
T
ChuckandClaude Opus 5.5 54fb42180f fix(web): schedule saves, restart_required, health and current-status agree with the rig
- POST /config/schedule and /config/dim-schedule accept a disabled per-day
  schedule with every day off (config.template.json's own shape); a day
  that is off keeps its posted times.
- POST /config/main sets restart_required from what the save changed:
  brightness, mode durations and plugin sections are applied live.
- /health is degraded (display_loop: stopped) when the service is inactive,
  the socket does not answer and there is no live heartbeat.
- /display/current-status no longer serves a stopped display's cached state
  when the socket and heartbeat both say it is gone (display_gone).

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
2026-10-03 22:11:32 -04:00

621 lines
27 KiB
Python

"""Routes with no larger group of their own: errors, integrations,
cache, sync, logs, health and hardware.
Routes decorate the shared `api_v3` Blueprint from the package `__init__`,
so their endpoint names are unchanged by living here.
"""
from web_interface.blueprints.api_v3 import (
_coerce_to_bool, _discovered_plugin_manifests,
ErrorCode, _JOURNALCTL, _MQTT_BRIDGE_CONFIG, _MQTT_BRIDGE_DEFAULTS,
_MQTT_BRIDGE_DIR, _SUDO, _coerce_mqtt_bridge_value,
_get_display_service_status, _mqtt_bridge_service_state,
_read_mqtt_bridge_config, api_v3, contextlib, describe_exception,
error_response, json, jsonify, logger, os, redact_text, request,
subprocess, success_response, tempfile,
)
from src.common.path_safety import safe_path_component
from src import display_watchdog
from src.common import sync_manager as _sync
from src import error_aggregator as _errors
from web_interface import display_preview, display_state
from web_interface.auth import request_is_authenticated
import web_interface.blueprints.api_v3 as _pkg
# Read through the module rather than bound by value: tests patch these
# as module attributes, and a value binding would not see the patch.
# Several are also called from helpers that live in __init__, so the
# package is the only patch point that covers every caller.
@api_v3.route('/health', methods=['GET'])
def get_health():
"""Get system health status"""
try:
health_status = {
'status': 'healthy',
'timestamp': _pkg.time.time(),
'services': {},
'checks': {}
}
# Stamp the start time before measuring against it: reading it with a
# fallback of time.time() and assigning it afterwards made the first
# call subtract two separate clock reads, a small negative uptime.
if not hasattr(get_health, '_start_time'):
get_health._start_time = _pkg.time.time()
health_status['services']['web_interface'] = {
'status': 'running',
'uptime_seconds': _pkg.time.time() - get_health._start_time
}
# Check display service
display_service_status = _get_display_service_status()
health_status['services']['display_service'] = {
'status': 'active' if display_service_status.get('active') else 'inactive',
'details': display_service_status
}
# Check config file accessibility
try:
if api_v3.config_manager:
api_v3.config_manager.load_config()
health_status['checks']['config_file'] = {
'status': 'accessible',
'readable': True
}
else:
health_status['checks']['config_file'] = {
'status': 'unknown',
'readable': False
}
except Exception:
logger.warning("Health check could not read the config file", exc_info=True)
health_status['checks']['config_file'] = {
'status': 'error',
'readable': False,
'error': 'see logs for details'
}
# Check plugin system
try:
if api_v3.plugin_catalog:
plugin_count = len(_discovered_plugin_manifests())
health_status['checks']['plugin_system'] = {
'status': 'operational',
'plugin_count': plugin_count
}
else:
health_status['checks']['plugin_system'] = {
'status': 'not_initialized'
}
except Exception:
logger.warning("Health check could not count plugins", exc_info=True)
health_status['checks']['plugin_system'] = {
'status': 'error',
'error': 'see logs for details'
}
# Is the render loop still going round? The display rewrites this
# heartbeat every few seconds from the render thread itself, so a
# thread stuck inside a plugin lets it go stale even though the
# service is "active". No heartbeat at all (the dev server, an older
# display) is not a failure: the preview-frame check below is then
# the only signal, as it always was.
# The display reports the same beat's age over the control socket's
# state stream, measured in memory; the file is the fallback.
snapshot = None
try:
snapshot = display_state.read_state()
if snapshot is not None:
source = 'socket'
age = display_state.loop_heartbeat_age(snapshot)
else:
source = 'heartbeat_file'
heartbeat = display_watchdog.read_heartbeat(display_watchdog.HEARTBEAT_PATH)
age = display_watchdog.heartbeat_age(heartbeat) if heartbeat else None
if age is None:
health_status['checks']['display_loop'] = {
'status': 'not_reported',
'note': 'The display is not writing a heartbeat (not started yet, '
'or a version or setup without one)',
'source': source,
}
else:
fresh = age < display_watchdog.HEARTBEAT_STALE_SECONDS
health_status['checks']['display_loop'] = {
'status': 'running' if fresh else 'stalled',
'heartbeat_age_seconds': round(age, 1),
'source': source,
}
except Exception:
logger.warning("Health check could not read the display heartbeat", exc_info=True)
health_status['checks']['display_loop'] = {
'status': 'unknown',
'error': 'see logs for details'
}
# A stopped display service. The heartbeat's absence alone says
# nothing (the dev server, the emulator and Windows write none), so
# the overall status stayed "healthy" with the display down. Together
# the three signals are definite: systemd says the service is not
# active, the control socket does not answer, and there is no live
# heartbeat (display_state.display_gone, which is never true where
# the platform has no socket or it is switched off).
try:
if (not display_service_status.get('active')
and display_state.display_gone(snapshot)):
health_status['checks']['display_loop'] = {
'status': 'stopped',
'note': 'The display service is not running',
'source': 'service',
}
except Exception:
logger.warning("Health check could not tell whether the display is stopped",
exc_info=True)
# Check hardware connectivity (if display manager available)
try:
snapshot_path = display_preview.SNAPSHOT_PATH
if os.path.exists(snapshot_path):
# Check if snapshot is recent (updated in last 60 seconds)
mtime = os.path.getmtime(snapshot_path)
age_seconds = _pkg.time.time() - mtime
health_status['checks']['hardware'] = {
'status': 'connected' if age_seconds < 60 else 'stale',
'snapshot_age_seconds': round(age_seconds, 1)
}
else:
health_status['checks']['hardware'] = {
'status': 'no_snapshot',
'note': 'Display service may not be running'
}
except Exception:
logger.warning("Health check could not read the preview snapshot", exc_info=True)
health_status['checks']['hardware'] = {
'status': 'unknown',
'error': 'see logs for details'
}
# Determine overall health
# 'not_reported' is the absence of a signal, not a bad one.
all_healthy = all(
check.get('status') in ['accessible', 'operational', 'connected', 'running', 'active',
'not_reported']
for check in health_status['checks'].values()
)
if not all_healthy:
health_status['status'] = 'degraded'
if not request_is_authenticated():
# Web login is on and this caller has not logged in: the route
# stays open for uptime monitors, but says only up or degraded.
return jsonify({'status': 'success',
'data': {'status': health_status['status']}})
return jsonify({'status': 'success', 'data': health_status})
except Exception as e:
logger.error("%s failed", request.path, exc_info=True)
return jsonify({
'status': 'error',
'message': 'An error occurred; see logs for details',
'details': describe_exception(e),
'data': {'status': 'unhealthy'}
}), 500
@api_v3.route('/hardware/status', methods=['GET'])
def get_hardware_status():
"""Return LED matrix hardware initialization status written by display_manager at startup."""
status_path = "/tmp/led_matrix_hw_status.json" # nosec B108
try:
with open(status_path) as f:
hw_data = json.load(f)
return jsonify({"status": "success", "data": hw_data})
except FileNotFoundError:
return jsonify({"status": "success", "data": {"ok": None, "error": "Display service not yet started"}})
except PermissionError:
logger.warning("Permission denied reading hardware status file; display service may be running as a different user")
return jsonify({"status": "success", "data": {"ok": False, "error": "Hardware status temporarily unavailable"}})
except json.JSONDecodeError:
logger.error("Failed to parse hardware status file", exc_info=True)
return jsonify({"status": "success", "data": {"ok": False, "error": "Hardware status file corrupted"}})
except Exception:
logger.error("Unexpected error reading hardware status", exc_info=True)
return jsonify({"status": "error", "message": "Unable to read hardware status"}), 500
@api_v3.route('/logs', methods=['GET'])
def get_logs():
"""Get system logs from journalctl"""
try:
if not _JOURNALCTL:
return jsonify({'status': 'error', 'message': 'journalctl not found on this system'}), 503
# Get recent logs from journalctl
_cmd = ([_SUDO, _JOURNALCTL] if _SUDO else [_JOURNALCTL]) + [
'-u', 'ledmatrix.service', '-u', 'ledmatrix-web.service',
'-n', '100', '--no-pager', '--output=short-iso']
result = subprocess.run(
_cmd,
capture_output=True,
text=True,
timeout=5
)
if result.returncode == 0:
logs_text = result.stdout.strip()
return jsonify({
'status': 'success',
'data': {
'logs': logs_text if logs_text else 'No logs available from ledmatrix or ledmatrix-web service'
}
})
else:
return jsonify({
'status': 'error',
'message': f'Failed to get logs: {result.stderr}'
}), 500
except subprocess.TimeoutExpired:
return jsonify({
'status': 'error',
'message': 'Timeout while fetching logs'
}), 500
# Multi-Display Sync Endpoints
@api_v3.route('/sync/status', methods=['GET'])
def get_sync_status():
"""Return live multi-display sync status written by the display process."""
# The display process writes this file; read it where it is written.
status_file = _sync.STATUS_FILE
# Also surface config so the UI can show the configured role even before
# the display process has written a status file.
cfg_role = "standalone"
cfg_port = _sync.SYNC_PORT
if api_v3.config_manager:
try:
cfg = api_v3.config_manager.load_config().get("sync", {})
cfg_role = cfg.get("role", "standalone")
cfg_port = int(cfg.get("port", _sync.SYNC_PORT))
except Exception:
pass
if os.path.exists(status_file):
try:
with open(status_file) as f:
live = json.load(f)
return jsonify({"status": "success", "data": live})
except Exception:
pass
# Status file not yet written — return config-only placeholder
return jsonify({
"status": "success",
"data": {
"role": cfg_role,
"port": cfg_port,
"state": "starting",
}
})
@api_v3.route('/cache/list', methods=['GET'])
def list_cache_files():
"""List all cache files with metadata"""
if not api_v3.cache_manager:
# Initialize cache manager if not already initialized
from src.cache_manager import CacheManager
api_v3.cache_manager = CacheManager()
cache_files = api_v3.cache_manager.list_cache_files()
cache_dir = api_v3.cache_manager.get_cache_dir()
return jsonify({
'status': 'success',
'data': {
'cache_files': cache_files,
'cache_dir': cache_dir,
'total_files': len(cache_files)
}
})
@api_v3.route('/cache/delete', methods=['POST'])
def delete_cache_file():
"""Delete a specific cache file by key"""
if not api_v3.cache_manager:
# Initialize cache manager if not already initialized
from src.cache_manager import CacheManager
api_v3.cache_manager = CacheManager()
data = request.get_json(silent=True)
if not data or 'key' not in data:
return jsonify({'status': 'error', 'message': 'cache key is required'}), 400
cache_key = data['key']
# The key names the file about to be removed. DiskCache refuses an
# unusable key on its own, but silently: say so here instead of
# reporting a deletion that never happened.
if safe_path_component(cache_key) is None:
return jsonify({'status': 'error', 'message': 'Invalid cache key'}), 400
# Delete the cache file
api_v3.cache_manager.clear_cache(cache_key)
return jsonify({
'status': 'success',
'message': f'Cache file for key "{cache_key}" deleted successfully'
})
def _errors_cache():
"""The shared cache the display service publishes its errors to."""
if not api_v3.cache_manager:
from src.cache_manager import CacheManager
api_v3.cache_manager = CacheManager()
return api_v3.cache_manager
def _redact_error_text(text, keep_lines=False):
"""Credentials out of plugin exception text, which can quote a URL with
an API key in it. Stack traces keep their line breaks and indentation."""
if not isinstance(text, str):
return text
if not keep_lines:
return redact_text(text, max_length=len(text) + 1)
return '\n'.join(
line[:len(line) - len(line.lstrip())] + redact_text(line, max_length=len(line) + 1)
for line in text.splitlines()
)
def _redact_error_record(record):
if not isinstance(record, dict):
return record
record = dict(record)
record['message'] = _redact_error_text(record.get('message'))
record['stack_trace'] = _redact_error_text(record.get('stack_trace'), keep_lines=True)
if isinstance(record.get('context'), dict):
record['context'] = {k: _redact_error_text(v) for k, v in record['context'].items()}
return record
def _read_errors():
snapshot, clear_request = _errors.read_error_report(_errors_cache())
return snapshot, clear_request
@api_v3.route('/errors/summary', methods=['GET'])
def get_error_summary():
"""
Get summary of all errors for monitoring and debugging.
Returns error counts, detected patterns, and recent errors, as last
reported by the display service (which runs the plugins, so it is the
only process that records their errors). ``snapshot_available`` is false
until it has reported; ``generated_at`` says when it did.
"""
try:
summary = _errors.error_summary_from_report(*_read_errors())
summary['recent_errors'] = [_redact_error_record(r) for r in summary['recent_errors']]
for pattern in summary['active_patterns'].values():
if isinstance(pattern, dict) and isinstance(pattern.get('sample_messages'), list):
pattern['sample_messages'] = [_redact_error_text(m) for m in pattern['sample_messages']]
message = ("Error summary retrieved" if summary['snapshot_available']
else "The display service has not reported any errors yet")
return success_response(data=summary, message=message)
except Exception as e:
logger.error(f"Error getting error summary: {e}", exc_info=True)
return error_response(
error_code=ErrorCode.SYSTEM_ERROR,
message="Failed to retrieve error summary",
status_code=500
)
@api_v3.route('/errors/plugin/<plugin_id>', methods=['GET'])
def get_plugin_errors(plugin_id):
"""
Get error health status for a specific plugin.
Args:
plugin_id: Plugin identifier
Returns health status and error statistics for the plugin, from the
display service's last report (see get_error_summary). A plugin with no
recorded errors is "healthy".
"""
try:
health = _errors.plugin_health_from_report(*_read_errors(), plugin_id)
health['last_error'] = _redact_error_record(health['last_error'])
return success_response(data=health, message="Plugin health retrieved")
except Exception as e:
logger.error(f"Error getting plugin health for {plugin_id}: {e}", exc_info=True)
return error_response(
error_code=ErrorCode.SYSTEM_ERROR,
message=f"Failed to retrieve health for plugin {plugin_id}",
status_code=500
)
@api_v3.route('/errors/clear', methods=['POST'])
def clear_old_errors():
"""
Clear error records older than specified age.
Request body (optional):
max_age_hours: Maximum age in hours (default: 24, max: 8760 = 1 year)
all: true clears every error recorded so far (max_age_hours ignored)
The errors live in the display service, so this records a clear request
that it applies within a few seconds. Reads hide the cleared errors from
the moment the request is recorded.
"""
try:
data = request.get_json(silent=True) or {}
clear_all = _coerce_to_bool(data.get('all'))
raw_max_age = data.get('max_age_hours', 24)
# Validate and coerce max_age_hours
max_age_hours = None
if not clear_all:
try:
max_age_hours = int(raw_max_age)
if max_age_hours < 1:
return error_response(
error_code=ErrorCode.INVALID_INPUT,
message="max_age_hours must be at least 1",
context={'provided_value': raw_max_age},
status_code=400
)
if max_age_hours > 8760: # 1 year max
return error_response(
error_code=ErrorCode.INVALID_INPUT,
message="max_age_hours cannot exceed 8760 (1 year)",
context={'provided_value': raw_max_age},
status_code=400
)
except (ValueError, TypeError, OverflowError):
return error_response(
error_code=ErrorCode.INVALID_INPUT,
message="max_age_hours must be a valid integer",
context={'provided_value': str(raw_max_age)},
status_code=400
)
now = _pkg.time.time()
cutoff = now if clear_all else now - max_age_hours * 3600
try:
result = _errors.request_error_clear(_errors_cache(), cutoff)
except OSError as e:
logger.error("Could not record an error clear request: %s", e)
return error_response(
error_code=ErrorCode.SYSTEM_ERROR,
message="Could not record the clear request in the shared cache",
status_code=500
)
scope = "all errors" if clear_all else f"errors older than {max_age_hours} hours"
return success_response(
data=result,
message=(f"Clear of {scope} requested; the display service applies it "
f"within about {int(_errors.SNAPSHOT_TICK_INTERVAL)} seconds")
)
except Exception as e:
logger.error(f"Error clearing old errors: {e}", exc_info=True)
return error_response(
error_code=ErrorCode.SYSTEM_ERROR,
message="Failed to clear old errors",
status_code=500
)
@api_v3.route('/integrations/mqtt-bridge', methods=['GET'])
def get_mqtt_bridge():
"""Bridge service state and its settings, minus the password."""
try:
config = _read_mqtt_bridge_config()
password = config.get('mqtt_password')
safe = {key: config.get(key, default)
for key, default in _MQTT_BRIDGE_DEFAULTS.items()}
return jsonify({
'status': 'success',
'data': {
'service': _mqtt_bridge_service_state(),
'config_exists': _MQTT_BRIDGE_CONFIG.is_file(),
'config_path': str(_MQTT_BRIDGE_CONFIG),
'config': safe,
# Enough to render "a password is set" without disclosing it.
'password_set': bool(password),
# Likewise the web-login API token (only needed off-Pi).
'api_token_set': bool(config.get('ledmatrix_api_token')),
'env_override_prefix': 'LEDMATRIX_MQTT_',
}
})
except Exception as e:
logger.exception('Error reading MQTT bridge settings')
return jsonify({'status': 'error', 'message': 'Could not read bridge settings',
'details': describe_exception(e)}), 500
@api_v3.route('/integrations/mqtt-bridge/config', methods=['PUT'])
def update_mqtt_bridge_config():
"""Write bridge_config.json.
The password is write-only: omit it to leave whatever is stored alone, send
a value to replace it, or send clear_password to remove it. It is never
returned by the GET above, so a form that round-tripped it would otherwise
have to blank it on every save.
"""
try:
# No `or {}` here: get_json(silent=True) returns None for a missing or
# unparseable body, and `None or {}` produced an empty dict that then
# satisfied the isinstance check below -- so malformed JSON, `null`,
# `[]` and `false` all reported success while applying nothing.
data = request.get_json(silent=True)
if not isinstance(data, dict):
return jsonify({'status': 'error', 'message': 'Body must be a JSON object'}), 400
config = _read_mqtt_bridge_config()
existing_password = config.get('mqtt_password')
updates = {}
for key in _MQTT_BRIDGE_DEFAULTS:
if key not in data:
continue
value, err = _coerce_mqtt_bridge_value(key, data[key])
if err:
return jsonify({'status': 'error', 'message': err}), 400
updates[key] = value
config.update(updates)
# Coerced, not merely truthy: the string "false" is truthy in Python,
# so a client echoing the field back as a string would have wiped a
# stored password it meant to keep.
if _coerce_to_bool(data.get('clear_password')):
config['mqtt_password'] = None
elif 'mqtt_password' in data and str(data['mqtt_password']) != '':
new_password = str(data['mqtt_password'])
if len(new_password) > 300:
return jsonify({'status': 'error', 'message': 'Password is too long'}), 400
config['mqtt_password'] = new_password
else:
config['mqtt_password'] = existing_password
# The web-login API token is write-only the same way.
if _coerce_to_bool(data.get('clear_api_token')):
config['ledmatrix_api_token'] = None
elif 'ledmatrix_api_token' in data and str(data['ledmatrix_api_token']).strip() != '':
new_token = str(data['ledmatrix_api_token']).strip()
if len(new_token) > 200:
return jsonify({'status': 'error', 'message': 'API token is too long'}), 400
config['ledmatrix_api_token'] = new_token
# CWE-319: a password with TLS off is sent in the clear. On a trusted
# LAN that is a normal, deliberate setup, so this is refused rather
# than forbidden -- allow_insecure_mqtt is the explicit acknowledgement.
insecure = bool(config.get('mqtt_password')) and not config.get('mqtt_tls')
if insecure and not config.get('allow_insecure_mqtt'):
return jsonify({
'status': 'error',
'message': 'MQTT credentials would cross the network in cleartext '
'with TLS disabled. Enable mqtt_tls, or set '
'allow_insecure_mqtt to accept that on a trusted network.'
}), 400
if insecure:
logger.warning('MQTT bridge: a password is set without TLS and '
'allow_insecure_mqtt is on; credentials will cross the '
'network in cleartext')
_MQTT_BRIDGE_DIR.mkdir(parents=True, exist_ok=True)
# Write via a temp file in the same directory so a crash mid-write
# cannot leave a half-written config the bridge would refuse to load.
fd, tmp_path = tempfile.mkstemp(dir=str(_MQTT_BRIDGE_DIR), prefix='.bridge_config.')
try:
with os.fdopen(fd, 'w', encoding='utf-8') as handle:
json.dump(config, handle, indent=2, sort_keys=True)
handle.write('\n')
os.chmod(tmp_path, 0o600)
os.replace(tmp_path, _MQTT_BRIDGE_CONFIG)
except Exception:
with contextlib.suppress(OSError):
os.unlink(tmp_path)
raise
service = _mqtt_bridge_service_state()
message = 'Bridge settings saved.'
if service['active']:
message += ' Restart the bridge for them to take effect.'
return jsonify({'status': 'success', 'message': message,
'data': {'password_set': bool(config.get('mqtt_password')),
'api_token_set': bool(config.get('ledmatrix_api_token')),
'restart_required': service['active']}})
except Exception as e:
logger.exception('Error saving MQTT bridge settings')
return jsonify({'status': 'error', 'message': 'Could not save bridge settings',
'details': describe_exception(e)}), 500