Replaced health checker with Prometheus metrics service.
Audit / Dependencies (push) Failing after 8s
CD / Build (push) Successful in 8s
CI / Formatting (push) Successful in 6s
CI / Linting (push) Successful in 6s
CI / Tests (push) Successful in 11s
CI / Type Checking (push) Successful in 14s
CI / Spelling (push) Successful in 13s
Audit / Dependencies (push) Failing after 8s
CD / Build (push) Successful in 8s
CI / Formatting (push) Successful in 6s
CI / Linting (push) Successful in 6s
CI / Tests (push) Successful in 11s
CI / Type Checking (push) Successful in 14s
CI / Spelling (push) Successful in 13s
This commit is contained in:
+4
-4
@@ -1,4 +1,4 @@
|
||||
# Health check endpoint URL.
|
||||
# If configured, a GET request will be sent to this URL after each successful update cycle.
|
||||
# Leave empty to disable health check reporting.
|
||||
health_check_endpoint: ""
|
||||
# Prometheus metrics endpoint.
|
||||
# When enabled, a /metrics endpoint is exposed via the maubot webapp.
|
||||
# Disabled by default.
|
||||
metrics_enabled: false
|
||||
|
||||
+15
-12
@@ -1,12 +1,15 @@
|
||||
maubot: 0.1.0
|
||||
id: dev.logal.owncastsentry
|
||||
version: 1.1.0
|
||||
license: Apache-2.0
|
||||
modules:
|
||||
- owncastsentry
|
||||
main_class: OwncastSentry
|
||||
database: true
|
||||
database_type: asyncpg
|
||||
config: true
|
||||
extra_files:
|
||||
- base-config.yaml
|
||||
maubot: 0.1.0
|
||||
id: dev.logal.owncastsentry
|
||||
version: 1.1.0
|
||||
license: Apache-2.0
|
||||
modules:
|
||||
- owncastsentry
|
||||
main_class: OwncastSentry
|
||||
database: true
|
||||
database_type: asyncpg
|
||||
config: true
|
||||
webapp: true
|
||||
dependencies:
|
||||
- prometheus_client>=0.24.1
|
||||
extra_files:
|
||||
- base-config.yaml
|
||||
|
||||
+66
-26
@@ -16,13 +16,15 @@
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from aiohttp import web
|
||||
from maubot import MessageEvent, Plugin # type: ignore[attr-defined]
|
||||
from maubot.handlers import command
|
||||
from prometheus_client.exposition import choose_encoder
|
||||
|
||||
from .commands import CommandHandler
|
||||
from .config import Config
|
||||
from .database import StreamRepository, SubscriptionRepository
|
||||
from .health_checker import HealthChecker
|
||||
from .metrics import ErrorSource, MetricsService
|
||||
from .migrations import get_upgrade_table
|
||||
from .notification_service import NotificationService
|
||||
from .owncast_client import OwncastClient
|
||||
@@ -57,8 +59,21 @@ class OwncastSentry(Plugin):
|
||||
config.load_and_update()
|
||||
db: Database = self.database # type: ignore[assignment]
|
||||
|
||||
# Initialize metrics service and register web endpoint (if enabled)
|
||||
self.metrics_service = MetricsService()
|
||||
self.metrics_service.set_build_info(str(self.loader.meta.version))
|
||||
if config.metrics_enabled and self.webapp is not None:
|
||||
self.webapp.add_route(
|
||||
method="GET", path="/metrics", handler=self._metrics_endpoint
|
||||
)
|
||||
|
||||
# Initialize the Owncast API client
|
||||
self.owncast_client = OwncastClient(self.log, str(self.loader.meta.version))
|
||||
self.owncast_client = OwncastClient(
|
||||
self.log, str(self.loader.meta.version), metrics=self.metrics_service
|
||||
)
|
||||
self.metrics_service.register_open_connections_gauge(
|
||||
lambda: self.owncast_client.open_connection_count
|
||||
)
|
||||
|
||||
# Initialize repositories
|
||||
self.stream_repo = StreamRepository(db)
|
||||
@@ -66,7 +81,10 @@ class OwncastSentry(Plugin):
|
||||
|
||||
# Initialize notification service
|
||||
self.notification_service = NotificationService(
|
||||
self.client, self.subscription_repo, self.log
|
||||
self.client,
|
||||
self.subscription_repo,
|
||||
self.log,
|
||||
metrics=self.metrics_service,
|
||||
)
|
||||
|
||||
# Initialize stream monitor
|
||||
@@ -76,13 +94,7 @@ class OwncastSentry(Plugin):
|
||||
self.subscription_repo,
|
||||
self.notification_service,
|
||||
self.log,
|
||||
)
|
||||
|
||||
# Initialize health checker
|
||||
self.health_checker = HealthChecker(
|
||||
db,
|
||||
self.owncast_client,
|
||||
self.log,
|
||||
metrics=self.metrics_service,
|
||||
)
|
||||
|
||||
# Initialize command handler
|
||||
@@ -97,41 +109,69 @@ class OwncastSentry(Plugin):
|
||||
self.sched.run_periodically(60, self._update_all_stream_states)
|
||||
|
||||
async def _update_all_stream_states(self) -> None:
|
||||
"""Update all stream states and perform health check."""
|
||||
# Get list of all stream domains with active subscriptions
|
||||
subscribed_domains = await self.subscription_repo.get_all_subscribed_domains()
|
||||
"""Update all stream states."""
|
||||
try:
|
||||
# Get list of all stream domains with active subscriptions
|
||||
subscribed_domains = (
|
||||
await self.subscription_repo.get_all_subscribed_domains()
|
||||
)
|
||||
|
||||
# Delegate to stream monitor and get results
|
||||
update_result = await self.stream_monitor.update_all_streams(subscribed_domains)
|
||||
|
||||
# Perform health check
|
||||
config: Config = self.config # type: ignore[assignment]
|
||||
await self.health_checker.perform_health_check(
|
||||
update_result,
|
||||
config.health_check_endpoint,
|
||||
)
|
||||
# Delegate to stream monitor
|
||||
await self.stream_monitor.update_all_streams(subscribed_domains)
|
||||
except Exception:
|
||||
self.metrics_service.record_error(ErrorSource.SCHEDULER_LOOP)
|
||||
self.log.exception("Unhandled exception in scheduler loop.")
|
||||
|
||||
@command.new(help="Subscribes to a new Owncast stream.")
|
||||
@command.argument("url")
|
||||
async def subscribe(self, evt: MessageEvent, url: str) -> None:
|
||||
"""Delegate subscribe command to CommandHandler."""
|
||||
await self.command_handler.subscribe(evt, url)
|
||||
try:
|
||||
await self.command_handler.subscribe(evt, url)
|
||||
except Exception:
|
||||
self.metrics_service.record_error(ErrorSource.COMMAND)
|
||||
self.log.exception("Unhandled exception in subscribe command.")
|
||||
await evt.reply("An unexpected error occurred. Please try again later.")
|
||||
|
||||
@command.new(help="Unsubscribes from an Owncast stream.")
|
||||
@command.argument("url")
|
||||
async def unsubscribe(self, evt: MessageEvent, url: str) -> None:
|
||||
"""Delegate unsubscribe command to CommandHandler."""
|
||||
await self.command_handler.unsubscribe(evt, url)
|
||||
try:
|
||||
await self.command_handler.unsubscribe(evt, url)
|
||||
except Exception:
|
||||
self.metrics_service.record_error(ErrorSource.COMMAND)
|
||||
self.log.exception("Unhandled exception in unsubscribe command.")
|
||||
await evt.reply("An unexpected error occurred. Please try again later.")
|
||||
|
||||
@command.new(help="Lists all stream subscriptions in this room.")
|
||||
async def subscriptions(self, evt: MessageEvent) -> None:
|
||||
"""Delegate subscriptions command to CommandHandler."""
|
||||
await self.command_handler.subscriptions(evt)
|
||||
try:
|
||||
await self.command_handler.subscriptions(evt)
|
||||
except Exception:
|
||||
self.metrics_service.record_error(ErrorSource.COMMAND)
|
||||
self.log.exception("Unhandled exception in subscriptions command.")
|
||||
await evt.reply("An unexpected error occurred. Please try again later.")
|
||||
|
||||
@command.new(help="Lists currently live streams in this room.")
|
||||
async def live(self, evt: MessageEvent) -> None:
|
||||
"""Delegate live command to CommandHandler."""
|
||||
await self.command_handler.live(evt)
|
||||
try:
|
||||
await self.command_handler.live(evt)
|
||||
except Exception:
|
||||
self.metrics_service.record_error(ErrorSource.COMMAND)
|
||||
self.log.exception("Unhandled exception in live command.")
|
||||
await evt.reply("An unexpected error occurred. Please try again later.")
|
||||
|
||||
async def _metrics_endpoint(self, request: web.Request) -> web.Response:
|
||||
"""Serve Prometheus metrics."""
|
||||
accept = request.headers.get("Accept", "")
|
||||
encoder, content_type = choose_encoder(accept)
|
||||
output = encoder(self.metrics_service.registry)
|
||||
response = web.Response(body=output)
|
||||
response.headers["Content-Type"] = content_type
|
||||
return response
|
||||
|
||||
async def stop(self) -> None:
|
||||
"""Clean up resources by closing the HTTP session."""
|
||||
|
||||
@@ -25,9 +25,9 @@ class Config(BaseProxyConfig):
|
||||
|
||||
:param helper: ConfigUpdateHelper for copying values.
|
||||
"""
|
||||
helper.copy("health_check_endpoint")
|
||||
helper.copy("metrics_enabled")
|
||||
|
||||
@property
|
||||
def health_check_endpoint(self) -> str:
|
||||
"""Return the configured health check endpoint URL."""
|
||||
return self["health_check_endpoint"] # type: ignore[no-any-return]
|
||||
def metrics_enabled(self) -> bool:
|
||||
"""Return whether the Prometheus metrics endpoint is enabled."""
|
||||
return self["metrics_enabled"] # type: ignore[no-any-return]
|
||||
|
||||
@@ -1,141 +0,0 @@
|
||||
# Copyright 2026 Logan Fick
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Health checking service for OwncastSentry."""
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import logging
|
||||
|
||||
from mautrix.util.async_db import Database
|
||||
|
||||
from .models import UpdateResult
|
||||
from .owncast_client import OwncastClient
|
||||
|
||||
|
||||
@dataclass
|
||||
class HealthStatus:
|
||||
"""Represents the health status of the plugin."""
|
||||
|
||||
database_healthy: bool
|
||||
http_healthy: bool
|
||||
|
||||
@property
|
||||
def is_healthy(self) -> bool:
|
||||
"""Check if all health components are healthy.
|
||||
|
||||
:return: True if all checks pass.
|
||||
"""
|
||||
return self.database_healthy and self.http_healthy
|
||||
|
||||
|
||||
class HealthChecker:
|
||||
"""Service for performing health checks on the plugin."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
database: Database,
|
||||
owncast_client: OwncastClient,
|
||||
logger: logging.Logger,
|
||||
):
|
||||
"""Initialize the health checker.
|
||||
|
||||
:param database: The maubot database instance.
|
||||
:param owncast_client: Client for making HTTP requests.
|
||||
:param logger: Logger instance for debugging.
|
||||
"""
|
||||
self.db = database
|
||||
self.owncast_client = owncast_client
|
||||
self.log = logger
|
||||
|
||||
async def check_database(self) -> bool:
|
||||
"""Check if the database is functioning by executing a simple query.
|
||||
|
||||
:return: True if database is healthy, False otherwise.
|
||||
"""
|
||||
try:
|
||||
async with self.db.acquire() as conn: # type: ignore[var-annotated]
|
||||
await conn.fetchval("SELECT 1")
|
||||
return True
|
||||
except Exception as e: # broad catch - DB backends raise varied errors
|
||||
self.log.warning(f"Database health check failed: {e}")
|
||||
return False
|
||||
|
||||
async def perform_health_check(
|
||||
self,
|
||||
update_result: UpdateResult,
|
||||
endpoint: str,
|
||||
) -> None:
|
||||
"""Perform health check and report to configured endpoint if all healthy.
|
||||
|
||||
:param update_result: Result of the stream update cycle.
|
||||
:param endpoint: Health check endpoint URL (empty string to skip reporting).
|
||||
"""
|
||||
# Check database health
|
||||
database_healthy = await self.check_database()
|
||||
|
||||
# Evaluate HTTP health from update results
|
||||
http_healthy = update_result.http_healthy
|
||||
|
||||
# Create health status
|
||||
status = HealthStatus(
|
||||
database_healthy=database_healthy,
|
||||
http_healthy=http_healthy,
|
||||
)
|
||||
|
||||
self.log.debug(
|
||||
f"Health check: database={database_healthy}, http={http_healthy}, "
|
||||
f"streams={update_result.total_streams}, "
|
||||
f"succeeded={update_result.successful_checks}, "
|
||||
f"failed={update_result.failed_checks}"
|
||||
)
|
||||
|
||||
# Skip endpoint notification if not configured
|
||||
if not endpoint.strip():
|
||||
self.log.debug("Health check endpoint not configured, skipping report.")
|
||||
return
|
||||
|
||||
# Only send to endpoint if ALL checks pass
|
||||
if not status.is_healthy:
|
||||
self.log.warning(
|
||||
f"Health check failed, not reporting to endpoint. "
|
||||
f"database={database_healthy}, http={http_healthy}"
|
||||
)
|
||||
return
|
||||
|
||||
# Send GET request to health endpoint
|
||||
await self._send_health_report(endpoint)
|
||||
|
||||
async def _send_health_report(self, endpoint: str) -> None:
|
||||
"""Send a GET request to the health check endpoint.
|
||||
|
||||
:param endpoint: The endpoint URL.
|
||||
"""
|
||||
try:
|
||||
async with self.owncast_client.session.get(
|
||||
endpoint, allow_redirects=True
|
||||
) as response:
|
||||
if 200 <= response.status < 300:
|
||||
self.log.debug(
|
||||
f"Health check reported successfully (status={response.status})"
|
||||
)
|
||||
else:
|
||||
self.log.warning(
|
||||
"Health check endpoint returned "
|
||||
f"non-success status: {response.status}"
|
||||
)
|
||||
except Exception as e:
|
||||
self.log.warning(f"Failed to report health check to endpoint: {e}")
|
||||
@@ -0,0 +1,234 @@
|
||||
# Copyright 2026 Logan Fick
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Prometheus metrics service for OwncastSentry."""
|
||||
|
||||
import time
|
||||
from contextlib import contextmanager, suppress
|
||||
from enum import StrEnum
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from prometheus_client import CollectorRegistry, Counter, Gauge, Info
|
||||
|
||||
from .models import StreamStatus
|
||||
|
||||
|
||||
class NotificationType(StrEnum):
|
||||
"""Notification type labels for the delivery counter."""
|
||||
|
||||
LIVE = "live"
|
||||
TITLE_CHANGE = "title_change"
|
||||
CLEANUP_WARNING = "cleanup_warning"
|
||||
CLEANUP_DELETION = "cleanup_deletion"
|
||||
|
||||
|
||||
class ErrorSource(StrEnum):
|
||||
"""Error source labels for the error counter."""
|
||||
|
||||
SCHEDULER_LOOP = "scheduler_loop"
|
||||
COMMAND = "command"
|
||||
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import Callable, Generator
|
||||
|
||||
# Mapping from StreamStatus enum to numeric gauge values
|
||||
_STATUS_VALUES: dict[StreamStatus, float] = {
|
||||
StreamStatus.ONLINE: 1.0,
|
||||
StreamStatus.OFFLINE: 0.0,
|
||||
StreamStatus.UNKNOWN: -1.0,
|
||||
}
|
||||
|
||||
|
||||
class _ResponseTimer:
|
||||
"""Timer that only records an observation when explicitly marked successful."""
|
||||
|
||||
__slots__ = ("_domain", "_gauge", "_should_observe", "_start")
|
||||
|
||||
def __init__(self, gauge: Gauge, domain: str) -> None:
|
||||
self._gauge = gauge
|
||||
self._domain = domain
|
||||
self._start = time.monotonic()
|
||||
self._should_observe = False
|
||||
|
||||
def success(self) -> None:
|
||||
"""Mark the request as successful so the duration is recorded."""
|
||||
self._should_observe = True
|
||||
|
||||
def _finalize(self) -> None:
|
||||
if self._should_observe:
|
||||
self._gauge.labels(domain=self._domain).set(
|
||||
max(time.monotonic() - self._start, 0)
|
||||
)
|
||||
else:
|
||||
with suppress(KeyError):
|
||||
self._gauge.remove(self._domain)
|
||||
|
||||
|
||||
class MetricsService:
|
||||
"""Manages Prometheus metrics with an isolated registry."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Initialize metrics with a custom collector registry."""
|
||||
self.registry = CollectorRegistry()
|
||||
|
||||
self.notification_delivery_total = Counter(
|
||||
"owncastsentry_notification_delivery_total",
|
||||
"Total notification delivery attempts to individual rooms",
|
||||
["type", "result"],
|
||||
registry=self.registry,
|
||||
)
|
||||
self.stream_status = Gauge(
|
||||
"owncastsentry_stream_status",
|
||||
"Current stream status (1=online, 0=offline, -1=unknown)",
|
||||
["domain"],
|
||||
registry=self.registry,
|
||||
)
|
||||
self.stream_subscriptions = Gauge(
|
||||
"owncastsentry_stream_subscriptions",
|
||||
"Number of room subscriptions per stream domain",
|
||||
["domain"],
|
||||
registry=self.registry,
|
||||
)
|
||||
self.check_failures = Gauge(
|
||||
"owncastsentry_check_failures",
|
||||
"Consecutive check failure count per stream domain",
|
||||
["domain"],
|
||||
registry=self.registry,
|
||||
)
|
||||
self.api_response_seconds = Gauge(
|
||||
"owncastsentry_api_response_seconds",
|
||||
"Last successful HTTP response time in seconds per stream domain",
|
||||
["domain"],
|
||||
registry=self.registry,
|
||||
)
|
||||
self.build_info = Info(
|
||||
"owncastsentry",
|
||||
"OwncastSentry build information",
|
||||
registry=self.registry,
|
||||
)
|
||||
self.open_connections = Gauge(
|
||||
"owncastsentry_http_connections_open",
|
||||
"Total number of open HTTP connections (idle and active)",
|
||||
registry=self.registry,
|
||||
)
|
||||
self.errors_total = Counter(
|
||||
"owncastsentry_errors_total",
|
||||
"Internal errors by source",
|
||||
["source"],
|
||||
registry=self.registry,
|
||||
)
|
||||
|
||||
# Initialize all known label combinations so they start at 0
|
||||
for notification_type in NotificationType:
|
||||
for result in ("success", "failure"):
|
||||
self.notification_delivery_total.labels(
|
||||
type=notification_type, result=result
|
||||
)
|
||||
for source in ErrorSource:
|
||||
self.errors_total.labels(source=source)
|
||||
|
||||
def record_delivery(
|
||||
self,
|
||||
notification_type: NotificationType,
|
||||
*,
|
||||
successful: int = 0,
|
||||
failed: int = 0,
|
||||
) -> None:
|
||||
"""Record notification delivery results.
|
||||
|
||||
:param notification_type: The type of notification delivered.
|
||||
:param successful: Number of successful room deliveries.
|
||||
:param failed: Number of failed room deliveries.
|
||||
"""
|
||||
self.notification_delivery_total.labels(
|
||||
type=notification_type, result="success"
|
||||
).inc(successful)
|
||||
self.notification_delivery_total.labels(
|
||||
type=notification_type, result="failure"
|
||||
).inc(failed)
|
||||
|
||||
def set_stream_status(self, domain: str, status: StreamStatus) -> None:
|
||||
"""Set the status gauge for a stream domain.
|
||||
|
||||
:param domain: The stream domain.
|
||||
:param status: The current stream status.
|
||||
"""
|
||||
self.stream_status.labels(domain=domain).set(_STATUS_VALUES[status])
|
||||
|
||||
def set_check_failures(self, domain: str, count: int) -> None:
|
||||
"""Set the consecutive failure count for a stream domain.
|
||||
|
||||
:param domain: The stream domain.
|
||||
:param count: The current failure counter value.
|
||||
"""
|
||||
self.check_failures.labels(domain=domain).set(count)
|
||||
|
||||
def set_subscription_count(self, domain: str, count: int) -> None:
|
||||
"""Set the subscription count for a stream domain.
|
||||
|
||||
:param domain: The stream domain.
|
||||
:param count: The number of room subscriptions.
|
||||
"""
|
||||
self.stream_subscriptions.labels(domain=domain).set(count)
|
||||
|
||||
@contextmanager
|
||||
def response_timer(self, domain: str) -> Generator[_ResponseTimer]:
|
||||
"""Return a context manager that times an HTTP request.
|
||||
|
||||
Call ``timer.success()`` inside the block to record the duration.
|
||||
If ``success()`` is never called, nothing is recorded.
|
||||
|
||||
:param domain: The stream domain being queried.
|
||||
"""
|
||||
timer = _ResponseTimer(self.api_response_seconds, domain)
|
||||
try:
|
||||
yield timer
|
||||
finally:
|
||||
timer._finalize()
|
||||
|
||||
def set_build_info(self, version: str) -> None:
|
||||
"""Set the build version info metric.
|
||||
|
||||
:param version: The plugin version string.
|
||||
"""
|
||||
self.build_info.info({"version": version})
|
||||
|
||||
def register_open_connections_gauge(self, callback: Callable[[], float]) -> None:
|
||||
"""Register a gauge that reads open connection count on scrape.
|
||||
|
||||
:param callback: Function returning the current open count.
|
||||
"""
|
||||
self.open_connections.set_function(callback)
|
||||
|
||||
def record_error(self, source: ErrorSource) -> None:
|
||||
"""Increment the internal error counter.
|
||||
|
||||
:param source: The source of the error.
|
||||
"""
|
||||
self.errors_total.labels(source=source).inc()
|
||||
|
||||
def remove_stream(self, domain: str) -> None:
|
||||
"""Remove a stream's gauge labels after cleanup deletion.
|
||||
|
||||
:param domain: The stream domain to remove.
|
||||
"""
|
||||
with suppress(KeyError):
|
||||
self.stream_status.remove(domain)
|
||||
with suppress(KeyError):
|
||||
self.check_failures.remove(domain)
|
||||
with suppress(KeyError):
|
||||
self.stream_subscriptions.remove(domain)
|
||||
with suppress(KeyError):
|
||||
self.api_response_seconds.remove(domain)
|
||||
@@ -99,20 +99,6 @@ class UpdateResult:
|
||||
successful_checks: int
|
||||
failed_checks: int
|
||||
|
||||
@property
|
||||
def http_healthy(self) -> bool:
|
||||
"""Determine HTTP health based on update results.
|
||||
|
||||
HTTP is considered healthy if:
|
||||
- No streams are subscribed (nothing to check), OR
|
||||
- At least one stream check succeeded
|
||||
|
||||
:return: True if HTTP is considered healthy.
|
||||
"""
|
||||
if self.total_streams == 0:
|
||||
return True
|
||||
return self.successful_checks > 0
|
||||
|
||||
|
||||
@dataclass
|
||||
class StreamConfig:
|
||||
|
||||
@@ -20,6 +20,7 @@ from typing import TYPE_CHECKING, Any
|
||||
|
||||
from mautrix.types import MessageType, TextMessageEventContent
|
||||
|
||||
from .metrics import NotificationType
|
||||
from .utils import (
|
||||
CLEANUP_DELETE_DAYS,
|
||||
CLEANUP_WARNING_DAYS,
|
||||
@@ -31,6 +32,7 @@ if TYPE_CHECKING:
|
||||
import logging
|
||||
|
||||
from .database import SubscriptionRepository
|
||||
from .metrics import MetricsService
|
||||
|
||||
|
||||
class NotificationService:
|
||||
@@ -41,16 +43,19 @@ class NotificationService:
|
||||
client: Any,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
logger: logging.Logger,
|
||||
metrics: MetricsService,
|
||||
) -> None:
|
||||
"""Initialize the notification service.
|
||||
|
||||
:param client: The Matrix client for sending messages.
|
||||
:param subscription_repo: Repository for managing subscriptions.
|
||||
:param logger: Logger instance for debugging.
|
||||
:param metrics: Metrics service for recording counters.
|
||||
"""
|
||||
self.client = client
|
||||
self.subscription_repo = subscription_repo
|
||||
self.log = logger
|
||||
self.metrics = metrics
|
||||
|
||||
# Cache for tracking when notifications were last sent
|
||||
self.notification_timers_cache: dict[str, float] = {}
|
||||
@@ -102,6 +107,12 @@ class NotificationService:
|
||||
f"{failed} failed."
|
||||
)
|
||||
|
||||
self.metrics.record_delivery(
|
||||
NotificationType.TITLE_CHANGE if title_change else NotificationType.LIVE,
|
||||
successful=successful,
|
||||
failed=failed,
|
||||
)
|
||||
|
||||
async def _send_notification(
|
||||
self, room_id: str, body_text: str, domain: str
|
||||
) -> None:
|
||||
@@ -231,6 +242,10 @@ class NotificationService:
|
||||
f"[{domain}] Sent cleanup warning to {successful} rooms ({failed} failed)."
|
||||
)
|
||||
|
||||
self.metrics.record_delivery(
|
||||
NotificationType.CLEANUP_WARNING, successful=successful, failed=failed
|
||||
)
|
||||
|
||||
async def send_cleanup_deletion(self, domain: str) -> None:
|
||||
"""Send cleanup deletion notification to all subscribed rooms.
|
||||
|
||||
@@ -251,3 +266,7 @@ class NotificationService:
|
||||
f"[{domain}] Sent cleanup deletion notice to "
|
||||
f"{successful} rooms ({failed} failed)."
|
||||
)
|
||||
|
||||
self.metrics.record_delivery(
|
||||
NotificationType.CLEANUP_DELETION, successful=successful, failed=failed
|
||||
)
|
||||
|
||||
@@ -29,17 +29,26 @@ from .utils import (
|
||||
if TYPE_CHECKING:
|
||||
import logging
|
||||
|
||||
from .metrics import MetricsService
|
||||
|
||||
|
||||
class OwncastClient:
|
||||
"""HTTP client for communicating with Owncast instances."""
|
||||
|
||||
def __init__(self, logger: logging.Logger, version: str) -> None:
|
||||
def __init__(
|
||||
self,
|
||||
logger: logging.Logger,
|
||||
version: str,
|
||||
metrics: MetricsService,
|
||||
) -> None:
|
||||
"""Initialize the Owncast client with an HTTP session.
|
||||
|
||||
:param logger: Logger instance for debugging
|
||||
:param version: Plugin version string for the User-Agent header
|
||||
:param metrics: Metrics service for recording response times.
|
||||
"""
|
||||
self.log = logger
|
||||
self.metrics = metrics
|
||||
|
||||
# Set up HTTP session configuration
|
||||
headers = {"User-Agent": user_agent(version)}
|
||||
@@ -100,21 +109,24 @@ class OwncastClient:
|
||||
:return: A StreamState if available, None on error.
|
||||
"""
|
||||
self.log.debug(f"[{domain}] Fetching current stream state...")
|
||||
new_state = await self._fetch_json(domain, OWNCAST_STATUS_PATH)
|
||||
if new_state is None:
|
||||
return None
|
||||
with self.metrics.response_timer(domain) as timer:
|
||||
new_state = await self._fetch_json(domain, OWNCAST_STATUS_PATH)
|
||||
|
||||
# Validate the response contains all basic info needed
|
||||
missing = REQUIRED_STATUS_FIELDS - new_state.keys()
|
||||
if missing:
|
||||
self.log.warning(
|
||||
f"[{domain}] Rejecting response to request on "
|
||||
f"{OWNCAST_STATUS_PATH} as it is missing "
|
||||
f"fields: {', '.join(sorted(missing))}"
|
||||
)
|
||||
return None
|
||||
if new_state is None:
|
||||
return None
|
||||
|
||||
return StreamState.from_api_response(new_state, domain)
|
||||
# Validate the response contains all basic info needed
|
||||
missing = REQUIRED_STATUS_FIELDS - new_state.keys()
|
||||
if missing:
|
||||
self.log.warning(
|
||||
f"[{domain}] Rejecting response to request on "
|
||||
f"{OWNCAST_STATUS_PATH} as it is missing "
|
||||
f"fields: {', '.join(sorted(missing))}"
|
||||
)
|
||||
return None
|
||||
|
||||
timer.success()
|
||||
return StreamState.from_api_response(new_state, domain)
|
||||
|
||||
async def get_stream_config(self, domain: str) -> StreamConfig | None:
|
||||
"""Get the current stream config for a given domain.
|
||||
@@ -126,11 +138,13 @@ class OwncastClient:
|
||||
:return: A StreamConfig, or None if fetch failed.
|
||||
"""
|
||||
self.log.debug(f"[{domain}] Fetching current stream config...")
|
||||
config = await self._fetch_json(domain, OWNCAST_CONFIG_PATH)
|
||||
if config is None:
|
||||
return None
|
||||
with self.metrics.response_timer(domain) as timer:
|
||||
config = await self._fetch_json(domain, OWNCAST_CONFIG_PATH)
|
||||
if config is None:
|
||||
return None
|
||||
|
||||
return StreamConfig.from_api_response(config)
|
||||
timer.success()
|
||||
return StreamConfig.from_api_response(config)
|
||||
|
||||
async def validate_instance(self, domain: str) -> bool:
|
||||
"""Validate that a domain is a valid Owncast instance.
|
||||
@@ -141,6 +155,16 @@ class OwncastClient:
|
||||
state = await self.get_stream_state(domain)
|
||||
return state is not None
|
||||
|
||||
@property
|
||||
def open_connection_count(self) -> int:
|
||||
"""Return the total number of open HTTP connections."""
|
||||
connector = self.session.connector
|
||||
if connector is None:
|
||||
return 0
|
||||
idle = sum(len(conns) for conns in connector._conns.values())
|
||||
active = len(connector._acquired)
|
||||
return idle + active
|
||||
|
||||
async def close(self) -> None:
|
||||
"""Close the HTTP session."""
|
||||
await self.session.close()
|
||||
|
||||
@@ -18,7 +18,7 @@ import asyncio
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from .models import StreamState, UpdateResult
|
||||
from .models import StreamState, StreamStatus, UpdateResult
|
||||
from .utils import (
|
||||
CLEANUP_DELETE_THRESHOLD,
|
||||
CLEANUP_WARNING_THRESHOLD,
|
||||
@@ -30,6 +30,7 @@ if TYPE_CHECKING:
|
||||
import logging
|
||||
|
||||
from .database import StreamRepository, SubscriptionRepository
|
||||
from .metrics import MetricsService
|
||||
from .notification_service import NotificationService
|
||||
from .owncast_client import OwncastClient
|
||||
|
||||
@@ -44,6 +45,7 @@ class StreamMonitor:
|
||||
subscription_repo: SubscriptionRepository,
|
||||
notification_service: NotificationService,
|
||||
logger: logging.Logger,
|
||||
metrics: MetricsService,
|
||||
) -> None:
|
||||
"""Initialize the stream monitor.
|
||||
|
||||
@@ -52,12 +54,14 @@ class StreamMonitor:
|
||||
:param subscription_repo: Repository for subscription data.
|
||||
:param notification_service: Service for sending notifications.
|
||||
:param logger: Logger instance for debugging.
|
||||
:param metrics: Metrics service for recording Prometheus metrics.
|
||||
"""
|
||||
self.owncast_client = owncast_client
|
||||
self.stream_repo = stream_repo
|
||||
self.subscription_repo = subscription_repo
|
||||
self.notification_service = notification_service
|
||||
self.log = logger
|
||||
self.metrics = metrics
|
||||
|
||||
# Cache for tracking when streams last went offline
|
||||
self.offline_timer_cache: dict[str, float] = {}
|
||||
@@ -90,6 +94,10 @@ class StreamMonitor:
|
||||
f"{failed_checks} failed."
|
||||
)
|
||||
|
||||
for domain in subscribed_domains:
|
||||
count = await self.subscription_repo.count_by_domain(domain)
|
||||
self.metrics.set_subscription_count(domain, count)
|
||||
|
||||
return UpdateResult(
|
||||
total_streams=total_streams,
|
||||
successful_checks=successful_checks,
|
||||
@@ -123,6 +131,10 @@ class StreamMonitor:
|
||||
)
|
||||
# Check cleanup thresholds even when skipping query
|
||||
await self._check_cleanup_thresholds(domain, failure_counter + 1)
|
||||
updated_state = await self.stream_repo.get_by_domain(domain)
|
||||
if updated_state is not None:
|
||||
self.metrics.set_stream_status(domain, updated_state.status)
|
||||
self.metrics.set_check_failures(domain, failure_counter + 1)
|
||||
# Backoff is expected behavior, not a failure
|
||||
return True
|
||||
|
||||
@@ -148,11 +160,16 @@ class StreamMonitor:
|
||||
)
|
||||
# Check cleanup thresholds after connection failure
|
||||
await self._check_cleanup_thresholds(domain, failure_counter + 1)
|
||||
updated_state = await self.stream_repo.get_by_domain(domain)
|
||||
if updated_state is not None:
|
||||
self.metrics.set_stream_status(domain, updated_state.status)
|
||||
self.metrics.set_check_failures(domain, failure_counter + 1)
|
||||
# Actual connection failure
|
||||
return False
|
||||
|
||||
# Fetch succeeded! Reset failure counter
|
||||
await self.stream_repo.reset_failure_counter(domain)
|
||||
self.metrics.set_check_failures(domain, 0)
|
||||
|
||||
# Initialize timer cache entries to prevent KeyError on first access
|
||||
self.offline_timer_cache.setdefault(domain, 0)
|
||||
@@ -300,6 +317,10 @@ class StreamMonitor:
|
||||
|
||||
# All done.
|
||||
self.log.debug(f"[{domain}] State update completed.")
|
||||
if new_state.last_connect_time is not None:
|
||||
self.metrics.set_stream_status(domain, StreamStatus.ONLINE)
|
||||
else:
|
||||
self.metrics.set_stream_status(domain, StreamStatus.OFFLINE)
|
||||
return True
|
||||
|
||||
async def _check_cleanup_thresholds(self, domain: str, counter: int) -> None:
|
||||
@@ -335,3 +356,4 @@ class StreamMonitor:
|
||||
f"Deleted {deleted_count} subscriptions "
|
||||
f"and stream record."
|
||||
)
|
||||
self.metrics.remove_stream(domain)
|
||||
|
||||
+1
-1
@@ -7,7 +7,7 @@ authors = [
|
||||
]
|
||||
license = "Apache-2.0"
|
||||
requires-python = ">=3.14"
|
||||
dependencies = []
|
||||
dependencies = ["prometheus_client>=0.24.1"]
|
||||
|
||||
[project.urls]
|
||||
Repository = "https://git.logal.dev/LogalDeveloper/OwncastSentry"
|
||||
|
||||
+7
-2
@@ -14,13 +14,12 @@
|
||||
|
||||
"""Shared test fixtures and stubs for OwncastSentry tests."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
import pytest
|
||||
from mautrix.util.async_db import Database
|
||||
from prometheus_client import generate_latest
|
||||
|
||||
from owncastsentry import OwncastSentry
|
||||
from owncastsentry.config import Config
|
||||
@@ -31,9 +30,15 @@ if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
from pathlib import Path
|
||||
|
||||
from owncastsentry.metrics import MetricsService
|
||||
from owncastsentry.models import StreamConfig, StreamState
|
||||
|
||||
|
||||
def generate_metrics_output(metrics: MetricsService) -> str:
|
||||
"""Generate Prometheus text format output from a MetricsService registry."""
|
||||
return generate_latest(metrics.registry).decode("utf-8")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def database(tmp_path: Path) -> AsyncIterator[Database]:
|
||||
"""Yield a real SQLite-backed mautrix Database with migrations applied."""
|
||||
|
||||
@@ -14,8 +14,6 @@
|
||||
|
||||
"""Tests for bot command handlers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from datetime import UTC, datetime, timedelta
|
||||
|
||||
@@ -14,8 +14,6 @@
|
||||
|
||||
"""Tests for database repository classes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
|
||||
@@ -1,172 +0,0 @@
|
||||
# Copyright 2026 Logan Fick
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Tests for the health checking service."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
from aioresponses import aioresponses
|
||||
|
||||
from owncastsentry.health_checker import HealthChecker, HealthStatus
|
||||
from owncastsentry.models import UpdateResult
|
||||
from owncastsentry.owncast_client import OwncastClient
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
from mautrix.util.async_db import Database
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def health_checker(database: Database) -> AsyncIterator[HealthChecker]:
|
||||
"""Yield a HealthChecker backed by a real OwncastClient."""
|
||||
client = OwncastClient(logger=logging.getLogger("test"), version="0.0.0")
|
||||
yield HealthChecker(database, client, logging.getLogger("test"))
|
||||
await client.close()
|
||||
|
||||
|
||||
class TestUpdateResultHttpHealthy:
|
||||
"""HTTP health derivation from update results."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("total", "successful", "failed", "expected"),
|
||||
[
|
||||
pytest.param(0, 0, 0, True, id="no-streams-is-healthy"),
|
||||
pytest.param(3, 2, 1, True, id="some-successes-is-healthy"),
|
||||
pytest.param(3, 0, 3, False, id="all-failures-is-unhealthy"),
|
||||
],
|
||||
)
|
||||
def test_http_healthy(
|
||||
self, total: int, successful: int, failed: int, expected: bool
|
||||
) -> None:
|
||||
"""Derive HTTP health from stream check results."""
|
||||
result = UpdateResult(
|
||||
total_streams=total,
|
||||
successful_checks=successful,
|
||||
failed_checks=failed,
|
||||
)
|
||||
assert result.http_healthy is expected
|
||||
|
||||
|
||||
class TestHealthStatusIsHealthy:
|
||||
"""Overall health status derivation."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("db_healthy", "http_healthy", "expected"),
|
||||
[
|
||||
pytest.param(True, True, True, id="all-healthy"),
|
||||
pytest.param(False, True, False, id="db-unhealthy"),
|
||||
pytest.param(True, False, False, id="http-unhealthy"),
|
||||
pytest.param(False, False, False, id="both-unhealthy"),
|
||||
],
|
||||
)
|
||||
def test_is_healthy(
|
||||
self, db_healthy: bool, http_healthy: bool, expected: bool
|
||||
) -> None:
|
||||
"""Derive overall health from component health."""
|
||||
status = HealthStatus(database_healthy=db_healthy, http_healthy=http_healthy)
|
||||
assert status.is_healthy is expected
|
||||
|
||||
|
||||
class TestCheckDatabase:
|
||||
"""Database health check."""
|
||||
|
||||
async def test_returns_true_for_healthy_db(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Return True when the database responds to queries."""
|
||||
assert await health_checker.check_database() is True
|
||||
|
||||
async def test_returns_false_for_stopped_db(
|
||||
self, health_checker: HealthChecker, database: Database
|
||||
) -> None:
|
||||
"""Return False when the database connection is closed."""
|
||||
await database.stop()
|
||||
assert await health_checker.check_database() is False
|
||||
|
||||
|
||||
class TestPerformHealthCheck:
|
||||
"""Health check orchestration and endpoint reporting."""
|
||||
|
||||
async def test_skips_report_when_no_endpoint(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Skip reporting when endpoint is empty."""
|
||||
result = UpdateResult(total_streams=0, successful_checks=0, failed_checks=0)
|
||||
with aioresponses():
|
||||
await health_checker.perform_health_check(result, "")
|
||||
|
||||
async def test_skips_report_when_unhealthy(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Skip health report when all stream checks failed."""
|
||||
result = UpdateResult(total_streams=3, successful_checks=0, failed_checks=3)
|
||||
with aioresponses():
|
||||
await health_checker.perform_health_check(
|
||||
result, "https://health.example.com/ping"
|
||||
)
|
||||
|
||||
async def test_sends_report_when_healthy(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Send GET to endpoint when all checks pass."""
|
||||
result = UpdateResult(total_streams=1, successful_checks=1, failed_checks=0)
|
||||
with aioresponses() as mocked:
|
||||
mocked.get("https://health.example.com/ping", status=200)
|
||||
await health_checker.perform_health_check(
|
||||
result, "https://health.example.com/ping"
|
||||
)
|
||||
|
||||
async def test_skips_report_for_whitespace_endpoint(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Skip reporting when endpoint is whitespace."""
|
||||
result = UpdateResult(total_streams=0, successful_checks=0, failed_checks=0)
|
||||
with aioresponses():
|
||||
await health_checker.perform_health_check(result, " ")
|
||||
|
||||
|
||||
class TestSendHealthReport:
|
||||
"""Health report HTTP delivery."""
|
||||
|
||||
async def test_handles_success_response(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Complete without error on a 2xx response."""
|
||||
with aioresponses() as mocked:
|
||||
mocked.get("https://health.example.com/ping", status=200)
|
||||
await health_checker._send_health_report("https://health.example.com/ping")
|
||||
|
||||
async def test_handles_non_success_response(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Complete without error on a non-2xx response."""
|
||||
with aioresponses() as mocked:
|
||||
mocked.get("https://health.example.com/ping", status=500)
|
||||
await health_checker._send_health_report("https://health.example.com/ping")
|
||||
|
||||
async def test_handles_connection_error(
|
||||
self, health_checker: HealthChecker
|
||||
) -> None:
|
||||
"""Complete without error on a connection failure."""
|
||||
with aioresponses() as mocked:
|
||||
mocked.get(
|
||||
"https://health.example.com/ping",
|
||||
exception=ConnectionError(),
|
||||
)
|
||||
await health_checker._send_health_report("https://health.example.com/ping")
|
||||
@@ -0,0 +1,314 @@
|
||||
# Copyright 2026 Logan Fick
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
"""Tests for the Prometheus metrics service."""
|
||||
|
||||
import pytest
|
||||
|
||||
from owncastsentry.metrics import ErrorSource, MetricsService, NotificationType
|
||||
from owncastsentry.models import StreamStatus
|
||||
from tests.conftest import generate_metrics_output
|
||||
|
||||
|
||||
class TestRecordDelivery:
|
||||
"""Notification delivery counter with type and result labels."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("notification_type", "successful", "failed", "expected_fragments"),
|
||||
[
|
||||
pytest.param(
|
||||
NotificationType.LIVE,
|
||||
3,
|
||||
0,
|
||||
['result="success",type="live"} 3.0'],
|
||||
id="live-success",
|
||||
),
|
||||
pytest.param(
|
||||
NotificationType.LIVE,
|
||||
0,
|
||||
2,
|
||||
['result="failure",type="live"} 2.0'],
|
||||
id="live-failure",
|
||||
),
|
||||
pytest.param(
|
||||
NotificationType.TITLE_CHANGE,
|
||||
1,
|
||||
0,
|
||||
['result="success",type="title_change"} 1.0'],
|
||||
id="title-change-success",
|
||||
),
|
||||
pytest.param(
|
||||
NotificationType.CLEANUP_WARNING,
|
||||
2,
|
||||
0,
|
||||
['result="success",type="cleanup_warning"} 2.0'],
|
||||
id="cleanup-warning-success",
|
||||
),
|
||||
pytest.param(
|
||||
NotificationType.CLEANUP_DELETION,
|
||||
1,
|
||||
1,
|
||||
[
|
||||
'result="success",type="cleanup_deletion"} 1.0',
|
||||
'result="failure",type="cleanup_deletion"} 1.0',
|
||||
],
|
||||
id="cleanup-deletion-mixed",
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_records_delivery(
|
||||
self,
|
||||
notification_type: NotificationType,
|
||||
successful: int,
|
||||
failed: int,
|
||||
expected_fragments: list[str],
|
||||
) -> None:
|
||||
"""Record delivery results with correct type and result labels."""
|
||||
service = MetricsService()
|
||||
service.record_delivery(notification_type, successful=successful, failed=failed)
|
||||
output = generate_metrics_output(service)
|
||||
for fragment in expected_fragments:
|
||||
assert fragment in output
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("notification_type", "result"),
|
||||
[
|
||||
pytest.param(t, r, id=f"{t}-{r}")
|
||||
for t in NotificationType
|
||||
for r in ("success", "failure")
|
||||
],
|
||||
)
|
||||
def test_all_combinations_initialized(
|
||||
self, notification_type: NotificationType, result: str
|
||||
) -> None:
|
||||
"""All type/result label combinations exist at zero on init."""
|
||||
service = MetricsService()
|
||||
output = generate_metrics_output(service)
|
||||
expected = (
|
||||
f"owncastsentry_notification_delivery_total"
|
||||
f'{{result="{result}",type="{notification_type}"}} 0.0'
|
||||
)
|
||||
assert expected in output
|
||||
|
||||
|
||||
class TestSetStreamStatus:
|
||||
"""Per-stream status gauge."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("status", "expected_value"),
|
||||
[
|
||||
pytest.param(StreamStatus.ONLINE, 1.0, id="online"),
|
||||
pytest.param(StreamStatus.OFFLINE, 0.0, id="offline"),
|
||||
pytest.param(StreamStatus.UNKNOWN, -1.0, id="unknown"),
|
||||
],
|
||||
)
|
||||
def test_sets_status(self, status: StreamStatus, expected_value: float) -> None:
|
||||
"""Set gauge to the correct value for each stream status."""
|
||||
service = MetricsService()
|
||||
service.set_stream_status("test.com", status)
|
||||
output = generate_metrics_output(service)
|
||||
expected = f'owncastsentry_stream_status{{domain="test.com"}} {expected_value}'
|
||||
assert expected in output
|
||||
|
||||
|
||||
class TestSetSubscriptionCount:
|
||||
"""Per-stream subscription count gauge."""
|
||||
|
||||
def test_sets_count(self) -> None:
|
||||
"""Set the subscription count for a domain."""
|
||||
service = MetricsService()
|
||||
service.set_subscription_count("test.com", 5)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_stream_subscriptions{domain="test.com"} 5.0' in output
|
||||
|
||||
def test_updates_count(self) -> None:
|
||||
"""Update the subscription count for a domain."""
|
||||
service = MetricsService()
|
||||
service.set_subscription_count("test.com", 5)
|
||||
service.set_subscription_count("test.com", 3)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_stream_subscriptions{domain="test.com"} 3.0' in output
|
||||
|
||||
|
||||
class TestSetCheckFailures:
|
||||
"""Check failure counter gauge per domain."""
|
||||
|
||||
def test_sets_count(self) -> None:
|
||||
"""Set the failure count for a domain."""
|
||||
service = MetricsService()
|
||||
service.set_check_failures("fail.com", 3)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_check_failures{domain="fail.com"} 3.0' in output
|
||||
|
||||
def test_resets_to_zero(self) -> None:
|
||||
"""Reset the failure count to zero."""
|
||||
service = MetricsService()
|
||||
service.set_check_failures("fail.com", 5)
|
||||
service.set_check_failures("fail.com", 0)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_check_failures{domain="fail.com"} 0.0' in output
|
||||
|
||||
|
||||
class TestResponseTimer:
|
||||
"""Response time gauge via context manager."""
|
||||
|
||||
def test_records_on_success(self) -> None:
|
||||
"""Record a response time when success() is called."""
|
||||
service = MetricsService()
|
||||
with service.response_timer("example.com") as timer:
|
||||
timer.success()
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' in output
|
||||
|
||||
def test_does_not_record_without_success(self) -> None:
|
||||
"""Do not record when success() is never called."""
|
||||
service = MetricsService()
|
||||
with service.response_timer("example.com"):
|
||||
pass
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' not in output
|
||||
|
||||
def test_overwrites_previous_value(self) -> None:
|
||||
"""Overwrite previous value with the latest response time."""
|
||||
service = MetricsService()
|
||||
with service.response_timer("example.com") as timer:
|
||||
timer.success()
|
||||
with service.response_timer("example.com") as timer:
|
||||
timer.success()
|
||||
output = generate_metrics_output(service)
|
||||
# Gauge should have exactly one line for this domain, not accumulated
|
||||
matches = [
|
||||
line
|
||||
for line in output.splitlines()
|
||||
if line.startswith("owncastsentry_api_response_seconds{")
|
||||
]
|
||||
assert len(matches) == 1
|
||||
|
||||
def test_does_not_record_on_exception(self) -> None:
|
||||
"""Do not record when the block raises an exception."""
|
||||
service = MetricsService()
|
||||
with (
|
||||
pytest.raises(ValueError, match="boom"),
|
||||
service.response_timer("example.com"),
|
||||
):
|
||||
raise ValueError("boom")
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' not in output
|
||||
|
||||
|
||||
class TestRemoveStream:
|
||||
"""Stale stream label cleanup."""
|
||||
|
||||
def test_removes_stream_label(self) -> None:
|
||||
"""Remove a stream's gauge labels after cleanup deletion."""
|
||||
service = MetricsService()
|
||||
service.set_stream_status("gone.com", StreamStatus.OFFLINE)
|
||||
service.set_subscription_count("gone.com", 2)
|
||||
assert 'domain="gone.com"' in generate_metrics_output(service)
|
||||
service.remove_stream("gone.com")
|
||||
assert 'domain="gone.com"' not in generate_metrics_output(service)
|
||||
|
||||
def test_remove_nonexistent_is_noop(self) -> None:
|
||||
"""Removing a nonexistent stream does not raise."""
|
||||
service = MetricsService()
|
||||
service.remove_stream("never.com")
|
||||
|
||||
|
||||
class TestRegisterOpenConnectionsGauge:
|
||||
"""Callback-based open connection gauge."""
|
||||
|
||||
def test_reads_value_from_callback(self) -> None:
|
||||
"""Read the open connection count from the callback at scrape time."""
|
||||
service = MetricsService()
|
||||
counter = [3]
|
||||
service.register_open_connections_gauge(lambda: counter[0])
|
||||
output = generate_metrics_output(service)
|
||||
assert "owncastsentry_http_connections_open 3.0" in output
|
||||
|
||||
def test_reflects_updated_value(self) -> None:
|
||||
"""Reflect changes in the callback value on subsequent scrapes."""
|
||||
service = MetricsService()
|
||||
counter = [1]
|
||||
service.register_open_connections_gauge(lambda: counter[0])
|
||||
counter[0] = 5
|
||||
output = generate_metrics_output(service)
|
||||
assert "owncastsentry_http_connections_open 5.0" in output
|
||||
|
||||
|
||||
class TestSetBuildInfo:
|
||||
"""Build version info metric."""
|
||||
|
||||
def test_sets_version(self) -> None:
|
||||
"""Set the build version info."""
|
||||
service = MetricsService()
|
||||
service.set_build_info("1.2.3")
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_info{version="1.2.3"} 1.0' in output
|
||||
|
||||
|
||||
class TestRecordError:
|
||||
"""Internal error counter."""
|
||||
|
||||
def test_increments_counter(self) -> None:
|
||||
"""Increment the error counter for a source."""
|
||||
service = MetricsService()
|
||||
service.record_error(ErrorSource.SCHEDULER_LOOP)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_errors_total{source="scheduler_loop"} 1.0' in output
|
||||
|
||||
def test_increments_multiple_sources(self) -> None:
|
||||
"""Increment error counters for different sources independently."""
|
||||
service = MetricsService()
|
||||
service.record_error(ErrorSource.SCHEDULER_LOOP)
|
||||
service.record_error(ErrorSource.COMMAND)
|
||||
service.record_error(ErrorSource.COMMAND)
|
||||
output = generate_metrics_output(service)
|
||||
assert 'owncastsentry_errors_total{source="scheduler_loop"} 1.0' in output
|
||||
assert 'owncastsentry_errors_total{source="command"} 2.0' in output
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"source",
|
||||
[pytest.param(s, id=s) for s in ErrorSource],
|
||||
)
|
||||
def test_all_sources_initialized(self, source: ErrorSource) -> None:
|
||||
"""All known source labels exist at zero on init."""
|
||||
service = MetricsService()
|
||||
output = generate_metrics_output(service)
|
||||
expected = f'owncastsentry_errors_total{{source="{source}"}} 0.0'
|
||||
assert expected in output
|
||||
|
||||
|
||||
class TestRegistryOutput:
|
||||
"""Prometheus registry output."""
|
||||
|
||||
def test_returns_string(self) -> None:
|
||||
"""Return a string (not bytes)."""
|
||||
service = MetricsService()
|
||||
output = generate_metrics_output(service)
|
||||
assert isinstance(output, str)
|
||||
|
||||
def test_contains_help_lines(self) -> None:
|
||||
"""Include HELP lines for registered metrics."""
|
||||
service = MetricsService()
|
||||
output = generate_metrics_output(service)
|
||||
assert "# HELP owncastsentry_notification_delivery_total" in output
|
||||
assert "# HELP owncastsentry_errors_total" in output
|
||||
assert "# HELP owncastsentry_info" in output
|
||||
|
||||
def test_uses_isolated_registry(self) -> None:
|
||||
"""Use a custom registry, not the global default."""
|
||||
service = MetricsService()
|
||||
output = generate_metrics_output(service)
|
||||
assert "python_gc" not in output
|
||||
assert "process_" not in output
|
||||
@@ -14,8 +14,6 @@
|
||||
|
||||
"""Tests for data models."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from owncastsentry.models import StreamConfig, StreamState, StreamStatus
|
||||
|
||||
@@ -14,17 +14,16 @@
|
||||
|
||||
"""Tests for the notification service."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
import pytest
|
||||
|
||||
from owncastsentry.metrics import MetricsService
|
||||
from owncastsentry.notification_service import NotificationService
|
||||
from owncastsentry.utils import SECONDS_BETWEEN_NOTIFICATIONS
|
||||
from tests.conftest import _StubMatrixClient
|
||||
from tests.conftest import _StubMatrixClient, generate_metrics_output
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from owncastsentry.database import StreamRepository, SubscriptionRepository
|
||||
@@ -34,12 +33,14 @@ def _make_service(
|
||||
*,
|
||||
client: _StubMatrixClient,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
metrics: MetricsService | None = None,
|
||||
) -> NotificationService:
|
||||
"""Build a NotificationService with a stub client and real repo."""
|
||||
return NotificationService(
|
||||
client=client,
|
||||
subscription_repo=subscription_repo,
|
||||
logger=logging.getLogger("test"),
|
||||
metrics=metrics or MetricsService(),
|
||||
)
|
||||
|
||||
|
||||
@@ -346,3 +347,111 @@ class TestSendCleanupDeletion:
|
||||
"If the instance comes online again and you want to "
|
||||
"resubscribe, run `!subscribe example.com`."
|
||||
)
|
||||
|
||||
|
||||
class TestNotificationMetrics:
|
||||
"""Notification metrics recording."""
|
||||
|
||||
async def test_records_live_notification(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record a live notification metric."""
|
||||
client = _StubMatrixClient()
|
||||
metrics = MetricsService()
|
||||
service = _make_service(
|
||||
client=client, subscription_repo=subscription_repo, metrics=metrics
|
||||
)
|
||||
await stream_repo.create("example.com")
|
||||
await subscription_repo.add("example.com", "!room:matrix.org")
|
||||
await service.notify_stream_live("example.com", "Stream", "Title", [])
|
||||
output = generate_metrics_output(metrics)
|
||||
expected = (
|
||||
"owncastsentry_notification_delivery_total"
|
||||
'{result="success",type="live"} 1.0'
|
||||
)
|
||||
assert expected in output
|
||||
|
||||
async def test_records_title_change_notification(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record a title_change notification metric."""
|
||||
client = _StubMatrixClient()
|
||||
metrics = MetricsService()
|
||||
service = _make_service(
|
||||
client=client, subscription_repo=subscription_repo, metrics=metrics
|
||||
)
|
||||
await stream_repo.create("example.com")
|
||||
await subscription_repo.add("example.com", "!room:matrix.org")
|
||||
await service.notify_stream_live(
|
||||
"example.com", "Stream", "Title", [], title_change=True
|
||||
)
|
||||
output = generate_metrics_output(metrics)
|
||||
expected = (
|
||||
"owncastsentry_notification_delivery_total"
|
||||
'{result="success",type="title_change"} 1.0'
|
||||
)
|
||||
assert expected in output
|
||||
|
||||
async def test_records_cleanup_warning_notification(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record a cleanup_warning notification metric."""
|
||||
client = _StubMatrixClient()
|
||||
metrics = MetricsService()
|
||||
service = _make_service(
|
||||
client=client, subscription_repo=subscription_repo, metrics=metrics
|
||||
)
|
||||
await stream_repo.create("example.com")
|
||||
await subscription_repo.add("example.com", "!room:matrix.org")
|
||||
await service.send_cleanup_warning("example.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
expected = (
|
||||
"owncastsentry_notification_delivery_total"
|
||||
'{result="success",type="cleanup_warning"} 1.0'
|
||||
)
|
||||
assert expected in output
|
||||
|
||||
async def test_records_cleanup_deletion_notification(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record a cleanup_deletion notification metric."""
|
||||
client = _StubMatrixClient()
|
||||
metrics = MetricsService()
|
||||
service = _make_service(
|
||||
client=client, subscription_repo=subscription_repo, metrics=metrics
|
||||
)
|
||||
await stream_repo.create("example.com")
|
||||
await subscription_repo.add("example.com", "!room:matrix.org")
|
||||
await service.send_cleanup_deletion("example.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
expected = (
|
||||
"owncastsentry_notification_delivery_total"
|
||||
'{result="success",type="cleanup_deletion"} 1.0'
|
||||
)
|
||||
assert expected in output
|
||||
|
||||
async def test_no_metric_when_rate_limited(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Do not record metric when notification is rate-limited."""
|
||||
client = _StubMatrixClient()
|
||||
metrics = MetricsService()
|
||||
service = _make_service(
|
||||
client=client, subscription_repo=subscription_repo, metrics=metrics
|
||||
)
|
||||
service.notification_timers_cache["example.com"] = time.monotonic()
|
||||
await stream_repo.create("example.com")
|
||||
await subscription_repo.add("example.com", "!room:matrix.org")
|
||||
await service.notify_stream_live("example.com", "Stream", "Title", [])
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'result="success",type="live"} 0.0' in output
|
||||
|
||||
@@ -14,8 +14,6 @@
|
||||
|
||||
"""Tests for the Owncast HTTP client."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from typing import TYPE_CHECKING
|
||||
@@ -23,8 +21,13 @@ from typing import TYPE_CHECKING
|
||||
import pytest
|
||||
from aioresponses import aioresponses
|
||||
|
||||
from owncastsentry.metrics import MetricsService
|
||||
from owncastsentry.owncast_client import OwncastClient
|
||||
from tests.conftest import VALID_CONFIG_RESPONSE, VALID_STATUS_RESPONSE
|
||||
from tests.conftest import (
|
||||
VALID_CONFIG_RESPONSE,
|
||||
VALID_STATUS_RESPONSE,
|
||||
generate_metrics_output,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
@@ -33,7 +36,11 @@ if TYPE_CHECKING:
|
||||
@pytest.fixture
|
||||
async def owncast_client() -> AsyncIterator[OwncastClient]:
|
||||
"""Create an OwncastClient and close it after the test."""
|
||||
client = OwncastClient(logger=logging.getLogger("test"), version="0.0.0")
|
||||
client = OwncastClient(
|
||||
logger=logging.getLogger("test"),
|
||||
version="0.0.0",
|
||||
metrics=MetricsService(),
|
||||
)
|
||||
yield client
|
||||
await client.close()
|
||||
|
||||
@@ -205,3 +212,69 @@ class TestValidateInstance:
|
||||
result = await owncast_client.validate_instance("invalid.com")
|
||||
|
||||
assert result is False
|
||||
|
||||
|
||||
class TestResponseTimeMetrics:
|
||||
"""Response time histogram recording."""
|
||||
|
||||
async def test_records_on_success(self) -> None:
|
||||
"""Record response time on a successful request."""
|
||||
metrics = MetricsService()
|
||||
client = OwncastClient(
|
||||
logger=logging.getLogger("test"),
|
||||
version="0.0.0",
|
||||
metrics=metrics,
|
||||
)
|
||||
with aioresponses() as mocked:
|
||||
mocked.get(
|
||||
"https://example.com/api/status",
|
||||
body=json.dumps(VALID_STATUS_RESPONSE).encode(),
|
||||
)
|
||||
await client.get_stream_state("example.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' in output
|
||||
await client.close()
|
||||
|
||||
async def test_no_observation_on_failure(self) -> None:
|
||||
"""Do not record response time when request fails."""
|
||||
metrics = MetricsService()
|
||||
client = OwncastClient(
|
||||
logger=logging.getLogger("test"),
|
||||
version="0.0.0",
|
||||
metrics=metrics,
|
||||
)
|
||||
with aioresponses() as mocked:
|
||||
mocked.get(
|
||||
"https://example.com/api/status",
|
||||
status=500,
|
||||
)
|
||||
await client.get_stream_state("example.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' not in output
|
||||
await client.close()
|
||||
|
||||
async def test_no_observation_on_connection_error(self) -> None:
|
||||
"""Do not record response time on connection error."""
|
||||
metrics = MetricsService()
|
||||
client = OwncastClient(
|
||||
logger=logging.getLogger("test"),
|
||||
version="0.0.0",
|
||||
metrics=metrics,
|
||||
)
|
||||
with aioresponses() as mocked:
|
||||
mocked.get(
|
||||
"https://example.com/api/status",
|
||||
exception=ConnectionError(),
|
||||
)
|
||||
await client.get_stream_state("example.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_api_response_seconds{domain="example.com"}' not in output
|
||||
await client.close()
|
||||
|
||||
|
||||
class TestOpenConnectionCount:
|
||||
"""Open connection count."""
|
||||
|
||||
async def test_zero_with_no_requests(self, owncast_client: OwncastClient) -> None:
|
||||
"""Return zero when no requests have been made."""
|
||||
assert owncast_client.open_connection_count == 0
|
||||
|
||||
@@ -14,13 +14,12 @@
|
||||
|
||||
"""Tests for the stream monitor."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import time
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from owncastsentry.models import StreamConfig, StreamState
|
||||
from owncastsentry.metrics import MetricsService
|
||||
from owncastsentry.models import StreamConfig, StreamState, StreamStatus
|
||||
from owncastsentry.notification_service import NotificationService
|
||||
from owncastsentry.stream_monitor import StreamMonitor
|
||||
from owncastsentry.utils import (
|
||||
@@ -28,7 +27,11 @@ from owncastsentry.utils import (
|
||||
CLEANUP_WARNING_THRESHOLD,
|
||||
SECONDS_BETWEEN_NOTIFICATIONS,
|
||||
)
|
||||
from tests.conftest import _StubMatrixClient, _StubOwncastClient
|
||||
from tests.conftest import (
|
||||
_StubMatrixClient,
|
||||
_StubOwncastClient,
|
||||
generate_metrics_output,
|
||||
)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from owncastsentry.database import StreamRepository, SubscriptionRepository
|
||||
@@ -40,13 +43,16 @@ def _make_monitor(
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
client: _StubMatrixClient,
|
||||
metrics: MetricsService | None = None,
|
||||
) -> tuple[StreamMonitor, NotificationService]:
|
||||
"""Build a StreamMonitor with stubs and a real NotificationService."""
|
||||
logger = logging.getLogger("test")
|
||||
metrics = metrics or MetricsService()
|
||||
notification_service = NotificationService(
|
||||
client=client,
|
||||
subscription_repo=subscription_repo,
|
||||
logger=logger,
|
||||
metrics=metrics,
|
||||
)
|
||||
monitor = StreamMonitor(
|
||||
owncast_client=owncast_client,
|
||||
@@ -54,6 +60,7 @@ def _make_monitor(
|
||||
subscription_repo=subscription_repo,
|
||||
notification_service=notification_service,
|
||||
logger=logger,
|
||||
metrics=metrics,
|
||||
)
|
||||
return monitor, notification_service
|
||||
|
||||
@@ -82,6 +89,25 @@ async def _seed_stream(
|
||||
await subscription_repo.add(domain, room_id)
|
||||
|
||||
|
||||
def _make_monitor_with_metrics(
|
||||
*,
|
||||
owncast_client: _StubOwncastClient,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
client: _StubMatrixClient,
|
||||
) -> tuple[StreamMonitor, NotificationService, MetricsService]:
|
||||
"""Build a StreamMonitor with stubs, a real NotificationService, and metrics."""
|
||||
metrics = MetricsService()
|
||||
monitor, notification_service = _make_monitor(
|
||||
owncast_client=owncast_client,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
metrics=metrics,
|
||||
)
|
||||
return monitor, notification_service, metrics
|
||||
|
||||
|
||||
class TestUpdateAllStreams:
|
||||
"""Parallel stream update orchestration."""
|
||||
|
||||
@@ -837,3 +863,178 @@ class TestUpdateAllStreamsMixed:
|
||||
assert result.total_streams == 2
|
||||
assert result.successful_checks == 1
|
||||
assert result.failed_checks == 1
|
||||
|
||||
|
||||
class TestStreamMonitorMetrics:
|
||||
"""Stream monitor metrics recording."""
|
||||
|
||||
async def test_records_stream_status_online(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record online stream status gauge."""
|
||||
owncast = _StubOwncastClient(
|
||||
stream_state=StreamState(
|
||||
domain="live.com",
|
||||
title="Title",
|
||||
last_connect_time="2026-01-01T12:00:00Z",
|
||||
last_disconnect_time="2026-01-01T10:00:00Z",
|
||||
),
|
||||
stream_config=StreamConfig(name="Live Stream"),
|
||||
)
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await _seed_stream(
|
||||
stream_repo,
|
||||
subscription_repo,
|
||||
domain="live.com",
|
||||
last_disconnect_time="2026-01-01T10:00:00Z",
|
||||
)
|
||||
monitor.offline_timer_cache["live.com"] = 0
|
||||
await monitor.update_stream("live.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_stream_status{domain="live.com"} 1.0' in output
|
||||
|
||||
async def test_records_stream_status_offline(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record offline stream status gauge."""
|
||||
owncast = _StubOwncastClient(
|
||||
stream_state=StreamState(
|
||||
domain="off.com",
|
||||
title="Title",
|
||||
last_disconnect_time="2026-01-01T12:00:00Z",
|
||||
),
|
||||
stream_config=StreamConfig(name="Off Stream"),
|
||||
)
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await _seed_stream(
|
||||
stream_repo,
|
||||
subscription_repo,
|
||||
domain="off.com",
|
||||
title="Title",
|
||||
last_disconnect_time="2026-01-01T12:00:00Z",
|
||||
)
|
||||
await monitor.update_stream("off.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_stream_status{domain="off.com"} 0.0' in output
|
||||
|
||||
async def test_records_check_failures_on_connection_failure(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record failure counter gauge when a stream check fails."""
|
||||
owncast = _StubOwncastClient(stream_state=None)
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await _seed_stream(
|
||||
stream_repo,
|
||||
subscription_repo,
|
||||
domain="fail.com",
|
||||
last_disconnect_time="2026-01-01T00:00:00Z",
|
||||
)
|
||||
await monitor.update_stream("fail.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_check_failures{domain="fail.com"} 1.0' in output
|
||||
|
||||
async def test_resets_check_failures_on_success(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Reset failure counter gauge to zero on successful check."""
|
||||
owncast = _StubOwncastClient(
|
||||
stream_state=StreamState(
|
||||
domain="recover.com",
|
||||
title="Title",
|
||||
last_disconnect_time="2026-01-01T12:00:00Z",
|
||||
),
|
||||
stream_config=StreamConfig(name="Recover"),
|
||||
)
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await _seed_stream(
|
||||
stream_repo,
|
||||
subscription_repo,
|
||||
domain="recover.com",
|
||||
last_disconnect_time="2026-01-01T12:00:00Z",
|
||||
)
|
||||
# Simulate prior failures
|
||||
for _ in range(3):
|
||||
await stream_repo.increment_failure_counter("recover.com")
|
||||
await monitor.update_stream("recover.com")
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_check_failures{domain="recover.com"} 0.0' in output
|
||||
|
||||
async def test_removes_stream_on_cleanup_deletion(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Remove stream gauge label on cleanup deletion."""
|
||||
owncast = _StubOwncastClient()
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await _seed_stream(stream_repo, subscription_repo, domain="delete.com")
|
||||
metrics.set_stream_status("delete.com", StreamStatus.OFFLINE)
|
||||
assert 'domain="delete.com"' in generate_metrics_output(metrics)
|
||||
|
||||
await monitor._check_cleanup_thresholds("delete.com", CLEANUP_DELETE_THRESHOLD)
|
||||
assert 'domain="delete.com"' not in generate_metrics_output(metrics)
|
||||
|
||||
async def test_records_subscription_counts(
|
||||
self,
|
||||
stream_repo: StreamRepository,
|
||||
subscription_repo: SubscriptionRepository,
|
||||
) -> None:
|
||||
"""Record per-domain subscription counts after update_all_streams."""
|
||||
owncast = _StubOwncastClient(
|
||||
stream_state=StreamState(
|
||||
domain="pop.com",
|
||||
last_connect_time="2026-01-01T00:00:00Z",
|
||||
),
|
||||
stream_config=StreamConfig(name="Popular"),
|
||||
)
|
||||
client = _StubMatrixClient()
|
||||
monitor, _, metrics = _make_monitor_with_metrics(
|
||||
owncast_client=owncast,
|
||||
stream_repo=stream_repo,
|
||||
subscription_repo=subscription_repo,
|
||||
client=client,
|
||||
)
|
||||
await stream_repo.create("pop.com")
|
||||
await subscription_repo.add("pop.com", "!room1:matrix.org")
|
||||
await subscription_repo.add("pop.com", "!room2:matrix.org")
|
||||
await monitor.update_all_streams(["pop.com"])
|
||||
output = generate_metrics_output(metrics)
|
||||
assert 'owncastsentry_stream_subscriptions{domain="pop.com"} 2.0' in output
|
||||
|
||||
@@ -14,8 +14,6 @@
|
||||
|
||||
"""Tests for utility functions and constants."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from owncastsentry.utils import (
|
||||
|
||||
@@ -924,6 +924,9 @@ wheels = [
|
||||
[[package]]
|
||||
name = "owncastsentry"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "prometheus-client" },
|
||||
]
|
||||
|
||||
[package.dev-dependencies]
|
||||
dev = [
|
||||
@@ -941,6 +944,7 @@ dev = [
|
||||
]
|
||||
|
||||
[package.metadata]
|
||||
requires-dist = [{ name = "prometheus-client", specifier = ">=0.24.1" }]
|
||||
|
||||
[package.metadata.requires-dev]
|
||||
dev = [
|
||||
@@ -1069,6 +1073,15 @@ wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/54/20/4d324d65cc6d9205fabedc306948156824eb9f0ee1633355a8f7ec5c66bf/pluggy-1.6.0-py3-none-any.whl", hash = "sha256:e920276dd6813095e9377c0bc5566d94c932c33b27a3e3945d8389c374dd4746", size = 20538, upload-time = "2025-05-15T12:30:06.134Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prometheus-client"
|
||||
version = "0.24.1"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/f0/58/a794d23feb6b00fc0c72787d7e87d872a6730dd9ed7c7b3e954637d8f280/prometheus_client-0.24.1.tar.gz", hash = "sha256:7e0ced7fbbd40f7b84962d5d2ab6f17ef88a72504dcf7c0b40737b43b2a461f9", size = 85616, upload-time = "2026-01-14T15:26:26.965Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/74/c3/24a2f845e3917201628ecaba4f18bab4d18a337834c1df2a159ee9d22a42/prometheus_client-0.24.1-py3-none-any.whl", hash = "sha256:150db128af71a5c2482b36e588fc8a6b95e498750da4b17065947c16070f4055", size = 64057, upload-time = "2026-01-14T15:26:24.42Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prompt-toolkit"
|
||||
version = "3.0.52"
|
||||
|
||||
Reference in New Issue
Block a user