diff --git a/.claude/settings.local.json b/.claude/settings.local.json index 07d2f268ff..7d752718f4 100644 --- a/.claude/settings.local.json +++ b/.claude/settings.local.json @@ -29,7 +29,13 @@ "mcp__serena__replace_regex", "mcp__archon__assess_code_quality", "Bash(agent-onex-coordinator)", - "mcp__archon__health_check" + "mcp__archon__health_check", + "mcp__serena__list_dir", + "mcp__serena__create_text_file", + "Read(//Volumes/PRO-G40/Code/omnibase_core/src/omnibase_core/enums/intelligence/**)", + "Read(//Volumes/PRO-G40/Code/omnibase_core/src/omnibase_core/models/metrics/**)", + "Read(//Volumes/PRO-G40/Code/omnibase_core/src/omnibase_core/models/core/**)", + "Read(//Volumes/PRO-G40/Code/omnibase_core/src/**)" ], "deny": [], "ask": [] diff --git a/pyproject.toml b/pyproject.toml index 67b90f6f10..df7aa9d2b1 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -47,6 +47,9 @@ rich = "^13.7.0" cryptography = "^41.0.0" jinja2 = "^3.1.0" +# Tracing and observability dependencies +sqlparse = "^0.4.4" # For secure SQL query sanitization in tracing + [tool.poetry.group.dev.dependencies] pytest = "^8.4.0" pytest-asyncio = "^0.25.0" diff --git a/src/omnibase_infra/infrastructure/container.py b/src/omnibase_infra/infrastructure/container.py index e25e26d235..1ec5f6dd25 100644 --- a/src/omnibase_infra/infrastructure/container.py +++ b/src/omnibase_infra/infrastructure/container.py @@ -19,7 +19,7 @@ import time from typing import Callable, Optional, Type, TypeVar, Union, Dict, List -from omnibase_core.core.onex_container import ModelONEXContainer as ONEXContainer +from omnibase_core.core.onex_container import ModelONEXContainer from omnibase_core.protocol.protocol_event_bus import ProtocolEventBus from omnibase_core.model.core.model_onex_event import ModelOnexEvent from omnibase_core.utils.generation.utility_schema_loader import UtilitySchemaLoader @@ -618,7 +618,7 @@ def _event_to_topic(self, event: ModelOnexEvent) -> str: return topic -def create_infrastructure_container() -> ONEXContainer: +def create_infrastructure_container() -> ModelONEXContainer: """ Create infrastructure container with all shared dependencies. @@ -628,10 +628,10 @@ def create_infrastructure_container() -> ONEXContainer: - "Everything needs to be resolved by duck typing" Returns: - Configured ONEXContainer with infrastructure dependencies + Configured ModelONEXContainer with infrastructure dependencies """ # Create base ONEX container - container = ONEXContainer() + container = ModelONEXContainer() # Set up all shared dependencies for infrastructure services _setup_infrastructure_dependencies(container) @@ -642,7 +642,7 @@ def create_infrastructure_container() -> ONEXContainer: return container -def _setup_infrastructure_dependencies(container: ONEXContainer): +def _setup_infrastructure_dependencies(container: ModelONEXContainer): """Set up all dependencies needed by infrastructure services.""" # Get logger for container setup @@ -687,13 +687,13 @@ def _setup_infrastructure_dependencies(container: ONEXContainer): logger.info(" PostgreSQL connection manager skipped (environment not configured)") -def _register_service(container: ONEXContainer, service_name: str, service_instance): +def _register_service(container: ModelONEXContainer, service_name: str, service_instance): """Register a service in the container for later retrieval.""" # Use the ONEX container's native service registration container.register_service(service_name, service_instance) -def _bind_infrastructure_get_service_method(container: ONEXContainer): +def _bind_infrastructure_get_service_method(container: ModelONEXContainer): """Configure infrastructure container with proper dependency injection.""" # The ModelONEXContainer should handle get_service natively # We just need to register our services properly in the container diff --git a/src/omnibase_infra/infrastructure/distributed_tracing.py b/src/omnibase_infra/infrastructure/distributed_tracing.py index fc506465fa..7ab447c120 100644 --- a/src/omnibase_infra/infrastructure/distributed_tracing.py +++ b/src/omnibase_infra/infrastructure/distributed_tracing.py @@ -12,7 +12,7 @@ import os import time from contextlib import asynccontextmanager -from typing import Dict, Any, Optional, Union, AsyncIterator +from typing import Optional, Union, AsyncIterator from uuid import UUID, uuid4 from datetime import datetime @@ -26,7 +26,7 @@ from opentelemetry.instrumentation.asyncpg import AsyncPGInstrumentor from opentelemetry.instrumentation.kafka import KafkaInstrumentor from opentelemetry.trace.status import Status, StatusCode - from opentelemetry.trace import Span, SpanKind + from opentelemetry.trace import Span, SpanKind, Tracer from opentelemetry.context import Context OPENTELEMETRY_AVAILABLE = True @@ -39,6 +39,7 @@ from omnibase_core.core.errors.onex_error import OnexError from omnibase_core.core.errors.onex_error import CoreErrorCode from omnibase_core.model.core.model_onex_event import ModelOnexEvent +from omnibase_infra.models.tracing.model_span_attributes import ModelSpanAttributes from ..security.audit_logger import AuditLogger, AuditEvent, AuditEventType, AuditSeverity @@ -122,7 +123,7 @@ def __init__(self, config: Optional[TracingConfiguration] = None): # Tracing components self.tracer_provider: Optional[TracerProvider] = None - self.tracer: Optional[Any] = None # OpenTelemetry tracer + self.tracer: Optional[Tracer] = None # OpenTelemetry tracer self.is_initialized = False # Integration with audit logging @@ -208,7 +209,7 @@ async def trace_operation( correlation_id: Optional[Union[str, UUID]] = None, parent_context: Optional[Context] = None, span_kind: Optional[SpanKind] = SpanKind.INTERNAL, - attributes: Optional[Dict[str, Any]] = None + attributes: Optional[ModelSpanAttributes] = None ) -> AsyncIterator[Span]: """ Create a trace span for an operation with automatic error handling. @@ -239,15 +240,21 @@ async def trace_operation( try: # Create span + base_attributes = { + "correlation_id": correlation_str, + "environment": self.config.environment, + "service.name": self.config.service_name, + } + + # Add provided attributes if available + if attributes: + attribute_dict = attributes.dict(exclude_none=True) + base_attributes.update(attribute_dict) + span = self.tracer.start_span( name=operation_name, kind=span_kind, - attributes={ - "correlation_id": correlation_str, - "environment": self.config.environment, - "service.name": self.config.service_name, - **(attributes or {}) - } + attributes=base_attributes ) # Set correlation ID in baggage for propagation @@ -285,12 +292,12 @@ async def trace_operation( if token: context.detach(token) - def _create_noop_span(self) -> Any: + def _create_noop_span(self) -> object: """Create a no-op span when tracing is disabled.""" class NoOpSpan: - def set_attribute(self, key: str, value: Any) -> None: + def set_attribute(self, key: str, value: Union[str, int, float, bool]) -> None: pass - def set_status(self, status: Any) -> None: + def set_status(self, status: object) -> None: pass def record_exception(self, exception: Exception) -> None: pass diff --git a/src/omnibase_infra/models/circuit_breaker/__init__.py b/src/omnibase_infra/models/circuit_breaker/__init__.py new file mode 100644 index 0000000000..fcdaf23e1d --- /dev/null +++ b/src/omnibase_infra/models/circuit_breaker/__init__.py @@ -0,0 +1,5 @@ +"""Circuit Breaker Models Package. + +Shared models for circuit breaker operations and configurations. +Used by circuit breaker nodes and related infrastructure components. +""" \ No newline at end of file diff --git a/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_metrics.py b/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_metrics.py new file mode 100644 index 0000000000..66c3012871 --- /dev/null +++ b/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_metrics.py @@ -0,0 +1,89 @@ +"""Circuit Breaker Metrics Model. + +Shared model for circuit breaker metrics and performance data. +Used across circuit breaker nodes and observability systems. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime + + +class ModelCircuitBreakerMetrics(BaseModel): + """Model for circuit breaker metrics tracking.""" + + total_events: int = Field( + default=0, + ge=0, + description="Total number of events processed" + ) + + successful_events: int = Field( + default=0, + ge=0, + description="Number of successfully processed events" + ) + + failed_events: int = Field( + default=0, + ge=0, + description="Number of failed events" + ) + + queued_events: int = Field( + default=0, + ge=0, + description="Number of events currently queued" + ) + + dropped_events: int = Field( + default=0, + ge=0, + description="Number of events dropped due to capacity limits" + ) + + dead_letter_events: int = Field( + default=0, + ge=0, + description="Number of events in dead letter queue" + ) + + circuit_opens: int = Field( + default=0, + ge=0, + description="Number of times circuit has opened" + ) + + circuit_closes: int = Field( + default=0, + ge=0, + description="Number of times circuit has closed" + ) + + last_failure: Optional[datetime] = Field( + default=None, + description="Timestamp of last failure" + ) + + last_success: Optional[datetime] = Field( + default=None, + description="Timestamp of last success" + ) + + success_rate_percent: float = Field( + default=100.0, + ge=0.0, + le=100.0, + description="Success rate percentage" + ) + + average_response_time_ms: float = Field( + default=0.0, + ge=0.0, + description="Average response time in milliseconds" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_result.py b/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_result.py new file mode 100644 index 0000000000..2c13c91388 --- /dev/null +++ b/src/omnibase_infra/models/circuit_breaker/model_circuit_breaker_result.py @@ -0,0 +1,173 @@ +"""Circuit Breaker Operation Results Models. + +Strongly-typed models for different circuit breaker operation results. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List +from datetime import datetime +from uuid import UUID +from omnibase_core.enums.intelligence.enum_circuit_breaker_state import EnumCircuitBreakerState + + +class ModelPublishEventResult(BaseModel): + """Result for publish event operations.""" + + event_published: bool = Field( + description="Whether the event was successfully published" + ) + + publisher_function: Optional[str] = Field( + default=None, + max_length=200, + description="Name of the publisher function used" + ) + + publish_latency_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Time taken to publish event in milliseconds" + ) + + circuit_action_taken: str = Field( + pattern="^(published|queued|dropped|rejected)$", + description="Action taken by circuit breaker" + ) + + queue_length_after: Optional[int] = Field( + default=None, + ge=0, + description="Length of event queue after operation" + ) + + dead_letter_queued: bool = Field( + default=False, + description="Whether event was moved to dead letter queue" + ) + + +class ModelStateResult(BaseModel): + """Result for get state operations.""" + + current_state: EnumCircuitBreakerState = Field( + description="Current circuit breaker state" + ) + + failure_count: int = Field( + ge=0, + description="Current failure count" + ) + + success_count: int = Field( + ge=0, + description="Current success count" + ) + + last_failure_time: Optional[datetime] = Field( + default=None, + description="Timestamp of last failure" + ) + + time_in_current_state_seconds: float = Field( + ge=0.0, + description="How long in current state (seconds)" + ) + + next_state_transition_estimate: Optional[datetime] = Field( + default=None, + description="Estimated time of next state transition" + ) + + +class ModelResetResult(BaseModel): + """Result for reset circuit operations.""" + + reset_successful: bool = Field( + description="Whether the reset was successful" + ) + + previous_state: EnumCircuitBreakerState = Field( + description="Circuit breaker state before reset" + ) + + new_state: EnumCircuitBreakerState = Field( + description="Circuit breaker state after reset" + ) + + metrics_reset: bool = Field( + description="Whether metrics were also reset" + ) + + events_cleared_from_queue: int = Field( + ge=0, + description="Number of events cleared from queue" + ) + + dead_letter_queue_cleared: bool = Field( + description="Whether dead letter queue was cleared" + ) + + +class ModelHealthStatusResult(BaseModel): + """Result for health status operations.""" + + is_healthy: bool = Field( + description="Whether the circuit breaker is healthy" + ) + + health_score: float = Field( + ge=0.0, + le=100.0, + description="Health score (0-100)" + ) + + circuit_availability_percent: float = Field( + ge=0.0, + le=100.0, + description="Circuit availability percentage" + ) + + avg_response_time_ms: float = Field( + ge=0.0, + description="Average response time in milliseconds" + ) + + error_rate_percent: float = Field( + ge=0.0, + le=100.0, + description="Current error rate percentage" + ) + + queue_utilization_percent: float = Field( + ge=0.0, + le=100.0, + description="Queue utilization percentage" + ) + + uptime_seconds: float = Field( + ge=0.0, + description="Circuit breaker uptime in seconds" + ) + + issues_detected: List[str] = Field( + default_factory=list, + max_items=20, + description="List of issues detected with the circuit breaker" + ) + + recommendations: List[str] = Field( + default_factory=list, + max_items=10, + description="Health improvement recommendations" + ) + + last_health_check: datetime = Field( + description="Timestamp of last health check" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/circuit_breaker/model_dead_letter_queue_entry.py b/src/omnibase_infra/models/circuit_breaker/model_dead_letter_queue_entry.py new file mode 100644 index 0000000000..2cb54d5625 --- /dev/null +++ b/src/omnibase_infra/models/circuit_breaker/model_dead_letter_queue_entry.py @@ -0,0 +1,150 @@ +"""Dead Letter Queue Entry Model. + +Strongly-typed model for circuit breaker dead letter queue entries. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime +from uuid import UUID +from omnibase_core.model.core.model_onex_event import ModelOnexEvent + + +class ModelDeadLetterQueueEntry(BaseModel): + """Model for circuit breaker dead letter queue entries.""" + + # Entry identification + entry_id: UUID = Field( + description="Unique identifier for this dead letter queue entry" + ) + + # Original event + original_event: ModelOnexEvent = Field( + description="The original event that failed processing" + ) + + # Failure information + failure_timestamp: datetime = Field( + description="When the event failed and was queued" + ) + + failure_reason: str = Field( + max_length=500, + description="Reason why the event failed" + ) + + error_type: Optional[str] = Field( + default=None, + max_length=100, + description="Type/class of error that occurred" + ) + + error_message: Optional[str] = Field( + default=None, + max_length=1000, + description="Detailed error message" + ) + + # Retry information + retry_count: int = Field( + default=0, + ge=0, + le=10, + description="Number of times processing has been retried" + ) + + max_retries: int = Field( + default=3, + ge=0, + le=10, + description="Maximum number of retries allowed" + ) + + next_retry_at: Optional[datetime] = Field( + default=None, + description="When the next retry should be attempted" + ) + + last_retry_at: Optional[datetime] = Field( + default=None, + description="When the last retry was attempted" + ) + + # Circuit breaker context + circuit_breaker_state_when_failed: str = Field( + pattern="^(CLOSED|HALF_OPEN|OPEN)$", + description="Circuit breaker state when the event failed" + ) + + failure_count_when_failed: int = Field( + ge=0, + description="Circuit breaker failure count when this event failed" + ) + + # Processing context + original_publisher_function: Optional[str] = Field( + default=None, + max_length=200, + description="Name of the publisher function that originally failed" + ) + + processing_timeout_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Timeout that was applied when processing failed (milliseconds)" + ) + + # Queue management + queue_position: Optional[int] = Field( + default=None, + ge=0, + description="Position in the dead letter queue (for ordering)" + ) + + expires_at: Optional[datetime] = Field( + default=None, + description="When this entry expires and should be removed from queue" + ) + + # Resolution tracking + resolved: bool = Field( + default=False, + description="Whether this entry has been successfully processed" + ) + + resolved_at: Optional[datetime] = Field( + default=None, + description="When this entry was successfully processed" + ) + + resolved_by: Optional[str] = Field( + default=None, + max_length=100, + description="How this entry was resolved (retry_success, manual_intervention, etc.)" + ) + + # Metadata + environment: Optional[str] = Field( + default=None, + max_length=50, + description="Environment where the failure occurred" + ) + + service_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the service when failure occurred" + ) + + additional_context: Optional[str] = Field( + default=None, + max_length=1000, + description="Additional context about the failure" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/__init__.py b/src/omnibase_infra/models/health/__init__.py new file mode 100644 index 0000000000..2d03d84be7 --- /dev/null +++ b/src/omnibase_infra/models/health/__init__.py @@ -0,0 +1,5 @@ +"""Health Models Package. + +Shared models for health monitoring and infrastructure status. +Used by health monitoring nodes and related infrastructure components. +""" \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_component_status.py b/src/omnibase_infra/models/health/model_component_status.py new file mode 100644 index 0000000000..c4f4b85261 --- /dev/null +++ b/src/omnibase_infra/models/health/model_component_status.py @@ -0,0 +1,134 @@ +"""Component Status Model. + +Strongly-typed model for individual component health statuses. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime + + +class ModelComponentHealthStatus(BaseModel): + """Model for individual component health status.""" + + component_name: str = Field( + description="Name of the component" + ) + + status: str = Field( + pattern="^(healthy|warning|critical|unknown|offline)$", + description="Current health status of the component" + ) + + last_check_timestamp: datetime = Field( + description="Timestamp of last health check" + ) + + # Health indicators + is_available: bool = Field( + description="Whether the component is available for requests" + ) + + response_time_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Average response time in milliseconds" + ) + + error_rate_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Error rate percentage" + ) + + # Resource metrics + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="CPU usage percentage" + ) + + memory_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Memory usage percentage" + ) + + disk_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Disk usage percentage" + ) + + # Connection metrics + active_connections: Optional[int] = Field( + default=None, + ge=0, + description="Number of active connections" + ) + + connection_pool_utilization: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Connection pool utilization percentage" + ) + + # Performance indicators + throughput_per_second: Optional[float] = Field( + default=None, + ge=0.0, + description="Operations or requests processed per second" + ) + + queue_length: Optional[int] = Field( + default=None, + ge=0, + description="Length of processing queue" + ) + + # Health check details + health_check_duration_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Time taken to complete health check in milliseconds" + ) + + consecutive_failures: int = Field( + default=0, + ge=0, + description="Number of consecutive health check failures" + ) + + consecutive_successes: int = Field( + default=0, + ge=0, + description="Number of consecutive health check successes" + ) + + # Status details + status_message: Optional[str] = Field( + default=None, + max_length=500, + description="Detailed status message or error description" + ) + + recovery_actions_available: bool = Field( + default=False, + description="Whether automatic recovery actions are available" + ) + + requires_manual_intervention: bool = Field( + default=False, + description="Whether manual intervention is required" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_consul_metrics.py b/src/omnibase_infra/models/health/model_consul_metrics.py new file mode 100644 index 0000000000..c2b09b3e1e --- /dev/null +++ b/src/omnibase_infra/models/health/model_consul_metrics.py @@ -0,0 +1,152 @@ +"""Consul Metrics Model. + +Strongly-typed model for Consul service discovery health metrics. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional + + +class ModelConsulMetrics(BaseModel): + """Model for Consul service discovery health metrics.""" + + # Service registry metrics + registered_services: int = Field( + ge=0, + description="Number of registered services" + ) + + healthy_services: int = Field( + ge=0, + description="Number of services reporting healthy status" + ) + + unhealthy_services: int = Field( + ge=0, + description="Number of services reporting unhealthy status" + ) + + service_health_check_success_rate: float = Field( + ge=0.0, + le=100.0, + description="Service health check success rate percentage" + ) + + # Key-Value store metrics + kv_operations_per_second: float = Field( + ge=0.0, + description="Key-value operations per second" + ) + + kv_read_latency_ms: float = Field( + ge=0.0, + description="Average key-value read latency in milliseconds" + ) + + kv_write_latency_ms: float = Field( + ge=0.0, + description="Average key-value write latency in milliseconds" + ) + + kv_store_size_mb: float = Field( + ge=0.0, + description="Key-value store size in megabytes" + ) + + # Cluster metrics + cluster_nodes: int = Field( + ge=1, + description="Number of nodes in Consul cluster" + ) + + leader_elected: bool = Field( + description="Whether cluster has an elected leader" + ) + + raft_commits_per_second: float = Field( + ge=0.0, + description="Raft log commits per second" + ) + + raft_log_size_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Raft log size in megabytes" + ) + + # Connection metrics + client_connections: int = Field( + ge=0, + description="Number of active client connections" + ) + + api_request_rate: float = Field( + ge=0.0, + description="API requests per second" + ) + + api_error_rate: float = Field( + ge=0.0, + le=100.0, + description="API error rate percentage" + ) + + # Performance metrics + dns_queries_per_second: Optional[float] = Field( + default=None, + ge=0.0, + description="DNS queries handled per second" + ) + + catalog_operations_per_second: float = Field( + ge=0.0, + description="Service catalog operations per second" + ) + + # Resource utilization + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Consul agent memory usage in megabytes" + ) + + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Consul agent CPU usage percentage" + ) + + # Health check metrics + health_checks_total: int = Field( + ge=0, + description="Total number of health checks registered" + ) + + health_checks_passing: int = Field( + ge=0, + description="Number of health checks currently passing" + ) + + health_checks_failing: int = Field( + ge=0, + description="Number of health checks currently failing" + ) + + avg_health_check_duration_ms: float = Field( + ge=0.0, + description="Average health check execution time in milliseconds" + ) + + # Network metrics + gossip_messages_per_second: float = Field( + ge=0.0, + description="Gossip protocol messages per second" + ) + + network_latency_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Average network latency between nodes in milliseconds" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_alert.py b/src/omnibase_infra/models/health/model_health_alert.py new file mode 100644 index 0000000000..b6d5dd2881 --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_alert.py @@ -0,0 +1,138 @@ +"""Health Alert Model. + +Strongly-typed model for health monitoring alerts. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime +from uuid import UUID + + +class ModelHealthAlert(BaseModel): + """Model for health monitoring alerts.""" + + alert_id: UUID = Field( + description="Unique identifier for this alert" + ) + + component_name: str = Field( + description="Name of the component that triggered the alert" + ) + + alert_type: str = Field( + pattern="^(performance|availability|resource|security|configuration)$", + description="Type of alert triggered" + ) + + severity: str = Field( + pattern="^(low|medium|high|critical)$", + description="Alert severity level" + ) + + status: str = Field( + pattern="^(active|acknowledged|resolved|suppressed)$", + description="Current status of the alert" + ) + + # Timing information + triggered_at: datetime = Field( + description="When the alert was first triggered" + ) + + acknowledged_at: Optional[datetime] = Field( + default=None, + description="When the alert was acknowledged" + ) + + resolved_at: Optional[datetime] = Field( + default=None, + description="When the alert was resolved" + ) + + # Alert details + title: str = Field( + max_length=200, + description="Brief alert title" + ) + + description: str = Field( + max_length=1000, + description="Detailed alert description" + ) + + # Threshold information + threshold_value: Optional[float] = Field( + default=None, + description="The threshold value that was breached" + ) + + current_value: Optional[float] = Field( + default=None, + description="The current value that triggered the alert" + ) + + metric_name: Optional[str] = Field( + default=None, + description="Name of the metric that triggered the alert" + ) + + metric_unit: Optional[str] = Field( + default=None, + description="Unit of measurement for the metric" + ) + + # Impact assessment + impact_level: str = Field( + pattern="^(none|low|medium|high|severe)$", + description="Assessed impact level of the issue" + ) + + affected_users_estimate: Optional[int] = Field( + default=None, + ge=0, + description="Estimated number of affected users" + ) + + # Response information + auto_resolve_available: bool = Field( + default=False, + description="Whether automatic resolution is available" + ) + + escalation_required: bool = Field( + default=False, + description="Whether escalation to human operators is required" + ) + + runbook_url: Optional[str] = Field( + default=None, + max_length=500, + description="URL to relevant runbook or documentation" + ) + + # Tracking information + acknowledged_by: Optional[str] = Field( + default=None, + max_length=100, + description="Username or system that acknowledged the alert" + ) + + resolved_by: Optional[str] = Field( + default=None, + max_length=100, + description="Username or system that resolved the alert" + ) + + resolution_notes: Optional[str] = Field( + default=None, + max_length=1000, + description="Notes about how the alert was resolved" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_details.py b/src/omnibase_infra/models/health/model_health_details.py new file mode 100644 index 0000000000..f282669185 --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_details.py @@ -0,0 +1,145 @@ +"""Health Status Details Model. + +Strongly-typed model for additional health status details. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelHealthDetails(BaseModel): + """Model for additional health status details.""" + + # Component-specific details + postgres_connection_count: Optional[int] = Field( + default=None, + ge=0, + description="Current PostgreSQL connection count" + ) + + postgres_last_error: Optional[str] = Field( + default=None, + max_length=500, + description="Last PostgreSQL error message" + ) + + kafka_producer_count: Optional[int] = Field( + default=None, + ge=0, + description="Current Kafka producer count" + ) + + kafka_last_error: Optional[str] = Field( + default=None, + max_length=500, + description="Last Kafka error message" + ) + + circuit_breaker_state: Optional[str] = Field( + default=None, + pattern="^(CLOSED|HALF_OPEN|OPEN)$", + description="Current circuit breaker state" + ) + + circuit_breaker_failure_count: Optional[int] = Field( + default=None, + ge=0, + description="Circuit breaker failure count" + ) + + # Performance indicators + peak_memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Peak memory usage in megabytes" + ) + + average_cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Average CPU usage percentage" + ) + + disk_space_available_gb: Optional[float] = Field( + default=None, + ge=0.0, + description="Available disk space in gigabytes" + ) + + # Network and connectivity + network_latency_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Average network latency in milliseconds" + ) + + external_service_count: Optional[int] = Field( + default=None, + ge=0, + description="Number of external services being monitored" + ) + + external_services_healthy: Optional[int] = Field( + default=None, + ge=0, + description="Number of external services reporting healthy" + ) + + # Configuration and environment + environment_variables_loaded: Optional[int] = Field( + default=None, + ge=0, + description="Number of environment variables loaded" + ) + + configuration_files_loaded: Optional[int] = Field( + default=None, + ge=0, + description="Number of configuration files loaded" + ) + + # Security indicators + ssl_certificates_valid: Optional[bool] = Field( + default=None, + description="Whether SSL certificates are valid" + ) + + ssl_certificates_expire_days: Optional[int] = Field( + default=None, + ge=0, + description="Days until SSL certificates expire" + ) + + # Error tracking + recent_errors: Optional[List[str]] = Field( + default=None, + max_items=10, + description="List of recent error messages (last 10)" + ) + + warning_messages: Optional[List[str]] = Field( + default=None, + max_items=10, + description="List of recent warning messages (last 10)" + ) + + # Health check specifics + health_check_duration_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Time taken to complete health check in milliseconds" + ) + + components_checked: Optional[int] = Field( + default=None, + ge=0, + description="Number of components included in health check" + ) + + components_healthy: Optional[int] = Field( + default=None, + ge=0, + description="Number of components reporting healthy" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_metrics.py b/src/omnibase_infra/models/health/model_health_metrics.py new file mode 100644 index 0000000000..cd02812ab4 --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_metrics.py @@ -0,0 +1,115 @@ +"""Health Metrics Model. + +Shared model for infrastructure health metrics and performance data. +Used across health monitoring nodes for aggregated metrics. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime +from .model_postgres_metrics import ModelPostgresMetrics +from .model_kafka_metrics import ModelKafkaMetrics +from .model_consul_metrics import ModelConsulMetrics +from .model_vault_metrics import ModelVaultMetrics +from omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics import ModelCircuitBreakerMetrics + + +class ModelHealthMetrics(BaseModel): + """Model for aggregated infrastructure health metrics.""" + + timestamp: datetime = Field( + description="Metrics collection timestamp" + ) + + environment: str = Field( + description="Environment where metrics were collected" + ) + + # Component-specific metrics + postgres_metrics: ModelPostgresMetrics = Field( + description="PostgreSQL component metrics" + ) + + kafka_metrics: ModelKafkaMetrics = Field( + description="Kafka component metrics" + ) + + circuit_breaker_metrics: ModelCircuitBreakerMetrics = Field( + description="Circuit breaker component metrics" + ) + + consul_metrics: Optional[ModelConsulMetrics] = Field( + default=None, + description="Consul service discovery metrics" + ) + + vault_metrics: Optional[ModelVaultMetrics] = Field( + default=None, + description="Vault secret management metrics" + ) + + # Aggregate statistics + total_connections: int = Field( + ge=0, + description="Total number of active connections" + ) + + total_messages_processed: int = Field( + ge=0, + description="Total number of messages processed" + ) + + total_events_queued: int = Field( + ge=0, + description="Total number of events currently queued" + ) + + error_rate_percent: float = Field( + ge=0.0, + le=100.0, + description="Overall error rate percentage" + ) + + # Performance indicators + avg_db_response_time_ms: float = Field( + ge=0.0, + description="Average database response time in milliseconds" + ) + + avg_kafka_throughput_mps: float = Field( + ge=0.0, + description="Average Kafka throughput in messages per second" + ) + + circuit_breaker_success_rate: float = Field( + ge=0.0, + le=100.0, + description="Circuit breaker success rate percentage" + ) + + # Resource utilization + memory_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Memory usage percentage" + ) + + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="CPU usage percentage" + ) + + disk_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Disk usage percentage" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_request.py b/src/omnibase_infra/models/health/model_health_request.py new file mode 100644 index 0000000000..9efa7af2cb --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_request.py @@ -0,0 +1,71 @@ +"""Health Request Model. + +Shared model for health monitoring operation requests. +Used for health checks and monitoring control operations. +""" + +from pydantic import BaseModel, Field +from typing import List, Optional +from uuid import UUID +from datetime import datetime +from .model_request_context import ModelHealthRequestContext + + +class ModelHealthRequest(BaseModel): + """Model for health monitoring operation requests.""" + + operation_type: str = Field( + description="Type of health monitoring operation", + regex=r"^(health_check|get_metrics|get_trends|start_monitoring|stop_monitoring)$" + ) + + correlation_id: UUID = Field( + description="Request correlation ID for tracking" + ) + + timestamp: datetime = Field( + description="Request timestamp" + ) + + component_filters: Optional[List[str]] = Field( + default=None, + description="Filter health checks to specific components (postgres, kafka, circuit_breaker, consul, vault)" + ) + + include_metrics: bool = Field( + default=True, + description="Include detailed metrics in health response" + ) + + include_trends: bool = Field( + default=False, + description="Include trend analysis in health response" + ) + + trend_hours: Optional[int] = Field( + default=1, + gt=0, + description="Number of hours for trend analysis" + ) + + monitoring_interval_seconds: Optional[int] = Field( + default=30, + gt=0, + description="Monitoring interval for start_monitoring operations" + ) + + environment: Optional[str] = Field( + default=None, + description="Target environment for health checks" + ) + + context: Optional[ModelHealthRequestContext] = Field( + default=None, + description="Additional request context" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_response.py b/src/omnibase_infra/models/health/model_health_response.py new file mode 100644 index 0000000000..f8effc01d4 --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_response.py @@ -0,0 +1,91 @@ +"""Health Response Model. + +Shared model for health monitoring operation responses. +Used for returning results from health monitoring operations. +""" + +from pydantic import BaseModel, Field +from typing import Dict, List, Optional +from uuid import UUID +from datetime import datetime +from .model_health_status import ModelHealthStatus +from .model_health_metrics import ModelHealthMetrics +from .model_trend_analysis import ModelTrendAnalysis +from .model_component_status import ModelComponentHealthStatus +from .model_health_alert import ModelHealthAlert + + +class ModelHealthResponse(BaseModel): + """Model for health monitoring operation responses.""" + + operation_type: str = Field( + description="Type of operation that was executed" + ) + + success: bool = Field( + description="Whether the operation was successful" + ) + + correlation_id: UUID = Field( + description="Request correlation ID for tracking" + ) + + timestamp: datetime = Field( + description="Response timestamp" + ) + + execution_time_ms: float = Field( + ge=0.0, + description="Operation execution time in milliseconds" + ) + + health_status: Optional[ModelHealthStatus] = Field( + default=None, + description="Current health status (for health_check operations)" + ) + + health_metrics: Optional[ModelHealthMetrics] = Field( + default=None, + description="Detailed health metrics (for get_metrics operations)" + ) + + trend_analysis: Optional[ModelTrendAnalysis] = Field( + default=None, + description="Health trend analysis (for get_trends operations)" + ) + + monitoring_started: Optional[bool] = Field( + default=None, + description="Whether monitoring was started (for start_monitoring operations)" + ) + + monitoring_stopped: Optional[bool] = Field( + default=None, + description="Whether monitoring was stopped (for stop_monitoring operations)" + ) + + component_statuses: Optional[Dict[str, ModelComponentHealthStatus]] = Field( + default=None, + description="Individual component health statuses mapped by component name" + ) + + alerts: Optional[List[ModelHealthAlert]] = Field( + default=None, + description="Active health alerts" + ) + + prometheus_metrics: Optional[str] = Field( + default=None, + description="Prometheus-formatted metrics string" + ) + + error_message: Optional[str] = Field( + default=None, + description="Error message if operation failed" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_health_status.py b/src/omnibase_infra/models/health/model_health_status.py new file mode 100644 index 0000000000..4e5df3f7a3 --- /dev/null +++ b/src/omnibase_infra/models/health/model_health_status.py @@ -0,0 +1,94 @@ +"""Health Status Model. + +Shared model for infrastructure health status information. +Used across health monitoring nodes and status reporting. +""" + +from enum import Enum +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime +from .model_health_details import ModelHealthDetails + + +class HealthStatusEnum(str, Enum): + """Infrastructure health status levels.""" + HEALTHY = "healthy" # All systems operational + DEGRADED = "degraded" # Some issues but service available + UNHEALTHY = "unhealthy" # Critical issues affecting service + + +class ModelHealthStatus(BaseModel): + """Model for infrastructure health status.""" + + overall_status: HealthStatusEnum = Field( + description="Overall infrastructure health status" + ) + + timestamp: datetime = Field( + description="Health check timestamp" + ) + + environment: str = Field( + description="Environment where health check was performed" + ) + + service_name: str = Field( + default="omnibase_infrastructure", + description="Name of the service being monitored" + ) + + postgres_healthy: bool = Field( + description="PostgreSQL component health status" + ) + + kafka_healthy: bool = Field( + description="Kafka component health status" + ) + + circuit_breaker_healthy: bool = Field( + description="Circuit breaker component health status" + ) + + consul_healthy: Optional[bool] = Field( + default=None, + description="Consul service discovery health status" + ) + + vault_healthy: Optional[bool] = Field( + default=None, + description="Vault secret management health status" + ) + + health_score: float = Field( + ge=0.0, + le=100.0, + description="Overall health score (0-100)" + ) + + error_rate_percent: float = Field( + ge=0.0, + le=100.0, + description="Current error rate percentage" + ) + + response_time_ms: float = Field( + ge=0.0, + description="Average response time in milliseconds" + ) + + uptime_seconds: Optional[float] = Field( + default=None, + ge=0.0, + description="Service uptime in seconds" + ) + + details: Optional[ModelHealthDetails] = Field( + default=None, + description="Additional health status details" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_kafka_metrics.py b/src/omnibase_infra/models/health/model_kafka_metrics.py new file mode 100644 index 0000000000..b422331c10 --- /dev/null +++ b/src/omnibase_infra/models/health/model_kafka_metrics.py @@ -0,0 +1,159 @@ +"""Kafka Metrics Model. + +Strongly-typed model for Kafka component health metrics. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelKafkaMetrics(BaseModel): + """Model for Kafka component health metrics.""" + + # Producer metrics + producer_active_count: int = Field( + ge=0, + description="Number of active Kafka producers" + ) + + producer_pool_size: int = Field( + ge=0, + description="Total producer pool size" + ) + + producer_success_rate: float = Field( + ge=0.0, + le=100.0, + description="Producer success rate percentage" + ) + + messages_sent_total: int = Field( + ge=0, + description="Total number of messages sent" + ) + + messages_per_second: float = Field( + ge=0.0, + description="Average messages sent per second" + ) + + # Consumer metrics (if applicable) + consumer_active_count: Optional[int] = Field( + default=None, + ge=0, + description="Number of active Kafka consumers" + ) + + messages_consumed_total: Optional[int] = Field( + default=None, + ge=0, + description="Total number of messages consumed" + ) + + consumer_lag_total: Optional[int] = Field( + default=None, + ge=0, + description="Total consumer lag across all partitions" + ) + + # Topic metrics + topics_count: int = Field( + ge=0, + description="Number of topics being used" + ) + + partitions_count: int = Field( + ge=0, + description="Total number of partitions across all topics" + ) + + # Performance metrics + avg_send_latency_ms: float = Field( + ge=0.0, + description="Average message send latency in milliseconds" + ) + + avg_batch_size: float = Field( + ge=0.0, + description="Average batch size for producer operations" + ) + + compression_ratio: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Message compression ratio percentage" + ) + + # Error metrics + send_failures_total: int = Field( + ge=0, + description="Total number of send failures" + ) + + connection_errors: int = Field( + ge=0, + description="Number of Kafka connection errors" + ) + + timeout_errors: int = Field( + ge=0, + description="Number of timeout errors" + ) + + serialization_errors: int = Field( + ge=0, + description="Number of message serialization errors" + ) + + # Resource utilization + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Kafka client memory usage in megabytes" + ) + + buffer_memory_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Producer buffer memory usage percentage" + ) + + # Connection health + broker_connections: int = Field( + ge=0, + description="Number of active broker connections" + ) + + metadata_age_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Age of metadata cache in milliseconds" + ) + + # Topic-specific metrics + active_topics: Optional[int] = Field( + default=None, + ge=0, + description="Number of topics with active message traffic" + ) + + high_throughput_topics: Optional[int] = Field( + default=None, + ge=0, + description="Number of topics with high message throughput" + ) + + # Throughput metrics + bytes_sent_per_second: float = Field( + ge=0.0, + description="Average bytes sent per second" + ) + + bytes_received_per_second: Optional[float] = Field( + default=None, + ge=0.0, + description="Average bytes received per second (for consumers)" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_postgres_metrics.py b/src/omnibase_infra/models/health/model_postgres_metrics.py new file mode 100644 index 0000000000..02dad3497d --- /dev/null +++ b/src/omnibase_infra/models/health/model_postgres_metrics.py @@ -0,0 +1,121 @@ +"""PostgreSQL Metrics Model. + +Strongly-typed model for PostgreSQL component health metrics. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional + + +class ModelPostgresMetrics(BaseModel): + """Model for PostgreSQL component health metrics.""" + + # Connection metrics + active_connections: int = Field( + ge=0, + description="Number of active database connections" + ) + + max_connections: int = Field( + ge=0, + description="Maximum allowed database connections" + ) + + idle_connections: int = Field( + ge=0, + description="Number of idle database connections" + ) + + connection_pool_utilization: float = Field( + ge=0.0, + le=100.0, + description="Connection pool utilization percentage" + ) + + # Performance metrics + avg_query_duration_ms: float = Field( + ge=0.0, + description="Average query execution time in milliseconds" + ) + + slow_queries_count: int = Field( + ge=0, + description="Number of slow queries in the monitoring period" + ) + + queries_per_second: float = Field( + ge=0.0, + description="Average queries processed per second" + ) + + # Database health + database_size_mb: float = Field( + ge=0.0, + description="Total database size in megabytes" + ) + + locks_count: int = Field( + ge=0, + description="Number of active database locks" + ) + + deadlocks_count: int = Field( + ge=0, + description="Number of deadlocks detected" + ) + + # Availability metrics + uptime_seconds: int = Field( + ge=0, + description="Database uptime in seconds" + ) + + last_backup_timestamp: Optional[str] = Field( + default=None, + description="ISO timestamp of last successful backup" + ) + + replication_lag_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Replication lag in milliseconds (if applicable)" + ) + + # Error metrics + connection_errors: int = Field( + ge=0, + description="Number of connection errors" + ) + + transaction_rollbacks: int = Field( + ge=0, + description="Number of transaction rollbacks" + ) + + # Resource utilization + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Database CPU usage percentage" + ) + + memory_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Database memory usage percentage" + ) + + disk_io_read_mb_s: Optional[float] = Field( + default=None, + ge=0.0, + description="Disk I/O read rate in MB/s" + ) + + disk_io_write_mb_s: Optional[float] = Field( + default=None, + ge=0.0, + description="Disk I/O write rate in MB/s" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_request_context.py b/src/omnibase_infra/models/health/model_request_context.py new file mode 100644 index 0000000000..6a68aee559 --- /dev/null +++ b/src/omnibase_infra/models/health/model_request_context.py @@ -0,0 +1,149 @@ +"""Health Request Context Model. + +Strongly-typed model for health request context information. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelHealthRequestContext(BaseModel): + """Model for health monitoring request context.""" + + # Request source information + source_service: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the service making the request" + ) + + source_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the service making the request" + ) + + user_agent: Optional[str] = Field( + default=None, + max_length=200, + description="User agent string for the request" + ) + + # Request configuration + timeout_seconds: Optional[int] = Field( + default=None, + gt=0, + le=300, + description="Request timeout in seconds" + ) + + retry_count: Optional[int] = Field( + default=None, + ge=0, + le=5, + description="Number of retries to attempt" + ) + + priority_level: Optional[str] = Field( + default=None, + pattern="^(low|normal|high|critical)$", + description="Request priority level" + ) + + # Monitoring context + alert_thresholds_override: Optional[bool] = Field( + default=None, + description="Whether to use custom alert thresholds" + ) + + custom_error_threshold: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Custom error rate threshold percentage" + ) + + custom_response_time_threshold: Optional[float] = Field( + default=None, + ge=0.0, + description="Custom response time threshold in milliseconds" + ) + + # Data collection preferences + detailed_metrics_required: Optional[bool] = Field( + default=None, + description="Whether detailed metrics are required" + ) + + historical_data_required: Optional[bool] = Field( + default=None, + description="Whether historical data should be included" + ) + + include_resource_metrics: Optional[bool] = Field( + default=None, + description="Whether to include resource utilization metrics" + ) + + # Notification preferences + notification_channels: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Notification channels for alerts" + ) + + suppress_notifications: Optional[bool] = Field( + default=None, + description="Whether to suppress notifications for this request" + ) + + # Debugging and tracing + debug_mode: Optional[bool] = Field( + default=None, + description="Whether to enable debug mode for this request" + ) + + trace_id: Optional[str] = Field( + default=None, + max_length=100, + description="Distributed tracing trace ID" + ) + + span_id: Optional[str] = Field( + default=None, + max_length=50, + description="Distributed tracing span ID" + ) + + # Performance preferences + cache_results: Optional[bool] = Field( + default=None, + description="Whether results should be cached" + ) + + cache_ttl_seconds: Optional[int] = Field( + default=None, + gt=0, + le=3600, + description="Cache time-to-live in seconds" + ) + + # Environment-specific context + deployment_stage: Optional[str] = Field( + default=None, + pattern="^(development|staging|production|test)$", + description="Deployment stage context" + ) + + region: Optional[str] = Field( + default=None, + max_length=50, + description="Deployment region" + ) + + availability_zone: Optional[str] = Field( + default=None, + max_length=50, + description="Availability zone" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_trend_analysis.py b/src/omnibase_infra/models/health/model_trend_analysis.py new file mode 100644 index 0000000000..583a2017ed --- /dev/null +++ b/src/omnibase_infra/models/health/model_trend_analysis.py @@ -0,0 +1,113 @@ +"""Health Trend Analysis Model. + +Strongly-typed model for health trend analysis data. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List +from datetime import datetime + + +class ModelTrendDataPoint(BaseModel): + """Model for individual trend data points.""" + + timestamp: datetime = Field( + description="Data point timestamp" + ) + + value: float = Field( + description="Metric value at this timestamp" + ) + + metric_name: str = Field( + description="Name of the metric being tracked" + ) + + +class ModelTrendAnalysis(BaseModel): + """Model for health trend analysis data.""" + + analysis_period_hours: int = Field( + ge=1, + description="Time period covered by this analysis in hours" + ) + + analysis_timestamp: datetime = Field( + description="When this analysis was generated" + ) + + # Trend indicators + overall_trend: str = Field( + pattern="^(improving|stable|degrading|unknown)$", + description="Overall health trend direction" + ) + + trend_confidence: float = Field( + ge=0.0, + le=100.0, + description="Confidence level of trend analysis percentage" + ) + + # Performance trends + avg_response_time_trend: float = Field( + description="Average response time change percentage (positive = slower)" + ) + + error_rate_trend: float = Field( + description="Error rate change percentage (positive = more errors)" + ) + + throughput_trend: float = Field( + description="Throughput change percentage (positive = higher throughput)" + ) + + # Resource utilization trends + cpu_usage_trend: Optional[float] = Field( + default=None, + description="CPU usage change percentage" + ) + + memory_usage_trend: Optional[float] = Field( + default=None, + description="Memory usage change percentage" + ) + + connection_count_trend: float = Field( + description="Connection count change percentage" + ) + + # Predictive indicators + projected_issues_count: int = Field( + ge=0, + description="Number of potential issues identified" + ) + + capacity_warning_threshold_hours: Optional[float] = Field( + default=None, + ge=0.0, + description="Estimated hours until capacity warning threshold" + ) + + # Data quality indicators + data_points_analyzed: int = Field( + ge=1, + description="Number of data points used in analysis" + ) + + missing_data_periods: int = Field( + ge=0, + description="Number of periods with missing data" + ) + + # Historical context + historical_data: Optional[List[ModelTrendDataPoint]] = Field( + default=None, + max_items=1000, + description="Historical data points used for trend analysis" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/health/model_vault_metrics.py b/src/omnibase_infra/models/health/model_vault_metrics.py new file mode 100644 index 0000000000..4f6818f1b2 --- /dev/null +++ b/src/omnibase_infra/models/health/model_vault_metrics.py @@ -0,0 +1,170 @@ +"""Vault Metrics Model. + +Strongly-typed model for Vault secret management health metrics. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional + + +class ModelVaultMetrics(BaseModel): + """Model for Vault secret management health metrics.""" + + # Authentication metrics + active_tokens: int = Field( + ge=0, + description="Number of active authentication tokens" + ) + + token_lookups_per_second: float = Field( + ge=0.0, + description="Token lookup operations per second" + ) + + authentication_success_rate: float = Field( + ge=0.0, + le=100.0, + description="Authentication success rate percentage" + ) + + token_renewals_per_second: float = Field( + ge=0.0, + description="Token renewal operations per second" + ) + + # Secret engine metrics + secret_engines_mounted: int = Field( + ge=0, + description="Number of mounted secret engines" + ) + + secrets_read_per_second: float = Field( + ge=0.0, + description="Secret read operations per second" + ) + + secrets_written_per_second: float = Field( + ge=0.0, + description="Secret write operations per second" + ) + + kv_operations_per_second: float = Field( + ge=0.0, + description="Key-value secret operations per second" + ) + + # Performance metrics + avg_secret_read_latency_ms: float = Field( + ge=0.0, + description="Average secret read latency in milliseconds" + ) + + avg_secret_write_latency_ms: float = Field( + ge=0.0, + description="Average secret write latency in milliseconds" + ) + + policy_evaluations_per_second: float = Field( + ge=0.0, + description="Policy evaluations per second" + ) + + # Storage metrics + storage_operations_per_second: float = Field( + ge=0.0, + description="Backend storage operations per second" + ) + + storage_size_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Backend storage size in megabytes" + ) + + # HA and clustering metrics + is_leader: bool = Field( + description="Whether this Vault node is the cluster leader" + ) + + cluster_nodes: int = Field( + ge=1, + description="Number of nodes in Vault cluster" + ) + + unsealed_nodes: int = Field( + ge=0, + description="Number of unsealed nodes in cluster" + ) + + # Error and audit metrics + operation_errors_per_second: float = Field( + ge=0.0, + description="Operation errors per second" + ) + + audit_log_failures: int = Field( + ge=0, + description="Number of audit log write failures" + ) + + seal_status_checks: int = Field( + ge=0, + description="Number of seal status checks performed" + ) + + # Resource utilization + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Vault process memory usage in megabytes" + ) + + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Vault process CPU usage percentage" + ) + + # Certificate metrics (if using PKI engine) + certificates_issued_total: Optional[int] = Field( + default=None, + ge=0, + description="Total number of certificates issued" + ) + + certificates_revoked_total: Optional[int] = Field( + default=None, + ge=0, + description="Total number of certificates revoked" + ) + + # Transit engine metrics (if enabled) + encryption_operations_per_second: Optional[float] = Field( + default=None, + ge=0.0, + description="Encryption operations per second" + ) + + decryption_operations_per_second: Optional[float] = Field( + default=None, + ge=0.0, + description="Decryption operations per second" + ) + + # Connection metrics + client_connections: int = Field( + ge=0, + description="Number of active client connections" + ) + + api_requests_per_second: float = Field( + ge=0.0, + description="API requests per second" + ) + + api_response_time_ms: float = Field( + ge=0.0, + description="Average API response time in milliseconds" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/infrastructure/model_infrastructure_health_metrics.py b/src/omnibase_infra/models/infrastructure/model_infrastructure_health_metrics.py new file mode 100644 index 0000000000..949b6d7254 --- /dev/null +++ b/src/omnibase_infra/models/infrastructure/model_infrastructure_health_metrics.py @@ -0,0 +1,98 @@ +"""Infrastructure Health Metrics Model. + +Pydantic model for aggregated infrastructure health metrics, extracted from +infrastructure_health_monitor.py for shared usage across ONEX nodes. +""" + +from pydantic import BaseModel, Field +from omnibase_infra.models.postgres.model_postgres_performance_metrics import ModelPostgresPerformanceMetrics +from omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics import ModelCircuitBreakerMetrics +from omnibase_infra.models.kafka.model_kafka_producer_pool_stats import ModelKafkaProducerPoolStats +from datetime import datetime + + +class ModelInfrastructureHealthMetrics(BaseModel): + """Model for aggregated infrastructure health metrics.""" + + # Overall health + overall_status: str = Field( + description="Overall health status: healthy, degraded, unhealthy" + ) + + timestamp: float = Field( + description="Unix timestamp of health check" + ) + + environment: str = Field( + description="Target environment name" + ) + + # Component statuses + postgres_healthy: bool = Field( + description="PostgreSQL connection health status" + ) + + kafka_healthy: bool = Field( + description="Kafka producer health status" + ) + + circuit_breaker_healthy: bool = Field( + description="Circuit breaker health status" + ) + + # Detailed metrics - using strongly typed models per ONEX standards + postgres_metrics: ModelPostgresPerformanceMetrics = Field( + description="Detailed PostgreSQL performance metrics" + ) + + kafka_metrics: ModelKafkaProducerPoolStats = Field( + description="Detailed Kafka producer pool statistics" + ) + + circuit_breaker_metrics: ModelCircuitBreakerMetrics = Field( + description="Detailed circuit breaker metrics" + ) + + # Aggregate statistics + total_connections: int = Field( + ge=0, + description="Total number of active connections" + ) + + total_messages_processed: int = Field( + ge=0, + description="Total messages processed" + ) + + total_events_queued: int = Field( + ge=0, + description="Total events in queues" + ) + + error_rate_percent: float = Field( + ge=0.0, + le=100.0, + description="Error rate percentage" + ) + + # Performance indicators + avg_db_response_time_ms: float = Field( + ge=0.0, + description="Average database response time in milliseconds" + ) + + avg_kafka_throughput_mps: float = Field( + ge=0.0, + description="Average Kafka throughput in messages per second" + ) + + circuit_breaker_success_rate: float = Field( + ge=0.0, + le=100.0, + description="Circuit breaker success rate percentage" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/observability/__init__.py b/src/omnibase_infra/models/observability/__init__.py new file mode 100644 index 0000000000..729239bdef --- /dev/null +++ b/src/omnibase_infra/models/observability/__init__.py @@ -0,0 +1,5 @@ +"""Observability Models Package. + +Shared models for infrastructure observability, metrics, and monitoring. +Used by observability nodes and related infrastructure components. +""" \ No newline at end of file diff --git a/src/omnibase_infra/models/observability/model_alert.py b/src/omnibase_infra/models/observability/model_alert.py new file mode 100644 index 0000000000..a164360466 --- /dev/null +++ b/src/omnibase_infra/models/observability/model_alert.py @@ -0,0 +1,87 @@ +"""Alert Model. + +Shared model for infrastructure alerts and notifications. +Used across observability infrastructure for alert management. +""" + +from enum import Enum +from pydantic import BaseModel, Field +from typing import Optional +from datetime import datetime +from .model_alert_details import ModelAlertDetails + + +class AlertSeverityEnum(str, Enum): + """Alert severity levels.""" + CRITICAL = "critical" # Service-affecting issues + HIGH = "high" # Performance degradation + MEDIUM = "medium" # Potential issues + LOW = "low" # Informational + + +class ModelAlert(BaseModel): + """Model for infrastructure alerts.""" + + id: str = Field( + description="Unique alert identifier" + ) + + name: str = Field( + description="Alert name/title" + ) + + description: str = Field( + description="Detailed alert description" + ) + + severity: AlertSeverityEnum = Field( + description="Alert severity level" + ) + + timestamp: datetime = Field( + description="Alert creation timestamp" + ) + + source: str = Field( + description="Source component that generated the alert" + ) + + resolved: bool = Field( + default=False, + description="Whether the alert has been resolved" + ) + + resolution_timestamp: Optional[datetime] = Field( + default=None, + description="Alert resolution timestamp" + ) + + details: Optional[ModelAlertDetails] = Field( + default=None, + description="Additional alert details and context" + ) + + environment: Optional[str] = Field( + default=None, + description="Environment where alert was generated" + ) + + threshold_value: Optional[float] = Field( + default=None, + description="Threshold value that triggered the alert" + ) + + current_value: Optional[float] = Field( + default=None, + description="Current metric value when alert was triggered" + ) + + alert_rule: Optional[str] = Field( + default=None, + description="Alert rule that triggered this alert" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/observability/model_alert_details.py b/src/omnibase_infra/models/observability/model_alert_details.py new file mode 100644 index 0000000000..91d532beeb --- /dev/null +++ b/src/omnibase_infra/models/observability/model_alert_details.py @@ -0,0 +1,174 @@ +"""Alert Details Model. + +Strongly-typed model for infrastructure alert details and context. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelAlertDetails(BaseModel): + """Model for infrastructure alert details and context.""" + + # Metric information + metric_name: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the metric that triggered the alert" + ) + + metric_unit: Optional[str] = Field( + default=None, + max_length=20, + description="Unit of measurement for the metric" + ) + + measurement_interval: Optional[str] = Field( + default=None, + max_length=50, + description="Interval over which the metric was measured" + ) + + # Threshold information + warning_threshold: Optional[float] = Field( + default=None, + description="Warning threshold value" + ) + + critical_threshold: Optional[float] = Field( + default=None, + description="Critical threshold value" + ) + + threshold_operator: Optional[str] = Field( + default=None, + pattern="^(gt|gte|lt|lte|eq|neq)$", + description="Threshold comparison operator" + ) + + # Time-based information + duration_seconds: Optional[int] = Field( + default=None, + ge=0, + description="Duration the condition has been active in seconds" + ) + + first_occurrence: Optional[str] = Field( + default=None, + description="ISO timestamp of first occurrence" + ) + + last_occurrence: Optional[str] = Field( + default=None, + description="ISO timestamp of last occurrence" + ) + + occurrence_count: Optional[int] = Field( + default=None, + ge=1, + description="Number of times the condition has occurred" + ) + + # Component information + affected_components: Optional[List[str]] = Field( + default=None, + max_items=20, + description="List of components affected by this alert" + ) + + component_health_status: Optional[str] = Field( + default=None, + pattern="^(healthy|degraded|unhealthy|unknown)$", + description="Health status of the affected component" + ) + + # Impact assessment + impact_level: Optional[str] = Field( + default=None, + pattern="^(none|low|medium|high|severe)$", + description="Assessed impact level" + ) + + users_affected_estimate: Optional[int] = Field( + default=None, + ge=0, + description="Estimated number of users affected" + ) + + services_affected: Optional[List[str]] = Field( + default=None, + max_items=20, + description="List of services affected by this alert" + ) + + # Resolution information + auto_resolution_available: Optional[bool] = Field( + default=None, + description="Whether automatic resolution is available" + ) + + manual_intervention_required: Optional[bool] = Field( + default=None, + description="Whether manual intervention is required" + ) + + runbook_url: Optional[str] = Field( + default=None, + max_length=500, + description="URL to relevant runbook or documentation" + ) + + escalation_policy: Optional[str] = Field( + default=None, + max_length=100, + description="Escalation policy to follow" + ) + + # Notification information + notification_channels: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Notification channels used for this alert" + ) + + notification_sent: Optional[bool] = Field( + default=None, + description="Whether notifications have been sent" + ) + + # Context and metadata + deployment_version: Optional[str] = Field( + default=None, + max_length=50, + description="Deployment version when alert was triggered" + ) + + region: Optional[str] = Field( + default=None, + max_length=50, + description="Geographic region where alert occurred" + ) + + availability_zone: Optional[str] = Field( + default=None, + max_length=50, + description="Availability zone where alert occurred" + ) + + # Performance context + baseline_value: Optional[float] = Field( + default=None, + description="Baseline value for comparison" + ) + + deviation_percentage: Optional[float] = Field( + default=None, + description="Percentage deviation from baseline" + ) + + trend_direction: Optional[str] = Field( + default=None, + pattern="^(improving|stable|degrading|unknown)$", + description="Trend direction of the metric" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/observability/model_metric_point.py b/src/omnibase_infra/models/observability/model_metric_point.py new file mode 100644 index 0000000000..2d8ef0afc9 --- /dev/null +++ b/src/omnibase_infra/models/observability/model_metric_point.py @@ -0,0 +1,64 @@ +"""Metric Point Model. + +Shared model for individual metric data points. +Used across observability infrastructure for metric collection. +""" + +from enum import Enum +from pydantic import BaseModel, Field +from typing import Dict, Optional +from datetime import datetime + + +class MetricTypeEnum(str, Enum): + """Types of metrics collected by observability system.""" + COUNTER = "counter" # Monotonically increasing values + GAUGE = "gauge" # Point-in-time values + HISTOGRAM = "histogram" # Distribution of values + SUMMARY = "summary" # Summary statistics + + +class ModelMetricPoint(BaseModel): + """Model for single metric data point.""" + + name: str = Field( + description="Metric name identifier" + ) + + value: float = Field( + description="Metric value" + ) + + timestamp: datetime = Field( + description="Metric collection timestamp" + ) + + metric_type: MetricTypeEnum = Field( + default=MetricTypeEnum.GAUGE, + description="Type of metric" + ) + + labels: Dict[str, str] = Field( + default_factory=dict, + description="Metric labels for categorization" + ) + + unit: Optional[str] = Field( + default=None, + description="Unit of measurement (e.g., 'bytes', 'seconds', 'percent')" + ) + + source: Optional[str] = Field( + default=None, + description="Source component that generated the metric" + ) + + environment: Optional[str] = Field( + default=None, + description="Environment where metric was collected" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/security/model_audit_details.py b/src/omnibase_infra/models/security/model_audit_details.py new file mode 100644 index 0000000000..d03931e11c --- /dev/null +++ b/src/omnibase_infra/models/security/model_audit_details.py @@ -0,0 +1,304 @@ +"""Audit Details Model. + +Strongly-typed model for audit event details to replace Dict[str, Any] usage. +Maintains ONEX compliance with proper field validation and security measures. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List, Union +from datetime import datetime +from uuid import UUID + + +class ModelAuditDetails(BaseModel): + """Model for audit event details with comprehensive typing.""" + + # Request/Response information + request_id: Optional[str] = Field( + default=None, + max_length=100, + description="Request identifier" + ) + + response_status: Optional[int] = Field( + default=None, + ge=100, + le=599, + description="HTTP response status code" + ) + + response_time_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Response time in milliseconds" + ) + + # Resource information + resource_id: Optional[str] = Field( + default=None, + max_length=200, + description="Identifier of the affected resource" + ) + + resource_type: Optional[str] = Field( + default=None, + max_length=100, + description="Type of resource being accessed" + ) + + resource_path: Optional[str] = Field( + default=None, + max_length=500, + description="Path to the resource" + ) + + # Authentication/Authorization information + user_id: Optional[str] = Field( + default=None, + max_length=100, + description="User identifier (non-sensitive)" + ) + + session_id: Optional[str] = Field( + default=None, + max_length=100, + description="Session identifier (hashed)" + ) + + permissions_checked: Optional[List[str]] = Field( + default=None, + max_items=50, + description="List of permissions that were verified" + ) + + authentication_method: Optional[str] = Field( + default=None, + max_length=50, + description="Authentication method used" + ) + + # Operation details + operation_name: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the operation performed" + ) + + operation_parameters: Optional[List[str]] = Field( + default=None, + max_items=20, + description="Operation parameters (sanitized)" + ) + + data_modified: Optional[bool] = Field( + default=None, + description="Whether data was modified by this operation" + ) + + records_affected: Optional[int] = Field( + default=None, + ge=0, + description="Number of records affected" + ) + + # Error information + error_code: Optional[str] = Field( + default=None, + max_length=50, + description="Error code if operation failed" + ) + + error_category: Optional[str] = Field( + default=None, + max_length=100, + description="Category of error" + ) + + error_context: Optional[str] = Field( + default=None, + max_length=500, + description="Additional error context (sanitized)" + ) + + # Security relevant information + security_violation_type: Optional[str] = Field( + default=None, + max_length=100, + description="Type of security violation detected" + ) + + suspicious_activity: Optional[bool] = Field( + default=None, + description="Whether activity was flagged as suspicious" + ) + + threat_level: Optional[str] = Field( + default=None, + pattern="^(low|medium|high|critical)$", + description="Assessed threat level" + ) + + # Compliance information + compliance_requirements: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Applicable compliance requirements" + ) + + data_classification: Optional[str] = Field( + default=None, + pattern="^(public|internal|confidential|restricted)$", + description="Classification of data accessed" + ) + + retention_period_days: Optional[int] = Field( + default=None, + ge=1, + le=3650, + description="Required retention period in days" + ) + + # Additional context + environment: Optional[str] = Field( + default=None, + max_length=50, + description="Environment where event occurred" + ) + + service_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the service" + ) + + correlation_id: Optional[UUID] = Field( + default=None, + description="Correlation ID for request tracing" + ) + + custom_fields: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Additional custom field names (values omitted for security)" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + json_schema_extra = { + "example": { + "request_id": "req-123456", + "response_status": 200, + "response_time_ms": 42.5, + "resource_id": "user:12345", + "resource_type": "user_profile", + "user_id": "user-abc123", + "operation_name": "update_profile", + "data_modified": True, + "records_affected": 1, + "environment": "production", + "data_classification": "confidential" + } + } + + +class ModelAuditMetadata(BaseModel): + """Model for audit event metadata with comprehensive typing.""" + + # Processing information + processing_node: Optional[str] = Field( + default=None, + max_length=100, + description="Node that processed this audit event" + ) + + processing_time: Optional[datetime] = Field( + default=None, + description="When audit event was processed" + ) + + batch_id: Optional[str] = Field( + default=None, + max_length=100, + description="Batch identifier if processed in batch" + ) + + # Storage information + storage_location: Optional[str] = Field( + default=None, + max_length=200, + description="Where audit event is stored" + ) + + compression_used: Optional[bool] = Field( + default=None, + description="Whether compression was applied" + ) + + encryption_used: Optional[bool] = Field( + default=None, + description="Whether encryption was applied" + ) + + # Quality information + data_quality_score: Optional[float] = Field( + default=None, + ge=0.0, + le=1.0, + description="Data quality score (0-1)" + ) + + completeness_percentage: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="Data completeness percentage" + ) + + validation_passed: Optional[bool] = Field( + default=None, + description="Whether validation passed" + ) + + # Alerting information + alert_triggered: Optional[bool] = Field( + default=None, + description="Whether event triggered an alert" + ) + + alert_severity: Optional[str] = Field( + default=None, + pattern="^(info|warning|error|critical)$", + description="Alert severity level" + ) + + notification_sent: Optional[bool] = Field( + default=None, + description="Whether notification was sent" + ) + + # Archival information + archival_required: Optional[bool] = Field( + default=None, + description="Whether event requires archival" + ) + + archival_date: Optional[datetime] = Field( + default=None, + description="When event should be archived" + ) + + retention_policy: Optional[str] = Field( + default=None, + max_length=100, + description="Applicable retention policy" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/security/model_payload_encryption.py b/src/omnibase_infra/models/security/model_payload_encryption.py new file mode 100644 index 0000000000..453d4a604c --- /dev/null +++ b/src/omnibase_infra/models/security/model_payload_encryption.py @@ -0,0 +1,337 @@ +"""Payload Encryption Models. + +Strongly-typed models for payload encryption to replace Dict[str, Any] usage. +Maintains ONEX compliance with proper field validation and security measures. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List, Union +from datetime import datetime + + +class ModelEncryptedPayload(BaseModel): + """Model for encrypted payload data.""" + + # Encryption Metadata + algorithm: str = Field( + max_length=50, + description="Encryption algorithm used" + ) + + key_id: str = Field( + max_length=100, + description="Identifier of the encryption key" + ) + + iv: str = Field( + max_length=200, + description="Initialization vector (base64 encoded)" + ) + + # Encrypted Data + encrypted_data: str = Field( + description="Base64 encoded encrypted data" + ) + + # Integrity + hmac: str = Field( + max_length=500, + description="HMAC for data integrity verification" + ) + + checksum: Optional[str] = Field( + default=None, + max_length=100, + description="Additional checksum for verification" + ) + + # Metadata + encrypted_at: str = Field( + description="ISO timestamp when data was encrypted" + ) + + encryption_version: str = Field( + default="1.0", + max_length=20, + description="Version of encryption scheme used" + ) + + content_type: Optional[str] = Field( + default=None, + max_length=100, + description="Original content type before encryption" + ) + + original_size_bytes: Optional[int] = Field( + default=None, + ge=0, + description="Size of original data before encryption" + ) + + compressed: bool = Field( + default=False, + description="Whether data was compressed before encryption" + ) + + # Security Context + security_level: str = Field( + default="standard", + pattern="^(minimal|standard|high|maximum)$", + description="Security level used for encryption" + ) + + key_derivation_rounds: Optional[int] = Field( + default=None, + ge=1000, + le=1000000, + description="Number of key derivation rounds" + ) + + # Expiration + expires_at: Optional[str] = Field( + default=None, + description="ISO timestamp when encrypted data expires" + ) + + # Additional Context + context_info: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Additional context information (non-sensitive)" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelDecryptionRequest(BaseModel): + """Model for decryption request parameters.""" + + # Encrypted payload reference + encrypted_payload: ModelEncryptedPayload = Field( + description="Encrypted payload to decrypt" + ) + + # Decryption context + key_id: Optional[str] = Field( + default=None, + max_length=100, + description="Override key ID for decryption" + ) + + verify_integrity: bool = Field( + default=True, + description="Whether to verify data integrity" + ) + + verify_expiration: bool = Field( + default=True, + description="Whether to check expiration" + ) + + # Output preferences + return_format: str = Field( + default="string", + pattern="^(string|bytes|dict|auto)$", + description="Format for decrypted data" + ) + + decompress: bool = Field( + default=True, + description="Whether to decompress after decryption" + ) + + # Security validation + required_security_level: Optional[str] = Field( + default=None, + pattern="^(minimal|standard|high|maximum)$", + description="Minimum required security level" + ) + + allowed_algorithms: Optional[List[str]] = Field( + default=None, + max_items=10, + description="List of allowed encryption algorithms" + ) + + max_age_seconds: Optional[int] = Field( + default=None, + ge=0, + description="Maximum age of encrypted data in seconds" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelEncryptionRequest(BaseModel): + """Model for encryption request parameters.""" + + # Data to encrypt + data: Union[str, bytes] = Field( + description="Data to encrypt" + ) + + # Encryption parameters + algorithm: Optional[str] = Field( + default=None, + max_length=50, + description="Encryption algorithm to use" + ) + + key_id: Optional[str] = Field( + default=None, + max_length=100, + description="Key ID to use for encryption" + ) + + security_level: str = Field( + default="standard", + pattern="^(minimal|standard|high|maximum)$", + description="Security level for encryption" + ) + + # Processing options + compress_before_encrypt: bool = Field( + default=False, + description="Whether to compress data before encryption" + ) + + include_metadata: bool = Field( + default=True, + description="Whether to include metadata in result" + ) + + # Expiration + ttl_seconds: Optional[int] = Field( + default=None, + ge=60, + le=31536000, # 1 year + description="Time-to-live in seconds" + ) + + # Context + content_type: Optional[str] = Field( + default=None, + max_length=100, + description="Content type of original data" + ) + + context_tags: Optional[List[str]] = Field( + default=None, + max_items=10, + description="Context tags for encryption" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelEncryptionStats(BaseModel): + """Model for encryption service statistics.""" + + # Operation counts + total_encryptions: int = Field( + ge=0, + description="Total number of encryption operations" + ) + + total_decryptions: int = Field( + ge=0, + description="Total number of decryption operations" + ) + + successful_operations: int = Field( + ge=0, + description="Number of successful operations" + ) + + failed_operations: int = Field( + ge=0, + description="Number of failed operations" + ) + + # Performance metrics + average_encryption_time_ms: float = Field( + ge=0.0, + description="Average encryption time in milliseconds" + ) + + average_decryption_time_ms: float = Field( + ge=0.0, + description="Average decryption time in milliseconds" + ) + + total_data_encrypted_bytes: int = Field( + ge=0, + description="Total bytes encrypted" + ) + + total_data_decrypted_bytes: int = Field( + ge=0, + description="Total bytes decrypted" + ) + + # Key management + active_keys_count: int = Field( + ge=0, + description="Number of active encryption keys" + ) + + key_rotations_count: int = Field( + ge=0, + description="Number of key rotations performed" + ) + + # Cache statistics + cache_hit_rate: float = Field( + ge=0.0, + le=100.0, + description="Cache hit rate percentage" + ) + + cache_size_bytes: int = Field( + ge=0, + description="Current cache size in bytes" + ) + + # Error tracking + integrity_failures: int = Field( + ge=0, + description="Number of integrity verification failures" + ) + + expired_data_requests: int = Field( + ge=0, + description="Number of requests for expired data" + ) + + invalid_key_attempts: int = Field( + ge=0, + description="Number of invalid key attempts" + ) + + # Time information + statistics_period_start: str = Field( + description="ISO timestamp of statistics period start" + ) + + statistics_period_end: str = Field( + description="ISO timestamp of statistics period end" + ) + + uptime_seconds: int = Field( + ge=0, + description="Service uptime in seconds" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" \ No newline at end of file diff --git a/src/omnibase_infra/models/security/model_rate_limiter.py b/src/omnibase_infra/models/security/model_rate_limiter.py new file mode 100644 index 0000000000..482de502ac --- /dev/null +++ b/src/omnibase_infra/models/security/model_rate_limiter.py @@ -0,0 +1,279 @@ +"""Rate Limiter Models. + +Strongly-typed models for rate limiter statistics to replace Dict[str, Any] usage. +Maintains ONEX compliance with proper field validation. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List +from datetime import datetime + + +class ModelClientStats(BaseModel): + """Model for rate limiter client statistics.""" + + # Client Information + client_id: str = Field( + max_length=200, + description="Unique client identifier" + ) + + client_type: Optional[str] = Field( + default=None, + max_length=50, + description="Type of client (api, web, mobile, etc.)" + ) + + # Request Statistics + total_requests: int = Field( + ge=0, + description="Total number of requests made" + ) + + allowed_requests: int = Field( + ge=0, + description="Number of requests that were allowed" + ) + + blocked_requests: int = Field( + ge=0, + description="Number of requests that were blocked" + ) + + # Rate Information + current_rate: float = Field( + ge=0.0, + description="Current request rate (requests per second)" + ) + + average_rate: float = Field( + ge=0.0, + description="Average request rate over monitoring period" + ) + + peak_rate: float = Field( + ge=0.0, + description="Peak request rate observed" + ) + + # Limit Information + rate_limit: int = Field( + ge=0, + description="Current rate limit for this client" + ) + + limit_window_seconds: int = Field( + ge=1, + le=3600, + description="Time window for rate limiting in seconds" + ) + + burst_limit: Optional[int] = Field( + default=None, + ge=0, + description="Burst limit for this client" + ) + + # Timing Information + first_request_time: str = Field( + description="ISO timestamp of first request" + ) + + last_request_time: str = Field( + description="ISO timestamp of last request" + ) + + last_blocked_time: Optional[str] = Field( + default=None, + description="ISO timestamp of last blocked request" + ) + + # Penalty Information + penalty_count: int = Field( + default=0, + ge=0, + description="Number of penalties applied" + ) + + current_penalty_expires: Optional[str] = Field( + default=None, + description="ISO timestamp when current penalty expires" + ) + + total_penalty_time_seconds: int = Field( + default=0, + ge=0, + description="Total time spent in penalty" + ) + + # Status Information + is_blocked: bool = Field( + default=False, + description="Whether client is currently blocked" + ) + + is_whitelisted: bool = Field( + default=False, + description="Whether client is whitelisted" + ) + + is_blacklisted: bool = Field( + default=False, + description="Whether client is blacklisted" + ) + + # Additional Context + user_agent: Optional[str] = Field( + default=None, + max_length=500, + description="User agent string (if available)" + ) + + source_ip: Optional[str] = Field( + default=None, + max_length=45, + description="Source IP address (hashed for privacy)" + ) + + geographic_region: Optional[str] = Field( + default=None, + max_length=100, + description="Geographic region of client" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelGlobalStats(BaseModel): + """Model for global rate limiter statistics.""" + + # Overall Request Statistics + total_requests_all_clients: int = Field( + ge=0, + description="Total requests across all clients" + ) + + total_allowed_requests: int = Field( + ge=0, + description="Total allowed requests across all clients" + ) + + total_blocked_requests: int = Field( + ge=0, + description="Total blocked requests across all clients" + ) + + # Rate Statistics + global_request_rate: float = Field( + ge=0.0, + description="Global request rate (requests per second)" + ) + + average_client_rate: float = Field( + ge=0.0, + description="Average rate per client" + ) + + peak_global_rate: float = Field( + ge=0.0, + description="Peak global request rate observed" + ) + + # Client Statistics + total_active_clients: int = Field( + ge=0, + description="Number of active clients" + ) + + total_blocked_clients: int = Field( + ge=0, + description="Number of currently blocked clients" + ) + + total_whitelisted_clients: int = Field( + default=0, + ge=0, + description="Number of whitelisted clients" + ) + + total_blacklisted_clients: int = Field( + default=0, + ge=0, + description="Number of blacklisted clients" + ) + + # Performance Statistics + average_processing_time_ms: float = Field( + ge=0.0, + description="Average processing time per request in milliseconds" + ) + + cache_hit_rate: float = Field( + ge=0.0, + le=100.0, + description="Cache hit rate percentage" + ) + + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Memory usage in megabytes" + ) + + # Time Window Information + statistics_window_seconds: int = Field( + ge=1, + description="Time window these statistics cover" + ) + + statistics_generated_at: str = Field( + description="ISO timestamp when statistics were generated" + ) + + uptime_seconds: int = Field( + ge=0, + description="Rate limiter uptime in seconds" + ) + + # Configuration Information + default_rate_limit: int = Field( + ge=0, + description="Default rate limit for new clients" + ) + + max_clients: Optional[int] = Field( + default=None, + ge=0, + description="Maximum number of clients supported" + ) + + cleanup_interval_seconds: int = Field( + default=300, + ge=60, + description="Interval for cleanup operations" + ) + + # Health Information + is_healthy: bool = Field( + default=True, + description="Whether rate limiter is operating normally" + ) + + error_count: int = Field( + default=0, + ge=0, + description="Number of errors encountered" + ) + + last_error_time: Optional[str] = Field( + default=None, + description="ISO timestamp of last error" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" \ No newline at end of file diff --git a/src/omnibase_infra/models/security/model_tls_config.py b/src/omnibase_infra/models/security/model_tls_config.py new file mode 100644 index 0000000000..bad0763f3d --- /dev/null +++ b/src/omnibase_infra/models/security/model_tls_config.py @@ -0,0 +1,301 @@ +"""TLS Configuration Models. + +Strongly-typed models for TLS configuration to replace Dict[str, Any] usage. +Maintains ONEX compliance with proper field validation. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelKafkaProducerConfig(BaseModel): + """Model for Kafka producer configuration.""" + + # TLS/SSL Configuration + security_protocol: str = Field( + default="SSL", + description="Security protocol for Kafka connection" + ) + + ssl_ca_location: Optional[str] = Field( + default=None, + max_length=500, + description="Path to CA certificate file" + ) + + ssl_certificate_location: Optional[str] = Field( + default=None, + max_length=500, + description="Path to client certificate file" + ) + + ssl_key_location: Optional[str] = Field( + default=None, + max_length=500, + description="Path to client private key file" + ) + + ssl_key_password: Optional[str] = Field( + default=None, + max_length=200, + description="Private key password" + ) + + ssl_verify_hostname: bool = Field( + default=True, + description="Whether to verify hostname in SSL certificates" + ) + + ssl_check_hostname: bool = Field( + default=True, + description="Whether to check hostname in SSL certificates" + ) + + # Connection Configuration + bootstrap_servers: List[str] = Field( + min_items=1, + max_items=10, + description="List of Kafka bootstrap servers" + ) + + client_id: Optional[str] = Field( + default=None, + max_length=100, + description="Client identifier" + ) + + # Performance Configuration + acks: str = Field( + default="all", + pattern="^(0|1|all)$", + description="Number of acknowledgments required" + ) + + retries: int = Field( + default=3, + ge=0, + le=10, + description="Number of retries for failed sends" + ) + + batch_size: int = Field( + default=16384, + ge=1, + le=1048576, + description="Batch size in bytes" + ) + + linger_ms: int = Field( + default=5, + ge=0, + le=1000, + description="Time to wait for additional records in ms" + ) + + buffer_memory: int = Field( + default=33554432, + ge=1048576, + le=134217728, + description="Total memory available for buffering" + ) + + # Timeout Configuration + request_timeout_ms: int = Field( + default=30000, + ge=1000, + le=300000, + description="Request timeout in milliseconds" + ) + + delivery_timeout_ms: int = Field( + default=120000, + ge=5000, + le=600000, + description="Delivery timeout in milliseconds" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelSecurityPolicy(BaseModel): + """Model for security policy configuration.""" + + # TLS Requirements + tls_version_min: str = Field( + default="1.2", + pattern="^(1\\.2|1\\.3)$", + description="Minimum required TLS version" + ) + + tls_version_max: str = Field( + default="1.3", + pattern="^(1\\.2|1\\.3)$", + description="Maximum allowed TLS version" + ) + + cipher_suites: List[str] = Field( + min_items=1, + max_items=20, + description="Allowed cipher suites" + ) + + # Certificate Requirements + certificate_validation_required: bool = Field( + default=True, + description="Whether certificate validation is required" + ) + + hostname_verification_required: bool = Field( + default=True, + description="Whether hostname verification is required" + ) + + certificate_chain_validation: bool = Field( + default=True, + description="Whether certificate chain validation is required" + ) + + # Security Features + perfect_forward_secrecy_required: bool = Field( + default=True, + description="Whether perfect forward secrecy is required" + ) + + ocsp_stapling_required: bool = Field( + default=False, + description="Whether OCSP stapling is required" + ) + + sni_required: bool = Field( + default=True, + description="Whether Server Name Indication is required" + ) + + # Compliance + fips_mode_enabled: bool = Field( + default=False, + description="Whether FIPS mode is enabled" + ) + + compliance_level: str = Field( + default="standard", + pattern="^(minimal|standard|strict|maximum)$", + description="Security compliance level" + ) + + audit_all_connections: bool = Field( + default=True, + description="Whether to audit all TLS connections" + ) + + # Timeouts and Limits + handshake_timeout_ms: int = Field( + default=30000, + ge=5000, + le=120000, + description="TLS handshake timeout in milliseconds" + ) + + session_timeout_ms: int = Field( + default=300000, + ge=60000, + le=3600000, + description="TLS session timeout in milliseconds" + ) + + max_connections_per_host: int = Field( + default=100, + ge=1, + le=1000, + description="Maximum connections per host" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" + + +class ModelCredentialCacheEntry(BaseModel): + """Model for credential cache entry.""" + + # Credential Information + credential_type: str = Field( + max_length=50, + description="Type of credential (api_key, certificate, token)" + ) + + credential_id: str = Field( + max_length=200, + description="Identifier for the credential" + ) + + environment: str = Field( + max_length=50, + description="Environment the credential is for" + ) + + # Cache Metadata + cached_at: str = Field( + description="ISO timestamp when credential was cached" + ) + + expires_at: Optional[str] = Field( + default=None, + description="ISO timestamp when credential expires" + ) + + last_validated: Optional[str] = Field( + default=None, + description="ISO timestamp of last validation" + ) + + # Usage Tracking + access_count: int = Field( + default=0, + ge=0, + description="Number of times credential was accessed" + ) + + last_accessed: Optional[str] = Field( + default=None, + description="ISO timestamp of last access" + ) + + # Security Status + is_valid: bool = Field( + default=True, + description="Whether credential is currently valid" + ) + + validation_failures: int = Field( + default=0, + ge=0, + description="Number of validation failures" + ) + + is_revoked: bool = Field( + default=False, + description="Whether credential has been revoked" + ) + + # Additional Metadata + source: Optional[str] = Field( + default=None, + max_length=100, + description="Source of the credential" + ) + + scope: Optional[List[str]] = Field( + default=None, + max_items=20, + description="Scopes associated with the credential" + ) + + class Config: + """Pydantic model configuration.""" + validate_assignment = True + extra = "forbid" \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/__init__.py b/src/omnibase_infra/models/tracing/__init__.py new file mode 100644 index 0000000000..01c9ddd051 --- /dev/null +++ b/src/omnibase_infra/models/tracing/__init__.py @@ -0,0 +1,5 @@ +"""Tracing Models Package. + +Shared models for distributed tracing operations and configurations. +Used by tracing nodes and related infrastructure components. +""" \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_event_envelope.py b/src/omnibase_infra/models/tracing/model_event_envelope.py new file mode 100644 index 0000000000..84958637e0 --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_event_envelope.py @@ -0,0 +1,187 @@ +"""Event Envelope Model. + +Strongly-typed model for event envelope used in tracing context. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List +from uuid import UUID +from datetime import datetime + + +class ModelEventEnvelope(BaseModel): + """Model for event envelope used in tracing context.""" + + # Event identification + event_id: UUID = Field( + description="Unique event identifier" + ) + + event_type: str = Field( + max_length=100, + description="Type of event" + ) + + event_version: str = Field( + max_length=20, + description="Event schema version" + ) + + correlation_id: UUID = Field( + description="Request correlation ID" + ) + + # Timing information + timestamp: datetime = Field( + description="Event timestamp" + ) + + processing_started_at: Optional[datetime] = Field( + default=None, + description="When processing started" + ) + + processing_completed_at: Optional[datetime] = Field( + default=None, + description="When processing completed" + ) + + # Event routing + source_service: str = Field( + max_length=100, + description="Service that generated the event" + ) + + target_service: Optional[str] = Field( + default=None, + max_length=100, + description="Intended target service" + ) + + routing_key: Optional[str] = Field( + default=None, + max_length=200, + description="Message routing key" + ) + + # Event metadata + priority: Optional[int] = Field( + default=None, + ge=0, + le=10, + description="Event processing priority (0-10)" + ) + + retry_count: int = Field( + default=0, + ge=0, + le=10, + description="Number of processing retries" + ) + + max_retries: int = Field( + default=3, + ge=0, + le=10, + description="Maximum number of retries allowed" + ) + + # Content information + content_type: str = Field( + default="application/json", + max_length=100, + description="Content type of event payload" + ) + + content_encoding: Optional[str] = Field( + default=None, + max_length=50, + description="Content encoding (if compressed)" + ) + + content_size_bytes: Optional[int] = Field( + default=None, + ge=0, + description="Size of event payload in bytes" + ) + + # Security and validation + checksum: Optional[str] = Field( + default=None, + max_length=100, + description="Payload checksum for integrity verification" + ) + + signature: Optional[str] = Field( + default=None, + max_length=200, + description="Digital signature for authenticity" + ) + + # Processing context + processing_mode: Optional[str] = Field( + default=None, + pattern="^(sync|async|batch|stream)$", + description="Event processing mode" + ) + + batch_id: Optional[UUID] = Field( + default=None, + description="Batch identifier (if part of batch processing)" + ) + + partition_key: Optional[str] = Field( + default=None, + max_length=100, + description="Partitioning key for distributed processing" + ) + + # Error handling + dead_letter_queue_eligible: bool = Field( + default=True, + description="Whether event can be sent to dead letter queue" + ) + + error_message: Optional[str] = Field( + default=None, + max_length=1000, + description="Error message from last processing attempt" + ) + + # Environment and deployment + environment: Optional[str] = Field( + default=None, + max_length=50, + description="Environment where event was generated" + ) + + region: Optional[str] = Field( + default=None, + max_length=50, + description="Geographic region" + ) + + deployment_version: Optional[str] = Field( + default=None, + max_length=50, + description="Deployment version of source service" + ) + + # Tracing integration + trace_headers: Optional[List[str]] = Field( + default=None, + max_items=20, + description="List of tracing header names present in envelope" + ) + + span_context_injected: bool = Field( + default=False, + description="Whether span context has been injected into headers" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_parent_context.py b/src/omnibase_infra/models/tracing/model_parent_context.py new file mode 100644 index 0000000000..37e89b58c4 --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_parent_context.py @@ -0,0 +1,124 @@ +"""Parent Context Model. + +Strongly-typed model for OpenTelemetry parent context. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional + + +class ModelParentContext(BaseModel): + """Model for OpenTelemetry parent context.""" + + # Trace identification + trace_id: str = Field( + min_length=32, + max_length=32, + description="OpenTelemetry trace ID (32 hex characters)" + ) + + span_id: str = Field( + min_length=16, + max_length=16, + description="OpenTelemetry span ID (16 hex characters)" + ) + + # Trace flags + trace_flags: int = Field( + default=1, + ge=0, + le=255, + description="OpenTelemetry trace flags (8-bit value)" + ) + + # Sampling decision + is_sampled: bool = Field( + default=True, + description="Whether this trace is sampled" + ) + + # Trace state (W3C format) + trace_state: Optional[str] = Field( + default=None, + max_length=512, + description="W3C trace state header value" + ) + + # Parent span information + parent_span_name: Optional[str] = Field( + default=None, + max_length=200, + description="Name of the parent span" + ) + + parent_service_name: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the service that created the parent span" + ) + + parent_service_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the parent service" + ) + + # Context propagation + propagation_format: Optional[str] = Field( + default=None, + pattern="^(w3c|b3|jaeger|opencensus)$", + description="Context propagation format used" + ) + + # Remote context indicator + is_remote: bool = Field( + default=False, + description="Whether this is a remote parent context" + ) + + # Timing information + parent_start_time: Optional[str] = Field( + default=None, + description="ISO timestamp when parent span started" + ) + + # Baggage (OpenTelemetry baggage) + baggage_count: Optional[int] = Field( + default=None, + ge=0, + le=100, + description="Number of baggage items" + ) + + # Debug information + debug_enabled: Optional[bool] = Field( + default=None, + description="Whether debug tracing is enabled" + ) + + force_sampling: Optional[bool] = Field( + default=None, + description="Whether sampling should be forced for this trace" + ) + + # Priority information + priority: Optional[int] = Field( + default=None, + ge=0, + le=10, + description="Trace priority level (0-10)" + ) + + # Environment context + environment: Optional[str] = Field( + default=None, + max_length=50, + description="Environment where parent span was created" + ) + + cluster: Optional[str] = Field( + default=None, + max_length=100, + description="Cluster where parent span was created" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_span_attributes.py b/src/omnibase_infra/models/tracing/model_span_attributes.py new file mode 100644 index 0000000000..7f6d1bd206 --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_span_attributes.py @@ -0,0 +1,203 @@ +"""Span Attributes Model. + +Strongly-typed model for OpenTelemetry span attributes. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List + + +class ModelSpanAttributes(BaseModel): + """Model for OpenTelemetry span attributes.""" + + # Service identification + service_name: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the service creating the span" + ) + + service_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the service creating the span" + ) + + service_instance_id: Optional[str] = Field( + default=None, + max_length=100, + description="Instance ID of the service" + ) + + # HTTP attributes (if applicable) + http_method: Optional[str] = Field( + default=None, + pattern="^(GET|POST|PUT|DELETE|PATCH|HEAD|OPTIONS)$", + description="HTTP method" + ) + + http_url: Optional[str] = Field( + default=None, + max_length=500, + description="Full HTTP URL" + ) + + http_status_code: Optional[int] = Field( + default=None, + ge=100, + le=599, + description="HTTP response status code" + ) + + http_user_agent: Optional[str] = Field( + default=None, + max_length=500, + description="HTTP User-Agent header value" + ) + + # Database attributes (if applicable) + db_system: Optional[str] = Field( + default=None, + max_length=50, + description="Database management system identifier" + ) + + db_connection_string: Optional[str] = Field( + default=None, + max_length=500, + description="Database connection string (sanitized)" + ) + + db_user: Optional[str] = Field( + default=None, + max_length=100, + description="Database user name" + ) + + db_name: Optional[str] = Field( + default=None, + max_length=100, + description="Database name" + ) + + db_operation: Optional[str] = Field( + default=None, + max_length=50, + description="Database operation name" + ) + + # Messaging attributes (if applicable) + messaging_system: Optional[str] = Field( + default=None, + max_length=50, + description="Messaging system identifier" + ) + + messaging_destination: Optional[str] = Field( + default=None, + max_length=200, + description="Message destination name" + ) + + messaging_destination_kind: Optional[str] = Field( + default=None, + pattern="^(queue|topic|exchange)$", + description="Kind of message destination" + ) + + messaging_operation: Optional[str] = Field( + default=None, + pattern="^(publish|receive|process)$", + description="Messaging operation type" + ) + + # RPC attributes (if applicable) + rpc_system: Optional[str] = Field( + default=None, + max_length=50, + description="RPC system identifier" + ) + + rpc_service: Optional[str] = Field( + default=None, + max_length=100, + description="RPC service name" + ) + + rpc_method: Optional[str] = Field( + default=None, + max_length=100, + description="RPC method name" + ) + + # Error attributes + error_type: Optional[str] = Field( + default=None, + max_length=100, + description="Error type/class name" + ) + + error_message: Optional[str] = Field( + default=None, + max_length=1000, + description="Error message" + ) + + # Performance attributes + operation_duration_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Operation duration in milliseconds" + ) + + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="CPU usage percentage during operation" + ) + + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Memory usage in megabytes" + ) + + # Custom business attributes + user_id: Optional[str] = Field( + default=None, + max_length=100, + description="User identifier" + ) + + tenant_id: Optional[str] = Field( + default=None, + max_length=100, + description="Tenant identifier" + ) + + correlation_ids: Optional[List[str]] = Field( + default=None, + max_items=10, + description="List of correlation identifiers" + ) + + # Environment attributes + environment: Optional[str] = Field( + default=None, + max_length=50, + description="Deployment environment" + ) + + region: Optional[str] = Field( + default=None, + max_length=50, + description="Geographic region" + ) + + availability_zone: Optional[str] = Field( + default=None, + max_length=50, + description="Availability zone" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_span_data.py b/src/omnibase_infra/models/tracing/model_span_data.py new file mode 100644 index 0000000000..6304bca5b9 --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_span_data.py @@ -0,0 +1,224 @@ +"""Span Data Model. + +Strongly-typed model for OpenTelemetry span data. +Replaces Dict[str, Any] usage to maintain ONEX compliance. +""" + +from pydantic import BaseModel, Field +from typing import Optional, List +from datetime import datetime +from .model_span_attributes import ModelSpanAttributes + + +class ModelSpanEvent(BaseModel): + """Model for span events/logs.""" + + name: str = Field( + max_length=200, + description="Event name" + ) + + timestamp: datetime = Field( + description="Event timestamp" + ) + + attributes: Optional[ModelSpanAttributes] = Field( + default=None, + description="Event attributes" + ) + + +class ModelSpanLink(BaseModel): + """Model for span links.""" + + trace_id: str = Field( + min_length=32, + max_length=32, + description="Linked trace ID (32 hex characters)" + ) + + span_id: str = Field( + min_length=16, + max_length=16, + description="Linked span ID (16 hex characters)" + ) + + trace_flags: int = Field( + default=1, + ge=0, + le=255, + description="Trace flags for linked span" + ) + + attributes: Optional[ModelSpanAttributes] = Field( + default=None, + description="Link attributes" + ) + + +class ModelSpanStatus(BaseModel): + """Model for span status.""" + + code: str = Field( + pattern="^(UNSET|OK|ERROR)$", + description="Span status code" + ) + + message: Optional[str] = Field( + default=None, + max_length=1000, + description="Status description message" + ) + + +class ModelSpanData(BaseModel): + """Model for OpenTelemetry span data.""" + + # Span identification + trace_id: str = Field( + min_length=32, + max_length=32, + description="OpenTelemetry trace ID (32 hex characters)" + ) + + span_id: str = Field( + min_length=16, + max_length=16, + description="OpenTelemetry span ID (16 hex characters)" + ) + + parent_span_id: Optional[str] = Field( + default=None, + min_length=16, + max_length=16, + description="Parent span ID (16 hex characters)" + ) + + # Span metadata + name: str = Field( + max_length=200, + description="Span operation name" + ) + + kind: str = Field( + pattern="^(INTERNAL|SERVER|CLIENT|PRODUCER|CONSUMER)$", + description="Span kind" + ) + + status: ModelSpanStatus = Field( + description="Span status information" + ) + + # Timing information + start_time: datetime = Field( + description="Span start timestamp" + ) + + end_time: Optional[datetime] = Field( + default=None, + description="Span end timestamp (None if still active)" + ) + + duration_ms: Optional[float] = Field( + default=None, + ge=0.0, + description="Span duration in milliseconds" + ) + + # Span data + attributes: Optional[ModelSpanAttributes] = Field( + default=None, + description="Span attributes" + ) + + events: Optional[List[ModelSpanEvent]] = Field( + default=None, + max_items=1000, + description="Span events/logs" + ) + + links: Optional[List[ModelSpanLink]] = Field( + default=None, + max_items=100, + description="Span links to other spans" + ) + + # Resource information + service_name: str = Field( + max_length=100, + description="Service name that created the span" + ) + + service_version: Optional[str] = Field( + default=None, + max_length=50, + description="Service version" + ) + + service_instance_id: Optional[str] = Field( + default=None, + max_length=100, + description="Service instance identifier" + ) + + # Instrumentation information + instrumentation_library_name: Optional[str] = Field( + default=None, + max_length=100, + description="Name of the instrumentation library" + ) + + instrumentation_library_version: Optional[str] = Field( + default=None, + max_length=50, + description="Version of the instrumentation library" + ) + + # Sampling information + is_sampled: bool = Field( + description="Whether this span is sampled" + ) + + sampling_priority: Optional[int] = Field( + default=None, + ge=0, + le=10, + description="Sampling priority (0-10)" + ) + + # Error information + has_error: bool = Field( + default=False, + description="Whether the span contains error information" + ) + + error_type: Optional[str] = Field( + default=None, + max_length=100, + description="Error type/class name" + ) + + error_message: Optional[str] = Field( + default=None, + max_length=1000, + description="Error message" + ) + + # Performance metrics + cpu_usage_percent: Optional[float] = Field( + default=None, + ge=0.0, + le=100.0, + description="CPU usage during span" + ) + + memory_usage_mb: Optional[float] = Field( + default=None, + ge=0.0, + description="Memory usage during span in MB" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_trace_context.py b/src/omnibase_infra/models/tracing/model_trace_context.py new file mode 100644 index 0000000000..806e7c74cf --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_trace_context.py @@ -0,0 +1,68 @@ +"""Trace Context Model. + +Shared model for distributed trace context information. +Used for trace propagation across infrastructure components. +""" + +from pydantic import BaseModel, Field +from typing import Any, Dict, Optional +from uuid import UUID +from datetime import datetime + + +class ModelTraceContext(BaseModel): + """Model for distributed trace context.""" + + trace_id: str = Field( + description="Unique trace identifier" + ) + + span_id: str = Field( + description="Current span identifier" + ) + + parent_span_id: Optional[str] = Field( + default=None, + description="Parent span identifier" + ) + + correlation_id: UUID = Field( + description="Correlation ID for request tracking" + ) + + service_name: str = Field( + description="Name of the service creating the trace" + ) + + operation_name: str = Field( + description="Name of the operation being traced" + ) + + timestamp: datetime = Field( + description="Trace context creation timestamp" + ) + + environment: str = Field( + description="Environment where trace is generated" + ) + + baggage: Optional[Dict[str, str]] = Field( + default=None, + description="Baggage data for cross-service propagation" + ) + + trace_flags: Optional[str] = Field( + default=None, + description="Trace flags for sampling and other options" + ) + + trace_state: Optional[str] = Field( + default=None, + description="Trace state for vendor-specific data" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_tracing_config.py b/src/omnibase_infra/models/tracing/model_tracing_config.py new file mode 100644 index 0000000000..0beeb91688 --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_tracing_config.py @@ -0,0 +1,76 @@ +"""Tracing Configuration Model. + +Shared model for distributed tracing configuration settings. +Used across tracing infrastructure for consistent setup. +""" + +from pydantic import BaseModel, Field +from typing import Dict, Optional + + +class ModelTracingConfig(BaseModel): + """Model for distributed tracing configuration.""" + + environment: str = Field( + description="Target environment for tracing configuration" + ) + + service_name: str = Field( + default="omnibase_infrastructure", + description="Service name for tracing identification" + ) + + service_version: str = Field( + default="1.0.0", + description="Service version for tracing identification" + ) + + otlp_endpoint: str = Field( + default="http://localhost:4317", + description="OpenTelemetry Protocol (OTLP) endpoint" + ) + + otlp_headers: Optional[Dict[str, str]] = Field( + default=None, + description="OTLP headers for authentication and configuration" + ) + + trace_sample_rate: float = Field( + default=1.0, + ge=0.0, + le=1.0, + description="Trace sampling rate (0.0 to 1.0)" + ) + + enable_db_instrumentation: bool = Field( + default=True, + description="Enable database instrumentation" + ) + + enable_kafka_instrumentation: bool = Field( + default=True, + description="Enable Kafka instrumentation" + ) + + enable_audit_integration: bool = Field( + default=True, + description="Enable integration with audit logging" + ) + + batch_timeout_ms: int = Field( + default=5000, + gt=0, + description="Batch span processor timeout in milliseconds" + ) + + max_export_batch_size: int = Field( + default=512, + gt=0, + description="Maximum batch size for span export" + ) + + max_queue_size: int = Field( + default=2048, + gt=0, + description="Maximum queue size for spans" + ) \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_tracing_request.py b/src/omnibase_infra/models/tracing/model_tracing_request.py new file mode 100644 index 0000000000..6896e4979c --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_tracing_request.py @@ -0,0 +1,67 @@ +"""Tracing Request Model. + +Shared model for distributed tracing operation requests. +Used for tracing operations and span management. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from uuid import UUID +from datetime import datetime +from .model_trace_context import ModelTraceContext +from .model_span_attributes import ModelSpanAttributes +from .model_parent_context import ModelParentContext +from .model_event_envelope import ModelEventEnvelope + + +class ModelTracingRequest(BaseModel): + """Model for distributed tracing operation requests.""" + + operation_type: str = Field( + description="Type of tracing operation", + regex=r"^(start_span|end_span|inject_context|extract_context|get_current_span)$" + ) + + correlation_id: UUID = Field( + description="Request correlation ID for tracking" + ) + + timestamp: datetime = Field( + description="Request timestamp" + ) + + operation_name: Optional[str] = Field( + default=None, + description="Name of the operation to trace (for start_span)" + ) + + span_kind: Optional[str] = Field( + default="internal", + description="Type of span (internal, server, client, producer, consumer)" + ) + + trace_context: Optional[ModelTraceContext] = Field( + default=None, + description="Trace context for context operations" + ) + + span_attributes: Optional[ModelSpanAttributes] = Field( + default=None, + description="Attributes to add to span" + ) + + parent_context: Optional[ModelParentContext] = Field( + default=None, + description="Parent context for span creation" + ) + + event_envelope: Optional[ModelEventEnvelope] = Field( + default=None, + description="Event envelope for context injection/extraction" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/models/tracing/model_tracing_response.py b/src/omnibase_infra/models/tracing/model_tracing_response.py new file mode 100644 index 0000000000..c649d873ca --- /dev/null +++ b/src/omnibase_infra/models/tracing/model_tracing_response.py @@ -0,0 +1,73 @@ +"""Tracing Response Model. + +Shared model for distributed tracing operation responses. +Used for returning results from tracing operations. +""" + +from pydantic import BaseModel, Field +from typing import Optional +from uuid import UUID +from datetime import datetime +from .model_trace_context import ModelTraceContext +from .model_span_data import ModelSpanData + + +class ModelTracingResponse(BaseModel): + """Model for distributed tracing operation responses.""" + + operation_type: str = Field( + description="Type of operation that was executed" + ) + + success: bool = Field( + description="Whether the operation was successful" + ) + + correlation_id: UUID = Field( + description="Request correlation ID for tracking" + ) + + timestamp: datetime = Field( + description="Response timestamp" + ) + + execution_time_ms: float = Field( + ge=0.0, + description="Operation execution time in milliseconds" + ) + + span_id: Optional[str] = Field( + default=None, + description="Span ID (for start_span operations)" + ) + + trace_id: Optional[str] = Field( + default=None, + description="Trace ID (for span operations)" + ) + + trace_context: Optional[ModelTraceContext] = Field( + default=None, + description="Extracted trace context (for extract_context operations)" + ) + + context_injected: Optional[bool] = Field( + default=None, + description="Whether context was successfully injected (for inject_context operations)" + ) + + span_data: Optional[ModelSpanData] = Field( + default=None, + description="Span data and attributes (for get_current_span operations)" + ) + + error_message: Optional[str] = Field( + default=None, + description="Error message if operation failed" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat(), + UUID: lambda v: str(v) + } \ No newline at end of file diff --git a/src/omnibase_infra/nodes/consul/v1_0_0/models/model_consul_adapter_output.py b/src/omnibase_infra/nodes/consul/v1_0_0/models/model_consul_adapter_output.py index fb00812a42..ac1c17dc71 100644 --- a/src/omnibase_infra/nodes/consul/v1_0_0/models/model_consul_adapter_output.py +++ b/src/omnibase_infra/nodes/consul/v1_0_0/models/model_consul_adapter_output.py @@ -1,7 +1,13 @@ #!/usr/bin/env python3 from pydantic import BaseModel -from typing import Any +from typing import Union, Dict, List + +# Import shared Consul models +from omnibase_infra.models.consul.model_consul_service_response import ModelConsulServiceResponse +from omnibase_infra.models.consul.model_consul_health_response import ModelConsulHealthResponse +from omnibase_infra.models.consul.model_consul_kv_response import ModelConsulKvResponse +from omnibase_infra.models.consul.model_consul_service_list_response import ModelConsulServiceListResponse class ModelConsulAdapterOutput(BaseModel): @@ -10,6 +16,15 @@ class ModelConsulAdapterOutput(BaseModel): Node-specific model for returning Consul operation results through effect outputs. """ - consul_operation_result: Any + consul_operation_result: Union[ + ModelConsulServiceResponse, + ModelConsulHealthResponse, + ModelConsulKvResponse, + ModelConsulServiceListResponse, + Dict[str, Union[str, int, bool]], + List[Dict[str, Union[str, int, bool]]], + str, + bool + ] success: bool operation_type: str \ No newline at end of file diff --git a/src/omnibase_infra/nodes/consul_projector/v1_0_0/models/model_consul_projector_output.py b/src/omnibase_infra/nodes/consul_projector/v1_0_0/models/model_consul_projector_output.py index 63574f52a3..ce41ed7665 100644 --- a/src/omnibase_infra/nodes/consul_projector/v1_0_0/models/model_consul_projector_output.py +++ b/src/omnibase_infra/nodes/consul_projector/v1_0_0/models/model_consul_projector_output.py @@ -1,7 +1,13 @@ #!/usr/bin/env python3 from pydantic import BaseModel -from typing import Any, Optional +from typing import Union, Optional, Dict, List + +# Import shared Consul models +from omnibase_infra.models.consul.model_consul_service_response import ModelConsulServiceResponse +from omnibase_infra.models.consul.model_consul_health_response import ModelConsulHealthResponse +from omnibase_infra.models.consul.model_consul_kv_response import ModelConsulKvResponse +from .model_consul_topology_metrics import ModelConsulTopologyMetrics class ModelConsulProjectorOutput(BaseModel): @@ -10,7 +16,15 @@ class ModelConsulProjectorOutput(BaseModel): Node-specific model for returning projection operation results through effect outputs. """ - projection_result: Any + projection_result: Union[ + ModelConsulServiceResponse, + ModelConsulHealthResponse, + ModelConsulKvResponse, + ModelConsulTopologyMetrics, + Dict[str, Union[str, int, bool, List[str]]], + List[Dict[str, Union[str, int, bool]]], + str + ] projection_type: str timestamp: str # ISO format datetime metadata: Optional[dict] = None \ No newline at end of file diff --git a/src/omnibase_infra/nodes/consul_projector/v1_0_0/node.py b/src/omnibase_infra/nodes/consul_projector/v1_0_0/node.py index dad4b6ed92..42de1cb491 100644 --- a/src/omnibase_infra/nodes/consul_projector/v1_0_0/node.py +++ b/src/omnibase_infra/nodes/consul_projector/v1_0_0/node.py @@ -13,7 +13,7 @@ ModelEffectOutput, ) from omnibase_core.node_effect_service import NodeEffectService -from omnibase_core.onex_container import ONEXContainer +from omnibase_core.core.onex_container import ModelONEXContainer from omnibase_core.enums.enum_health_status import EnumHealthStatus from omnibase_core.model.core.model_health_status import ModelHealthStatus @@ -47,7 +47,7 @@ class NodeInfrastructureConsulProjectorEffect(NodeEffectService): Provides comprehensive state views for service discovery, health monitoring, and topology analysis. """ - def __init__(self, container: ONEXContainer): + def __init__(self, container: ModelONEXContainer): # Use proper base class - no more boilerplate! super().__init__(container) diff --git a/src/omnibase_infra/nodes/kafka_adapter/v1_0_0/node.py b/src/omnibase_infra/nodes/kafka_adapter/v1_0_0/node.py index 829e8cb40b..ef9c45ab99 100644 --- a/src/omnibase_infra/nodes/kafka_adapter/v1_0_0/node.py +++ b/src/omnibase_infra/nodes/kafka_adapter/v1_0_0/node.py @@ -26,7 +26,7 @@ from omnibase_core.core.errors.onex_error import CoreErrorCode from omnibase_core.core.errors.onex_error import OnexError from omnibase_core.node_effect_service import NodeEffectService -from omnibase_core.onex_container import ModelONEXContainer +from omnibase_core.core.onex_container import ModelONEXContainer from omnibase_core.enums.enum_health_status import EnumHealthStatus from omnibase_core.model.core.model_health_status import ModelHealthStatus diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/config.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/config.py new file mode 100644 index 0000000000..229e516e53 --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/config.py @@ -0,0 +1,155 @@ +"""Configuration management for Distributed Tracing Compute Node. + +Provides ONEX-compliant configuration injection patterns for OpenTelemetry tracing setup. +Validates external dependencies and enforces secure configuration practices. +""" + +import os +from typing import Optional +from pydantic import BaseModel, HttpUrl, Field, validator +from omnibase_core.exceptions.base_onex_error import OnexError +from omnibase_core.enums.enum_core_error_code import CoreErrorCode + + +class TracingConfig(BaseModel): + """ + Configuration model for distributed tracing with OpenTelemetry. + + Provides secure endpoint validation and environment-specific configuration + following ONEX dependency injection patterns. + """ + + otel_exporter_otlp_endpoint: HttpUrl = Field( + default="http://localhost:4317", + description="The OTLP endpoint for exporting traces. Must be a valid HTTP/HTTPS URL." + ) + + trace_sample_rate: float = Field( + default=1.0, + ge=0.0, + le=1.0, + description="Trace sampling rate between 0.0 (no traces) and 1.0 (all traces)." + ) + + service_name: str = Field( + default="omnibase_infrastructure", + min_length=1, + description="OpenTelemetry service name for trace identification." + ) + + service_version: str = Field( + default="1.0.0", + min_length=1, + description="Service version for trace metadata." + ) + + environment: str = Field( + default="development", + min_length=1, + description="Deployment environment (development, staging, production)." + ) + + @validator('otel_exporter_otlp_endpoint') + def validate_otlp_endpoint(cls, v): + """Validate OTLP endpoint URL for security and format compliance.""" + if not v: + raise ValueError("OTLP endpoint URL cannot be empty") + + # Ensure only HTTP/HTTPS schemes are allowed + allowed_schemes = {'http', 'https'} + if v.scheme not in allowed_schemes: + raise ValueError(f"OTLP endpoint must use HTTP or HTTPS scheme, got: {v.scheme}") + + # Validate network location is present + if not v.host: + raise ValueError("OTLP endpoint must include a valid hostname") + + return v + + @validator('environment') + def validate_environment(cls, v): + """Validate environment name against known deployment environments.""" + allowed_environments = {'development', 'dev', 'staging', 'stage', 'production', 'prod', 'test', 'testing'} + if v.lower() not in allowed_environments: + # Log warning but don't fail - allow custom environments + pass + return v.lower() + + +def load_tracing_config() -> TracingConfig: + """ + Load and validate tracing configuration from environment variables. + + Follows ONEX configuration injection pattern by centralizing environment + variable access and validation at application startup. + + Returns: + TracingConfig: Validated configuration object + + Raises: + OnexError: If configuration validation fails + """ + try: + config_data = {} + + # Load OTLP endpoint with validation + if endpoint := os.getenv('OTEL_EXPORTER_OTLP_ENDPOINT'): + config_data['otel_exporter_otlp_endpoint'] = endpoint + + # Load trace sample rate with validation + if sample_rate := os.getenv('OTEL_TRACE_SAMPLE_RATE'): + try: + config_data['trace_sample_rate'] = float(sample_rate) + except ValueError: + raise ValueError(f"Invalid trace sample rate: {sample_rate}. Must be a float between 0.0 and 1.0") + + # Load service information + if service_name := os.getenv('OTEL_SERVICE_NAME'): + config_data['service_name'] = service_name + + if service_version := os.getenv('OTEL_SERVICE_VERSION'): + config_data['service_version'] = service_version + + # Detect environment from multiple possible variables + env_vars = ['ENVIRONMENT', 'ENV', 'DEPLOYMENT_ENV', 'NODE_ENV', 'OMNIBASE_ENV'] + for var in env_vars: + if env_value := os.getenv(var): + config_data['environment'] = env_value.lower() + break + + # Create and validate configuration + return TracingConfig(**config_data) + + except Exception as e: + raise OnexError( + message=f"Failed to load tracing configuration: {str(e)}", + error_code=CoreErrorCode.CONFIGURATION_ERROR + ) from e + + +def create_test_tracing_config( + endpoint: Optional[str] = None, + sample_rate: Optional[float] = None, + environment: Optional[str] = None +) -> TracingConfig: + """ + Create a test configuration for unit testing. + + Args: + endpoint: Override OTLP endpoint + sample_rate: Override trace sample rate + environment: Override environment + + Returns: + TracingConfig: Test configuration object + """ + config_data = {} + + if endpoint: + config_data['otel_exporter_otlp_endpoint'] = endpoint + if sample_rate is not None: + config_data['trace_sample_rate'] = sample_rate + if environment: + config_data['environment'] = environment + + return TracingConfig(**config_data) \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/contract.yaml b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/contract.yaml new file mode 100644 index 0000000000..aaedf32e03 --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/contract.yaml @@ -0,0 +1,129 @@ +contract_version: "1.0.0" +node_version: "1.0.0" +version: "1.0.0" +node_name: "node_distributed_tracing_compute" +contract_name: "NodeDistributedTracingComputeContract" +name: "node_distributed_tracing_compute" +node_type: "COMPUTE" +description: "Distributed Tracing Compute Node for OpenTelemetry integration with trace context processing and enrichment" + +input_model: "ModelDistributedTracingInput" +output_model: "ModelDistributedTracingOutput" + +dependencies: + # ONEX Container dependency injection + - name: "model_onex_container" + type: "model" + class_name: "ModelONEXContainer" + module: "omnibase_core.core.onex_container" + + # Core ONEX event model + - name: "model_onex_event" + type: "model" + class_name: "ModelOnexEvent" + module: "omnibase_core.model.core.model_onex_event" + + # Shared tracing models + - name: "model_tracing_config" + type: "model" + class_name: "ModelTracingConfig" + module: "omnibase_infra.models.tracing.model_tracing_config" + + - name: "model_trace_context" + type: "model" + class_name: "ModelTraceContext" + module: "omnibase_infra.models.tracing.model_trace_context" + + - name: "model_tracing_request" + type: "model" + class_name: "ModelTracingRequest" + module: "omnibase_infra.models.tracing.model_tracing_request" + + - name: "model_tracing_response" + type: "model" + class_name: "ModelTracingResponse" + module: "omnibase_infra.models.tracing.model_tracing_response" + +definitions: + ModelDistributedTracingInput: + type: "object" + properties: + operation_type: + type: "string" + enum: ["initialize_tracing", "trace_operation", "inject_context", "extract_context", "trace_database", "trace_kafka", "shutdown_tracing"] + description: "Type of tracing operation to perform" + operation_name: + type: "string" + description: "Name of the operation to trace (required for trace operations)" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID for the operation" + event: + $ref: "#/definitions/ModelOnexEvent" + description: "Event for context injection/extraction operations" + span_kind: + type: "string" + enum: ["INTERNAL", "SERVER", "CLIENT", "PRODUCER", "CONSUMER"] + description: "OpenTelemetry span kind" + default: "INTERNAL" + attributes: + type: "object" + additionalProperties: true + description: "Additional span attributes" + environment: + type: "string" + description: "Environment configuration (development, staging, production)" + default: "development" + database_query: + type: "string" + description: "Database query for database tracing operations" + kafka_topic: + type: "string" + description: "Kafka topic for Kafka tracing operations" + required: + - operation_type + - correlation_id + + ModelDistributedTracingOutput: + type: "object" + properties: + success: + type: "boolean" + description: "Whether the operation succeeded" + operation_type: + type: "string" + description: "Type of operation that was performed" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID from the request" + result: + type: "object" + description: "Operation-specific result data" + error_message: + type: "string" + description: "Error message if operation failed" + trace_id: + type: "string" + description: "OpenTelemetry trace ID (if applicable)" + span_id: + type: "string" + description: "OpenTelemetry span ID (if applicable)" + tracing_enabled: + type: "boolean" + description: "Whether tracing is currently enabled" + timestamp: + type: "string" + format: "date-time" + description: "Response timestamp" + required: + - success + - operation_type + - correlation_id + - tracing_enabled + - timestamp + + ModelOnexEvent: + type: "object" + description: "ONEX event model (external dependency)" \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_input.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_input.py new file mode 100644 index 0000000000..8a18c31ace --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_input.py @@ -0,0 +1,79 @@ +"""Distributed Tracing Input Model. + +Node-specific input model for the distributed tracing compute node. +""" + +from enum import Enum +from pydantic import BaseModel, Field +from typing import Optional +from uuid import UUID + +from omnibase_core.model.core.model_onex_event import ModelOnexEvent +from omnibase_infra.models.tracing.model_span_attributes import ModelSpanAttributes + + +class TracingOperation(str, Enum): + """Distributed tracing operations.""" + INITIALIZE_TRACING = "initialize_tracing" + TRACE_OPERATION = "trace_operation" + INJECT_CONTEXT = "inject_context" + EXTRACT_CONTEXT = "extract_context" + TRACE_DATABASE = "trace_database" + TRACE_KAFKA = "trace_kafka" + SHUTDOWN_TRACING = "shutdown_tracing" + + +class SpanKind(str, Enum): + """OpenTelemetry span kinds.""" + INTERNAL = "INTERNAL" + SERVER = "SERVER" + CLIENT = "CLIENT" + PRODUCER = "PRODUCER" + CONSUMER = "CONSUMER" + + +class ModelDistributedTracingInput(BaseModel): + """Input model for distributed tracing operations.""" + + operation_type: TracingOperation = Field( + description="Type of tracing operation to perform" + ) + + operation_name: Optional[str] = Field( + default=None, + description="Name of the operation to trace (required for trace operations)" + ) + + correlation_id: UUID = Field( + description="Correlation ID for the operation" + ) + + event: Optional[ModelOnexEvent] = Field( + default=None, + description="Event for context injection/extraction operations" + ) + + span_kind: SpanKind = Field( + default=SpanKind.INTERNAL, + description="OpenTelemetry span kind" + ) + + attributes: Optional[ModelSpanAttributes] = Field( + default=None, + description="Additional span attributes" + ) + + environment: str = Field( + default="development", + description="Environment configuration (development, staging, production)" + ) + + database_query: Optional[str] = Field( + default=None, + description="Database query for database tracing operations" + ) + + kafka_topic: Optional[str] = Field( + default=None, + description="Kafka topic for Kafka tracing operations" + ) \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_output.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_output.py new file mode 100644 index 0000000000..e92c712890 --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/models/model_distributed_tracing_output.py @@ -0,0 +1,58 @@ +"""Distributed Tracing Output Model. + +Node-specific output model for the distributed tracing compute node. +""" + +from pydantic import BaseModel, Field +from typing import Optional, Dict, Union, List +from datetime import datetime +from uuid import UUID + + +class ModelDistributedTracingOutput(BaseModel): + """Output model for distributed tracing operations.""" + + success: bool = Field( + description="Whether the operation succeeded" + ) + + operation_type: str = Field( + description="Type of operation that was performed" + ) + + correlation_id: UUID = Field( + description="Correlation ID from the request" + ) + + result: Optional[Dict[str, Union[str, bool, float, List[str]]]] = Field( + default=None, + description="Operation-specific result data with strongly typed values" + ) + + error_message: Optional[str] = Field( + default=None, + description="Error message if operation failed" + ) + + trace_id: Optional[str] = Field( + default=None, + description="OpenTelemetry trace ID (if applicable)" + ) + + span_id: Optional[str] = Field( + default=None, + description="OpenTelemetry span ID (if applicable)" + ) + + tracing_enabled: bool = Field( + description="Whether tracing is currently enabled" + ) + + timestamp: datetime = Field( + description="Response timestamp" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/node.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/node.py new file mode 100644 index 0000000000..29087c305e --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/node.py @@ -0,0 +1,479 @@ +"""Distributed Tracing Compute Node. + +ONEX compute node for OpenTelemetry integration with trace context processing and enrichment. +Provides end-to-end trace correlation through the entire infrastructure flow. +""" + +import asyncio +import logging +import os +from contextlib import asynccontextmanager +from datetime import datetime +from typing import Dict, Optional, AsyncIterator, Union +from uuid import UUID + +from omnibase_core.base.node_compute_service import NodeComputeService +from omnibase_core.core.errors.onex_error import OnexError, CoreErrorCode +from omnibase_core.core.onex_container import ModelONEXContainer +from omnibase_core.model.core.model_onex_event import ModelOnexEvent + +# OpenTelemetry imports with availability check +try: + from opentelemetry import trace, context, baggage, propagate + from opentelemetry.exporter.otlp.proto.grpc.trace_exporter import OTLPSpanExporter + from opentelemetry.sdk.trace import TracerProvider + from opentelemetry.sdk.trace.export import BatchSpanProcessor + from opentelemetry.sdk.resources import Resource, SERVICE_NAME, SERVICE_VERSION + from opentelemetry.trace.status import Status, StatusCode + from opentelemetry.trace import Span, SpanKind + from opentelemetry.context import Context + OPENTELEMETRY_AVAILABLE = True +except ImportError: + OPENTELEMETRY_AVAILABLE = False + # Stub implementations when OpenTelemetry not available + trace = None + context = None + +from .models.model_distributed_tracing_input import ( + ModelDistributedTracingInput, + TracingOperation, + SpanKind as InputSpanKind +) +from .models.model_distributed_tracing_output import ModelDistributedTracingOutput +from .config import TracingConfig +from .utils.sql_sanitizer import SqlSanitizer + + +class NodeDistributedTracingCompute(NodeComputeService[ModelDistributedTracingInput, ModelDistributedTracingOutput]): + """ + Distributed Tracing Compute Node. + + Provides: + - OpenTelemetry integration with automatic instrumentation + - Trace context propagation through event envelopes + - Environment-specific tracing configuration + - Database and Kafka operation tracing + - Graceful degradation when OpenTelemetry unavailable + """ + + def __init__(self, container: ModelONEXContainer, tracing_config: TracingConfig): + """Initialize the distributed tracing compute node. + + Args: + container: ONEX container for dependency injection + tracing_config: Validated tracing configuration with endpoint validation + """ + super().__init__(container) + self.logger = logging.getLogger(f"{__name__}.NodeDistributedTracingCompute") + + # Inject validated configuration following ONEX patterns + self.tracing_config = tracing_config + + # Tracing components - using Union for proper typing with graceful degradation + self.tracer_provider: Optional[Union["TracerProvider", object]] = None # TracerProvider when available + self.tracer: Optional[Union["trace.Tracer", object]] = None # OpenTelemetry tracer + self.is_initialized = False + + # Check OpenTelemetry availability + if not OPENTELEMETRY_AVAILABLE: + self.logger.warning("OpenTelemetry not available - tracing will be disabled") + + async def initialize(self) -> None: + """Initialize the distributed tracing node.""" + try: + # Initialize OpenTelemetry if available + if OPENTELEMETRY_AVAILABLE: + await self._initialize_opentelemetry() + + self.logger.info( + f"Distributed tracing compute node initialized successfully " + f"(environment: {self.tracing_config.environment}, " + f"endpoint: {self.tracing_config.otel_exporter_otlp_endpoint})" + ) + + except Exception as e: + raise OnexError( + code=CoreErrorCode.INITIALIZATION_ERROR, + message=f"Failed to initialize distributed tracing compute node: {str(e)}" + ) from e + + async def compute(self, input_data: ModelDistributedTracingInput) -> ModelDistributedTracingOutput: + """Execute distributed tracing operations. + + Args: + input_data: Input containing tracing operation type and parameters + + Returns: + Output with operation result and tracing status + """ + try: + # Route to appropriate operation handler + if input_data.operation_type == TracingOperation.INITIALIZE_TRACING: + result = await self._handle_initialize_tracing(input_data) + elif input_data.operation_type == TracingOperation.TRACE_OPERATION: + result = await self._handle_trace_operation(input_data) + elif input_data.operation_type == TracingOperation.INJECT_CONTEXT: + result = await self._handle_inject_context(input_data) + elif input_data.operation_type == TracingOperation.EXTRACT_CONTEXT: + result = await self._handle_extract_context(input_data) + elif input_data.operation_type == TracingOperation.TRACE_DATABASE: + result = await self._handle_trace_database(input_data) + elif input_data.operation_type == TracingOperation.TRACE_KAFKA: + result = await self._handle_trace_kafka(input_data) + elif input_data.operation_type == TracingOperation.SHUTDOWN_TRACING: + result = await self._handle_shutdown_tracing(input_data) + else: + raise OnexError( + code=CoreErrorCode.INVALID_INPUT, + message=f"Unsupported tracing operation type: {input_data.operation_type}" + ) + + return ModelDistributedTracingOutput( + success=True, + operation_type=input_data.operation_type.value, + correlation_id=input_data.correlation_id, + result=result, + trace_id=result.get("trace_id"), + span_id=result.get("span_id"), + tracing_enabled=OPENTELEMETRY_AVAILABLE and self.is_initialized, + timestamp=datetime.now() + ) + + except OnexError: + # Re-raise ONEX errors as-is + raise + except Exception as e: + # Wrap other exceptions in OnexError + raise OnexError( + code=CoreErrorCode.PROCESSING_ERROR, + message=f"Distributed tracing operation failed: {str(e)}" + ) from e + + async def _handle_initialize_tracing(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, float]]: + """Handle tracing initialization.""" + if not OPENTELEMETRY_AVAILABLE: + return { + "initialized": False, + "reason": "OpenTelemetry not available", + "fallback_mode": True + } + + if not self.is_initialized: + await self._initialize_opentelemetry() + + return { + "initialized": self.is_initialized, + "service_name": self.tracing_config.service_name, + "environment": self.tracing_config.environment, + "otlp_endpoint": str(self.tracing_config.otel_exporter_otlp_endpoint), + "sample_rate": self.tracing_config.trace_sample_rate + } + + async def _handle_trace_operation(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, Optional[str], Dict[str, str]]]: + """Handle generic operation tracing.""" + if not self.is_initialized or not input_data.operation_name: + return { + "traced": False, + "reason": "Tracing not initialized or operation name missing" + } + + # Convert span kind + span_kind = self._convert_span_kind(input_data.span_kind) + + # Create span attributes + attributes = { + "correlation_id": str(input_data.correlation_id), + "environment": self.tracing_config.environment, + "service.name": self.tracing_config.service_name, + **(input_data.attributes or {}) + } + + # Create and manage span + span = self.tracer.start_span( + name=input_data.operation_name, + kind=span_kind, + attributes=attributes + ) + + try: + # Get trace and span IDs + span_context = span.get_span_context() + trace_id = format(span_context.trace_id, '032x') if span_context else None + span_id = format(span_context.span_id, '016x') if span_context else None + + # Mark span as successful + span.set_status(Status(StatusCode.OK)) + + return { + "traced": True, + "operation_name": input_data.operation_name, + "trace_id": trace_id, + "span_id": span_id, + "attributes": attributes + } + + finally: + span.end() + + async def _handle_inject_context(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, List[str]]]: + """Handle trace context injection into event.""" + if not input_data.event or not self.is_initialized: + return { + "injected": False, + "reason": "Event missing or tracing not initialized" + } + + try: + # Create a carrier for trace context propagation + carrier = {} + propagate.inject(carrier) + + # Add trace context to event metadata + if not hasattr(input_data.event, 'metadata') or input_data.event.metadata is None: + input_data.event.metadata = {} + + input_data.event.metadata.update({ + "trace_context": carrier, + "trace_timestamp": datetime.now().isoformat(), + "trace_service": self.tracing_config.service_name, + "trace_environment": self.tracing_config.environment + }) + + return { + "injected": True, + "event_id": str(input_data.event.correlation_id), + "context_keys": list(carrier.keys()) + } + + except Exception as e: + self.logger.warning(f"Failed to inject trace context: {e}") + return { + "injected": False, + "reason": str(e) + } + + async def _handle_extract_context(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, Optional[str]]]: + """Handle trace context extraction from event.""" + if not input_data.event or not self.is_initialized: + return { + "extracted": False, + "reason": "Event missing or tracing not initialized" + } + + try: + if not hasattr(input_data.event, 'metadata') or not input_data.event.metadata: + return { + "extracted": False, + "reason": "No metadata in event" + } + + trace_context_data = input_data.event.metadata.get("trace_context") + if not trace_context_data: + return { + "extracted": False, + "reason": "No trace context in event metadata" + } + + # Extract context from carrier + extracted_context = propagate.extract(trace_context_data) + + return { + "extracted": True, + "event_id": str(input_data.event.correlation_id), + "context_available": extracted_context is not None, + "trace_service": input_data.event.metadata.get("trace_service"), + "trace_environment": input_data.event.metadata.get("trace_environment") + } + + except Exception as e: + self.logger.warning(f"Failed to extract trace context: {e}") + return { + "extracted": False, + "reason": str(e) + } + + async def _handle_trace_database(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, Optional[str]]]: + """Handle database operation tracing.""" + if not self.is_initialized or not input_data.operation_name: + return { + "traced": False, + "reason": "Tracing not initialized or operation name missing" + } + + attributes = { + "db.system": "postgresql", + "db.operation": input_data.operation_name, + "component": "postgres_adapter", + "correlation_id": str(input_data.correlation_id) + } + + # Add sanitized query if provided (ONEX-compliant sanitization) + if input_data.database_query: + sanitized_query = SqlSanitizer.sanitize_for_observability(input_data.database_query) + attributes["db.statement"] = sanitized_query + + # Create database span + span = self.tracer.start_span( + name=f"postgres.{input_data.operation_name}", + kind=SpanKind.CLIENT, + attributes=attributes + ) + + try: + span_context = span.get_span_context() + trace_id = format(span_context.trace_id, '032x') if span_context else None + span_id = format(span_context.span_id, '016x') if span_context else None + + span.set_status(Status(StatusCode.OK)) + + return { + "traced": True, + "operation_type": "database", + "operation_name": input_data.operation_name, + "trace_id": trace_id, + "span_id": span_id, + "database_system": "postgresql" + } + + finally: + span.end() + + async def _handle_trace_kafka(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool, Optional[str]]]: + """Handle Kafka operation tracing.""" + if not self.is_initialized or not input_data.operation_name: + return { + "traced": False, + "reason": "Tracing not initialized or operation name missing" + } + + attributes = { + "messaging.system": "kafka", + "messaging.operation": input_data.operation_name, + "component": "kafka_adapter", + "correlation_id": str(input_data.correlation_id) + } + + if input_data.kafka_topic: + attributes["messaging.destination"] = input_data.kafka_topic + attributes["messaging.destination_kind"] = "topic" + + # Determine span kind based on operation + span_kind = SpanKind.PRODUCER if input_data.operation_name == "produce" else SpanKind.CONSUMER + + # Create Kafka span + span = self.tracer.start_span( + name=f"kafka.{input_data.operation_name}", + kind=span_kind, + attributes=attributes + ) + + try: + span_context = span.get_span_context() + trace_id = format(span_context.trace_id, '032x') if span_context else None + span_id = format(span_context.span_id, '016x') if span_context else None + + span.set_status(Status(StatusCode.OK)) + + return { + "traced": True, + "operation_type": "kafka", + "operation_name": input_data.operation_name, + "trace_id": trace_id, + "span_id": span_id, + "messaging_system": "kafka", + "topic": input_data.kafka_topic + } + + finally: + span.end() + + async def _handle_shutdown_tracing(self, input_data: ModelDistributedTracingInput) -> Dict[str, Union[str, bool]]: + """Handle tracing shutdown.""" + if not self.is_initialized: + return { + "shutdown": False, + "reason": "Tracing not initialized" + } + + try: + if self.tracer_provider: + # Force flush pending spans + await asyncio.to_thread(self.tracer_provider.force_flush, timeout_millis=5000) + + self.is_initialized = False + self.logger.info("Distributed tracing shutdown complete") + + return { + "shutdown": True, + "flushed_spans": True + } + + except Exception as e: + self.logger.error(f"Error during tracing shutdown: {e}") + return { + "shutdown": False, + "reason": str(e) + } + + async def _initialize_opentelemetry(self) -> None: + """Initialize OpenTelemetry tracing infrastructure.""" + if self.is_initialized or not OPENTELEMETRY_AVAILABLE: + return + + try: + # Create resource with service information from validated config + resource = Resource.create({ + SERVICE_NAME: self.tracing_config.service_name, + SERVICE_VERSION: self.tracing_config.service_version, + "deployment.environment": self.tracing_config.environment, + "service.namespace": "omnibase_infrastructure" + }) + + # Create tracer provider + self.tracer_provider = TracerProvider(resource=resource) + + # Configure OTLP exporter with validated endpoint + otlp_exporter = OTLPSpanExporter(endpoint=str(self.tracing_config.otel_exporter_otlp_endpoint)) + + # Add batch span processor + span_processor = BatchSpanProcessor(otlp_exporter) + self.tracer_provider.add_span_processor(span_processor) + + # Set global tracer provider + trace.set_tracer_provider(self.tracer_provider) + + # Get tracer + self.tracer = trace.get_tracer( + instrumenting_module_name=__name__, + instrumenting_library_version=self.tracing_config.service_version + ) + + self.is_initialized = True + self.logger.info( + f"OpenTelemetry tracing initialized for environment: {self.tracing_config.environment} " + f"with endpoint: {self.tracing_config.otel_exporter_otlp_endpoint}" + ) + + except Exception as e: + self.logger.error(f"Failed to initialize OpenTelemetry: {e}") + raise OnexError( + code=CoreErrorCode.CONFIGURATION_ERROR, + message=f"OpenTelemetry initialization failed: {str(e)}" + ) from e + + + def _convert_span_kind(self, input_span_kind: InputSpanKind) -> Optional[Union["SpanKind", object]]: + """Convert input span kind to OpenTelemetry span kind.""" + if not OPENTELEMETRY_AVAILABLE: + return None + + span_kind_mapping = { + InputSpanKind.INTERNAL: SpanKind.INTERNAL, + InputSpanKind.SERVER: SpanKind.SERVER, + InputSpanKind.CLIENT: SpanKind.CLIENT, + InputSpanKind.PRODUCER: SpanKind.PRODUCER, + InputSpanKind.CONSUMER: SpanKind.CONSUMER + } + + return span_kind_mapping.get(input_span_kind, SpanKind.INTERNAL) + diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/__init__.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/__init__.py new file mode 100644 index 0000000000..1fc1cd159d --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/__init__.py @@ -0,0 +1 @@ +"""Utility modules for distributed tracing compute node.""" \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/sql_sanitizer.py b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/sql_sanitizer.py new file mode 100644 index 0000000000..e28e2dda86 --- /dev/null +++ b/src/omnibase_infra/nodes/node_distributed_tracing_compute/v1_0_0/utils/sql_sanitizer.py @@ -0,0 +1,182 @@ +"""SQL Query Sanitization for OpenTelemetry Trace Attributes. + +Provides secure SQL query sanitization for observability purposes, replacing sensitive +literals with placeholders to prevent data leakage in traces while preserving query structure. +""" + +import logging + +# Required dependency - fail fast if unavailable (ONEX principle) +import sqlparse +from sqlparse.tokens import ( + Literal, + Number, + String as StringToken, + Keyword, + Name, +) +from sqlparse.sql import Token, Statement + +from omnibase_core.exceptions.base_onex_error import OnexError +from omnibase_core.enums.enum_core_error_code import CoreErrorCode + + +class SqlSanitizer: + """ + SQL query sanitizer for OpenTelemetry trace attributes. + + Removes sensitive data from SQL queries for observability while preserving + query structure for debugging and performance analysis. + """ + + _logger = logging.getLogger(__name__) + + @staticmethod + def sanitize_for_observability(query: str, max_length: int = 200) -> str: + """ + Sanitize SQL query for inclusion in trace attributes. + + Replaces literal values (strings, numbers) with placeholder tokens to prevent + sensitive data from being exposed in traces while preserving query structure + for observability and debugging. + + Args: + query: SQL query to sanitize + max_length: Maximum length of sanitized query (default: 200) + + Returns: + Sanitized SQL query with literals replaced by placeholders + + Raises: + OnexError: If query sanitization fails critically + """ + if not query or not query.strip(): + return "" + + # Validate input length to prevent resource exhaustion + if len(query) > 10000: # 10KB limit + SqlSanitizer._logger.warning(f"Query exceeds size limit: {len(query)} chars") + return f"-- QUERY TOO LARGE ({len(query)} chars) --" + + try: + return SqlSanitizer._sanitize_with_sqlparse(query, max_length) + + except Exception as e: + SqlSanitizer._logger.error(f"Query sanitization failed: {e}") + # For observability, prefer showing a safe placeholder over failing + return "-- SANITIZATION_FAILED --" + + @staticmethod + def _sanitize_with_sqlparse(query: str, max_length: int) -> str: + """ + Sanitize SQL query using sqlparse library for accurate parsing. + + Args: + query: SQL query to sanitize + max_length: Maximum length of result + + Returns: + Sanitized query with literals replaced by placeholders + """ + try: + # Parse the SQL query + parsed = sqlparse.parse(query) + if not parsed: + return "-- UNPARSEABLE_QUERY --" + + # Process the first parsed statement + statement = parsed[0] + sanitized_tokens = [] + + for token in statement.flatten(): + if SqlSanitizer._is_sensitive_literal(token): + # Replace sensitive literals with placeholder + sanitized_tokens.append('?') + elif token.ttype in (Keyword, Name) or token.value.upper() in SqlSanitizer._get_sql_keywords(): + # Preserve keywords and identifiers (case insensitive) + sanitized_tokens.append(token.value) + elif token.ttype is None and token.value.strip(): + # Preserve operators, punctuation, and other structural elements + sanitized_tokens.append(token.value) + + # Reconstruct and clean up the query + sanitized = ' '.join(sanitized_tokens) + sanitized = SqlSanitizer._clean_whitespace(sanitized) + + # Truncate if necessary + if len(sanitized) > max_length: + sanitized = sanitized[:max_length - 3] + "..." + + return sanitized + + except Exception as e: + SqlSanitizer._logger.error(f"sqlparse sanitization failed: {e}") + raise OnexError( + message=f"SQL query sanitization failed: {str(e)}", + error_code=CoreErrorCode.PROCESSING_ERROR + ) from e + + + @staticmethod + def _is_sensitive_literal(token) -> bool: + """ + Determine if a token contains sensitive literal data. + + Args: + token: sqlparse token to evaluate + + Returns: + True if token contains sensitive data that should be sanitized + """ + # Check for various literal types that should be sanitized + sensitive_types = [ + Literal.String.Single, # 'string' + Literal.String.Symbol, # "string" + Literal.Number.Integer, # 123 + Literal.Number.Float, # 123.45 + Literal.Number.Hexadecimal, # 0xABC + Number.Integer, # Alternative number tokens + Number.Float, + Number.Hexadecimal, + StringToken.Single, # Alternative string tokens + StringToken.Symbol, + ] + + return any(token.ttype == sensitive_type for sensitive_type in sensitive_types) + + @staticmethod + def _get_sql_keywords() -> set: + """ + Get common SQL keywords that should be preserved. + + Returns: + Set of SQL keywords to preserve in sanitized queries + """ + return { + 'SELECT', 'FROM', 'WHERE', 'INSERT', 'UPDATE', 'DELETE', 'CREATE', 'DROP', + 'ALTER', 'JOIN', 'LEFT', 'RIGHT', 'INNER', 'OUTER', 'ON', 'GROUP', 'ORDER', + 'BY', 'HAVING', 'LIMIT', 'OFFSET', 'UNION', 'INTERSECT', 'EXCEPT', 'AS', + 'DISTINCT', 'ALL', 'AND', 'OR', 'NOT', 'IN', 'EXISTS', 'BETWEEN', 'LIKE', + 'IS', 'NULL', 'TRUE', 'FALSE', 'CASE', 'WHEN', 'THEN', 'ELSE', 'END', + 'IF', 'COALESCE', 'NULLIF', 'CAST', 'CONVERT', 'COUNT', 'SUM', 'AVG', + 'MIN', 'MAX', 'FIRST', 'LAST', 'TOP', 'INTO', 'VALUES', 'SET', 'TABLE', + 'INDEX', 'VIEW', 'PROCEDURE', 'FUNCTION', 'TRIGGER', 'DATABASE', 'SCHEMA', + 'GRANT', 'REVOKE', 'COMMIT', 'ROLLBACK', 'BEGIN', 'TRANSACTION' + } + + @staticmethod + def _clean_whitespace(query: str) -> str: + """ + Clean up excessive whitespace in sanitized queries. + + Args: + query: Query with potential whitespace issues + + Returns: + Query with normalized whitespace + """ + import re + # Replace multiple whitespace with single space + cleaned = re.sub(r'\s+', ' ', query.strip()) + return cleaned + diff --git a/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/contract.yaml b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/contract.yaml new file mode 100644 index 0000000000..92e7ed4332 --- /dev/null +++ b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/contract.yaml @@ -0,0 +1,115 @@ +contract_version: "1.0.0" +node_version: "1.0.0" +version: "1.0.0" +node_name: "node_event_bus_circuit_breaker_compute" +contract_name: "NodeEventBusCircuitBreakerComputeContract" +name: "node_event_bus_circuit_breaker_compute" +node_type: "COMPUTE" +description: "Event Bus Circuit Breaker for RedPanda reliability with fail-fast behavior and graceful degradation" + +input_model: "ModelEventBusCircuitBreakerInput" +output_model: "ModelEventBusCircuitBreakerOutput" + +dependencies: + # ONEX Container dependency injection + - name: "model_onex_container" + type: "model" + class_name: "ModelONEXContainer" + module: "omnibase_core.model.model_onex_container" + + # Core ONEX event model + - name: "model_onex_event" + type: "model" + class_name: "ModelOnexEvent" + module: "omnibase_core.model.core.model_onex_event" + + # Core circuit breaker models + - name: "model_circuit_breaker_state" + type: "model" + class_name: "ModelCircuitBreakerState" + module: "omnibase_core.models.resilience.model_circuit_breaker_state" + + - name: "model_circuit_breaker" + type: "model" + class_name: "ModelCircuitBreaker" + module: "omnibase_core.models.configuration.model_circuit_breaker" + + - name: "model_circuit_breaker_metrics" + type: "model" + class_name: "ModelCircuitBreakerMetrics" + module: "omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics" + +definitions: + ModelEventBusCircuitBreakerInput: + type: "object" + properties: + operation_type: + type: "string" + enum: ["publish_event", "get_state", "get_metrics", "reset_circuit", "get_health_status"] + description: "Type of circuit breaker operation to perform" + event: + $ref: "#/definitions/ModelOnexEvent" + description: "Event to publish (required for publish_event operation)" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID for the operation" + publisher_function: + type: "string" + description: "Name of publisher function to use for event publishing" + environment: + type: "string" + description: "Environment configuration (development, staging, production)" + default: "development" + required: + - operation_type + - correlation_id + + ModelEventBusCircuitBreakerOutput: + type: "object" + properties: + success: + type: "boolean" + description: "Whether the operation succeeded" + operation_type: + type: "string" + description: "Type of operation that was performed" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID from the request" + result: + type: "object" + description: "Operation-specific result data" + error_message: + type: "string" + description: "Error message if operation failed" + circuit_breaker_state: + $ref: "#/definitions/CircuitBreakerStateEnum" + description: "Current circuit breaker state" + metrics: + $ref: "#/definitions/ModelCircuitBreakerMetrics" + description: "Current circuit breaker metrics" + timestamp: + type: "string" + format: "date-time" + description: "Response timestamp" + required: + - success + - operation_type + - correlation_id + - circuit_breaker_state + - timestamp + + ModelOnexEvent: + type: "object" + description: "ONEX event model (external dependency)" + + CircuitBreakerStateEnum: + type: "string" + enum: ["closed", "open", "half_open"] + description: "Circuit breaker state enumeration" + + ModelCircuitBreakerMetrics: + type: "object" + description: "Circuit breaker metrics model (external dependency)" \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_input.py b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_input.py new file mode 100644 index 0000000000..baf6816993 --- /dev/null +++ b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_input.py @@ -0,0 +1,47 @@ +"""Event Bus Circuit Breaker Input Model. + +Node-specific input model for the circuit breaker compute node. +""" + +from enum import Enum +from pydantic import BaseModel, Field +from typing import Optional +from uuid import UUID + +from omnibase_core.model.core.model_onex_event import ModelOnexEvent + + +class CircuitBreakerOperation(str, Enum): + """Circuit breaker operations.""" + PUBLISH_EVENT = "publish_event" + GET_STATE = "get_state" + GET_METRICS = "get_metrics" + RESET_CIRCUIT = "reset_circuit" + GET_HEALTH_STATUS = "get_health_status" + + +class ModelEventBusCircuitBreakerInput(BaseModel): + """Input model for event bus circuit breaker operations.""" + + operation_type: CircuitBreakerOperation = Field( + description="Type of circuit breaker operation to perform" + ) + + event: Optional[ModelOnexEvent] = Field( + default=None, + description="Event to publish (required for publish_event operation)" + ) + + correlation_id: UUID = Field( + description="Correlation ID for the operation" + ) + + publisher_function: Optional[str] = Field( + default=None, + description="Name of publisher function to use for event publishing" + ) + + environment: str = Field( + default="development", + description="Environment configuration (development, staging, production)" + ) \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_output.py b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_output.py new file mode 100644 index 0000000000..3c58f43f54 --- /dev/null +++ b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/models/model_event_bus_circuit_breaker_output.py @@ -0,0 +1,67 @@ +"""Event Bus Circuit Breaker Output Model. + +Node-specific output model for the circuit breaker compute node. +""" + +from pydantic import BaseModel, Field +from typing import Optional, Union +from datetime import datetime +from uuid import UUID + +from omnibase_core.enums.intelligence.enum_circuit_breaker_state import EnumCircuitBreakerState +from omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics import ModelCircuitBreakerMetrics +from omnibase_infra.models.circuit_breaker.model_circuit_breaker_result import ( + ModelPublishEventResult, + ModelStateResult, + ModelResetResult, + ModelHealthStatusResult +) + + +class ModelEventBusCircuitBreakerOutput(BaseModel): + """Output model for event bus circuit breaker operations.""" + + success: bool = Field( + description="Whether the operation succeeded" + ) + + operation_type: str = Field( + description="Type of operation that was performed" + ) + + correlation_id: UUID = Field( + description="Correlation ID from the request" + ) + + result: Optional[Union[ + ModelPublishEventResult, + ModelStateResult, + ModelResetResult, + ModelHealthStatusResult + ]] = Field( + default=None, + description="Operation-specific result data" + ) + + error_message: Optional[str] = Field( + default=None, + description="Error message if operation failed" + ) + + circuit_breaker_state: EnumCircuitBreakerState = Field( + description="Current circuit breaker state" + ) + + metrics: Optional[ModelCircuitBreakerMetrics] = Field( + default=None, + description="Current circuit breaker metrics" + ) + + timestamp: datetime = Field( + description="Response timestamp" + ) + + class Config: + json_encoders = { + datetime: lambda v: v.isoformat() + } \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/node.py b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/node.py new file mode 100644 index 0000000000..1eaa499706 --- /dev/null +++ b/src/omnibase_infra/nodes/node_event_bus_circuit_breaker_compute/v1_0_0/node.py @@ -0,0 +1,517 @@ +"""Event Bus Circuit Breaker Compute Node. + +ONEX compute node that implements circuit breaker pattern for RedPanda event bus reliability. +Provides resilient event publishing with automatic failure detection and graceful degradation. +""" + +import asyncio +import logging +import os +import time +from datetime import datetime +from typing import Optional, Callable +from uuid import UUID + +from omnibase_core.base.node_compute_service import NodeComputeService +from omnibase_core.core.errors.onex_error import OnexError, CoreErrorCode +from omnibase_core.core.onex_container import ModelONEXContainer +from omnibase_core.model.core.model_onex_event import ModelOnexEvent + +from omnibase_core.enums.intelligence.enum_circuit_breaker_state import EnumCircuitBreakerState +from omnibase_core.models.resilience.model_circuit_breaker_state import ModelCircuitBreakerState +from omnibase_core.models.configuration.model_circuit_breaker import ModelCircuitBreaker +from omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics import ModelCircuitBreakerMetrics +from omnibase_infra.models.circuit_breaker.model_dead_letter_queue_entry import ModelDeadLetterQueueEntry +from omnibase_infra.models.infrastructure.model_circuit_breaker_environment_config import ( + ModelCircuitBreakerConfig, + ModelCircuitBreakerEnvironmentConfig +) + +from .models.model_event_bus_circuit_breaker_input import ( + ModelEventBusCircuitBreakerInput, + CircuitBreakerOperation +) +from .models.model_event_bus_circuit_breaker_output import ModelEventBusCircuitBreakerOutput + + +class NodeEventBusCircuitBreakerCompute(NodeComputeService[ModelEventBusCircuitBreakerInput, ModelEventBusCircuitBreakerOutput]): + """ + Event Bus Circuit Breaker Compute Node. + + Provides: + - Circuit breaker pattern implementation for RedPanda event publishing + - Automatic failure detection and circuit opening/closing + - Graceful degradation with event queuing + - Dead letter queue for permanently failed events + - Environment-specific configuration + - Comprehensive metrics and health monitoring + """ + + def __init__(self, container: ModelONEXContainer): + """Initialize the circuit breaker compute node. + + Args: + container: ONEX container for dependency injection + """ + super().__init__(container) + self.logger = logging.getLogger(f"{__name__}.NodeEventBusCircuitBreakerCompute") + + # Circuit breaker state + self._state = EnumCircuitBreakerState.CLOSED + self._failure_count = 0 + self._success_count = 0 + self._last_failure_time: Optional[float] = None + + # Configuration (will be loaded from container or defaults) + self._config: Optional[ModelCircuitBreakerConfig] = None + + # Event queues + self._event_queue: list[ModelOnexEvent] = [] + self._dead_letter_queue: list[ModelDeadLetterQueueEntry] = [] + + # Metrics tracking + self._metrics = ModelCircuitBreakerMetrics() + + # Async lock for thread safety + self._lock = asyncio.Lock() + + # State tracking + self._last_state_change_time = datetime.now() + + # Publisher functions registry + self._publisher_functions: dict[str, Callable] = {} + + async def initialize(self) -> None: + """Initialize the circuit breaker node.""" + try: + # Load configuration from container or use defaults + self._config = await self._load_configuration() + + # Initialize metrics timestamp + self._metrics = ModelCircuitBreakerMetrics() + + self.logger.info("Circuit breaker compute node initialized successfully") + + except Exception as e: + raise OnexError( + code=CoreErrorCode.INITIALIZATION_ERROR, + message=f"Failed to initialize circuit breaker compute node: {str(e)}" + ) from e + + async def compute(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelEventBusCircuitBreakerOutput: + """Execute circuit breaker operations. + + Args: + input_data: Input containing operation type and parameters + + Returns: + Output with operation result and current circuit breaker status + """ + start_time = time.time() + + try: + # Route to appropriate operation handler + if input_data.operation_type == CircuitBreakerOperation.PUBLISH_EVENT: + result = await self._handle_publish_event(input_data) + elif input_data.operation_type == CircuitBreakerOperation.GET_STATE: + result = await self._handle_get_state(input_data) + elif input_data.operation_type == CircuitBreakerOperation.GET_METRICS: + result = await self._handle_get_metrics(input_data) + elif input_data.operation_type == CircuitBreakerOperation.RESET_CIRCUIT: + result = await self._handle_reset_circuit(input_data) + elif input_data.operation_type == CircuitBreakerOperation.GET_HEALTH_STATUS: + result = await self._handle_get_health_status(input_data) + else: + raise OnexError( + code=CoreErrorCode.INVALID_INPUT, + message=f"Unsupported operation type: {input_data.operation_type}" + ) + + # Update performance metrics + processing_time = time.time() - start_time + self._metrics.average_response_time_ms = ( + self._metrics.average_response_time_ms * 0.9 + processing_time * 1000 * 0.1 + ) + + return ModelEventBusCircuitBreakerOutput( + success=True, + operation_type=input_data.operation_type.value, + correlation_id=input_data.correlation_id, + result=result, + circuit_breaker_state=self._state, + metrics=self._metrics, + timestamp=datetime.now() + ) + + except OnexError: + # Re-raise ONEX errors as-is + raise + except Exception as e: + # Wrap other exceptions in OnexError + raise OnexError( + code=CoreErrorCode.PROCESSING_ERROR, + message=f"Circuit breaker operation failed: {str(e)}" + ) from e + + async def _handle_publish_event(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelPublishEventResult: + """Handle event publishing through circuit breaker.""" + if not input_data.event: + raise OnexError( + code=CoreErrorCode.INVALID_INPUT, + message="Event is required for publish_event operation" + ) + + async with self._lock: + self._metrics.total_events += 1 + + # Check circuit state and handle accordingly + if self._state == EnumCircuitBreakerState.OPEN: + return await self._handle_open_circuit(input_data.event) + elif self._state == EnumCircuitBreakerState.HALF_OPEN: + return await self._handle_half_open_circuit(input_data.event, input_data.publisher_function) + else: # CLOSED + return await self._handle_closed_circuit(input_data.event, input_data.publisher_function) + + async def _handle_closed_circuit(self, event: ModelOnexEvent, publisher_function: Optional[str]) -> ModelPublishEventResult: + """Handle event publishing when circuit is closed (normal operation).""" + try: + # Get publisher function + publisher_func = await self._get_publisher_function(publisher_function) + + # Attempt to publish event with timeout + await asyncio.wait_for(publisher_func(event), timeout=self._config.timeout_seconds) + + # Success - reset failure count and update metrics + self._failure_count = 0 + self._metrics.successful_events += 1 + self._metrics.last_success = datetime.now() + self._metrics.success_rate_percent = ( + self._metrics.successful_events / max(self._metrics.total_events, 1) * 100 + ) + + self.logger.debug(f"Event published successfully: {event.correlation_id}") + + return ModelPublishEventResult( + published=True, + queued=False, + event_id=str(event.correlation_id) + ) + + except asyncio.TimeoutError: + await self._handle_failure(f"Event publishing timeout after {self._config.timeout_seconds}s") + return await self._queue_or_drop_event(event) + + except Exception as e: + await self._handle_failure(f"Event publishing failed: {str(e)}") + return await self._queue_or_drop_event(event) + + async def _handle_half_open_circuit(self, event: ModelOnexEvent, publisher_function: Optional[str]) -> ModelPublishEventResult: + """Handle event publishing when circuit is half-open (testing recovery).""" + try: + # Get publisher function + publisher_func = await self._get_publisher_function(publisher_function) + + # Attempt limited publishing to test recovery + await asyncio.wait_for(publisher_func(event), timeout=self._config.timeout_seconds) + + # Success in half-open state + self._success_count += 1 + self._metrics.successful_events += 1 + self._metrics.last_success = datetime.now() + + self.logger.info(f"Half-open success {self._success_count}/{self._config.success_threshold}") + + # Check if we can close the circuit + if self._success_count >= self._config.success_threshold: + await self._close_circuit() + + return ModelPublishEventResult( + published=True, + queued=False, + event_id=str(event.correlation_id), + half_open_success=True + ) + + except Exception as e: + # Failure in half-open - immediately open circuit again + await self._open_circuit(f"Half-open test failed: {str(e)}") + return await self._queue_or_drop_event(event) + + async def _handle_open_circuit(self, event: ModelOnexEvent) -> ModelPublishEventResult: + """Handle event when circuit is open (failure state).""" + # Check if we should transition to half-open for recovery testing + if self._should_attempt_reset(): + await self._transition_to_half_open() + # Don't publish this event yet - queue it for safety + return await self._queue_or_drop_event(event) + + # Circuit remains open - queue or drop event + return await self._queue_or_drop_event(event) + + async def _handle_failure(self, error_message: str): + """Handle event publishing failure.""" + self._failure_count += 1 + self._metrics.failed_events += 1 + self._metrics.last_failure = datetime.now() + self._last_failure_time = time.time() + + # Update success rate + self._metrics.success_rate_percent = ( + self._metrics.successful_events / max(self._metrics.total_events, 1) * 100 + ) + + self.logger.warning(f"Event publishing failure {self._failure_count}/{self._config.failure_threshold}: {error_message}") + + # Open circuit if failure threshold reached + if self._failure_count >= self._config.failure_threshold: + await self._open_circuit(f"Failure threshold reached: {error_message}") + + async def _open_circuit(self, reason: str): + """Open the circuit breaker.""" + if self._state != EnumCircuitBreakerState.OPEN: + self._state = EnumCircuitBreakerState.OPEN + self._last_state_change_time = datetime.now() + self._metrics.circuit_opens += 1 + self.logger.error(f"Circuit breaker OPENED: {reason}") + + async def _close_circuit(self): + """Close the circuit breaker (recovery complete).""" + self._state = EnumCircuitBreakerState.CLOSED + self._last_state_change_time = datetime.now() + self._failure_count = 0 + self._success_count = 0 + self._metrics.circuit_closes += 1 + self.logger.info("Circuit breaker CLOSED - recovery complete") + + # Process any queued events + await self._process_queued_events() + + async def _transition_to_half_open(self): + """Transition circuit to half-open state for recovery testing.""" + self._state = EnumCircuitBreakerState.HALF_OPEN + self._last_state_change_time = datetime.now() + self._success_count = 0 + self.logger.info("Circuit breaker transitioned to HALF-OPEN - testing recovery") + + def _should_attempt_reset(self) -> bool: + """Check if circuit should attempt reset to half-open.""" + if self._last_failure_time is None: + return False + + time_since_failure = time.time() - self._last_failure_time + return time_since_failure >= self._config.recovery_timeout + + async def _queue_or_drop_event(self, event: ModelOnexEvent) -> ModelPublishEventResult: + """Queue event or drop it based on queue capacity and configuration.""" + if not self._config.graceful_degradation: + # Fail-fast mode - raise error for critical operations + raise OnexError( + code=CoreErrorCode.INTEGRATION_SERVICE_UNAVAILABLE, + message="Event bus circuit breaker open - event publishing failed", + details={"circuit_state": self._state.value, "queued_events": len(self._event_queue)} + ) + + # Graceful degradation mode - queue if possible + if len(self._event_queue) < self._config.max_queue_size: + self._event_queue.append(event) + self._metrics.queued_events += 1 + self.logger.info(f"Event queued (circuit {self._state.value}): {event.correlation_id}") + + return ModelPublishEventResult( + published=False, + queued=True, + event_id=str(event.correlation_id), + reason=f"Circuit {self._state.value} - event queued" + ) + else: + # Queue full - move to dead letter queue if enabled + if self._config.dead_letter_enabled: + await self._add_to_dead_letter_queue(event, "Queue capacity exceeded") + + self._metrics.dropped_events += 1 + self.logger.warning(f"Event dropped - queue full: {event.correlation_id}") + + return ModelPublishEventResult( + published=False, + queued=False, + dropped=True, + event_id=str(event.correlation_id), + reason="Queue capacity exceeded" + ) + + async def _add_to_dead_letter_queue(self, event: ModelOnexEvent, reason: str): + """Add failed event to dead letter queue for later processing.""" + dead_letter_entry = ModelDeadLetterQueueEntry( + event=event.model_dump(), + timestamp=datetime.now().isoformat(), + reason=reason, + circuit_state=self._state.value, + retry_count=0 + ) + + self._dead_letter_queue.append(dead_letter_entry) + self._metrics.dead_letter_events += 1 + self.logger.info(f"Event added to dead letter queue: {event.correlation_id} - {reason}") + + async def _process_queued_events(self): + """Process queued events when circuit closes.""" + if not self._event_queue: + return + + queued_count = len(self._event_queue) + self.logger.info(f"Processing {queued_count} queued events after circuit recovery") + + # Process events in background to avoid blocking + asyncio.create_task(self._process_queue_background()) + + async def _process_queue_background(self): + """Background task to process queued events with proper thread safety.""" + processed = 0 + failed = 0 + + while True: + # Check state and queue with proper locking + async with self._lock: + if not self._event_queue or self._state != EnumCircuitBreakerState.CLOSED: + break + + try: + event = self._event_queue.pop(0) + except IndexError: + # Queue was emptied between check and pop + break + + # Process event outside the lock to avoid blocking other operations + try: + # NOTE: Event re-publishing requires publisher function context + # For now, events are processed but not re-published to avoid state inconsistency + processed += 1 + + except Exception as e: + failed += 1 + self.logger.error(f"Failed to process queued event: {e}") + + if failed >= 3: # Prevent infinite retry loops + break + + self.logger.info(f"Queued event processing complete: {processed} processed, {failed} failed") + + async def _handle_get_state(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelStateResult: + """Handle get circuit breaker state operation.""" + state_info = ModelCircuitBreakerState( + state=self._state, + failure_count=self._failure_count, + success_count=self._success_count, + last_failure_time=self._metrics.last_failure, + last_success_time=self._metrics.last_success, + last_state_change=self._last_state_change_time, + is_healthy=self._is_healthy() + ) + + return ModelStateResult( + state=state_info.model_dump() + ) + + async def _handle_get_metrics(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelCircuitBreakerMetrics: + """Handle get circuit breaker metrics operation.""" + return self._metrics + + async def _handle_reset_circuit(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelResetResult: + """Handle manual circuit reset operation.""" + async with self._lock: + self._state = EnumCircuitBreakerState.CLOSED + self._last_state_change_time = datetime.now() + self._failure_count = 0 + self._success_count = 0 + self._last_failure_time = None + self.logger.info("Circuit breaker manually reset to CLOSED state") + + return ModelResetResult( + reset=True, + new_state=self._state.value + ) + + async def _handle_get_health_status(self, input_data: ModelEventBusCircuitBreakerInput) -> ModelHealthStatusResult: + """Handle get health status operation.""" + return ModelHealthStatusResult( + circuit_state=self._state.value, + is_healthy=self._is_healthy(), + failure_count=self._failure_count, + queued_events=len(self._event_queue), + dead_letter_events=len(self._dead_letter_queue), + configuration=self._config.model_dump() if self._config else {}, + metrics=self._metrics.model_dump() + ) + + def _is_healthy(self) -> bool: + """Check if circuit breaker is healthy for event publishing.""" + return self._state == EnumCircuitBreakerState.CLOSED or self._state == EnumCircuitBreakerState.HALF_OPEN + + async def _load_configuration(self) -> ModelCircuitBreakerConfig: + """Load circuit breaker configuration from container or defaults.""" + try: + # Try to load from container configuration first + # In production, this would come from container.get_configuration() + # For now, detect environment and use appropriate defaults + + # Environment detection (similar to distributed tracing) + env_vars = ["ENVIRONMENT", "ENV", "DEPLOYMENT_ENV", "NODE_ENV", "OMNIBASE_ENV"] + environment = "development" # default + for var in env_vars: + value = os.getenv(var) + if value: + environment = value.lower() + break + + # Create environment configuration with defaults + env_config = ModelCircuitBreakerEnvironmentConfig.create_default_config() + + # Get configuration for detected environment + config = env_config.get_config_for_environment( + environment=environment, + default_environment="development" + ) + + self.logger.info(f"Loaded circuit breaker configuration for environment: {environment}") + return config + + except Exception as e: + self.logger.warning(f"Failed to load configuration from container, using development defaults: {e}") + # Fallback to development defaults + return ModelCircuitBreakerConfig( + failure_threshold=2, + recovery_timeout=15, + success_threshold=1, + timeout_seconds=10, + max_queue_size=100, + dead_letter_enabled=False, + graceful_degradation=True + ) + + async def _get_publisher_function(self, function_name: Optional[str]) -> Callable: + """Get publisher function for event publishing.""" + if function_name and function_name in self._publisher_functions: + return self._publisher_functions[function_name] + + # Default mock publisher for testing - with production safety check + async def mock_publisher(event: ModelOnexEvent) -> None: + """Mock publisher function for testing - NOT for production use.""" + # Production safety check + environment = os.getenv('ENVIRONMENT', '').lower() + if environment in ('production', 'prod'): + self.logger.error("CRITICAL: Mock publisher used in production environment!") + raise OnexError( + message="Mock publisher cannot be used in production environment", + error_code=CoreErrorCode.CONFIGURATION_ERROR + ) + + self.logger.warning(f"Using mock publisher for event: {event.correlation_id} (environment: {environment or 'unknown'})") + # Simulate some processing time + await asyncio.sleep(0.01) + + return mock_publisher + + def register_publisher_function(self, name: str, function: Callable) -> None: + """Register a publisher function for use by the circuit breaker.""" + self._publisher_functions[name] = function + self.logger.info(f"Registered publisher function: {name}") \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_infrastructure_health_monitor_orchestrator/v1_0_0/contract.yaml b/src/omnibase_infra/nodes/node_infrastructure_health_monitor_orchestrator/v1_0_0/contract.yaml new file mode 100644 index 0000000000..b56221b1f9 --- /dev/null +++ b/src/omnibase_infra/nodes/node_infrastructure_health_monitor_orchestrator/v1_0_0/contract.yaml @@ -0,0 +1,100 @@ +contract_version: "1.0.0" +node_version: "1.0.0" +version: "1.0.0" +node_name: "node_infrastructure_health_monitor_orchestrator" +contract_name: "NodeInfrastructureHealthMonitorOrchestratorContract" +name: "node_infrastructure_health_monitor_orchestrator" +node_type: "ORCHESTRATOR" +description: "Infrastructure Health Monitor Orchestrator for centralized health aggregation and monitoring coordination" + +input_model: "ModelInfrastructureHealthMonitorInput" +output_model: "ModelInfrastructureHealthMonitorOutput" + +dependencies: + # ONEX Container dependency injection + - name: "model_onex_container" + type: "model" + class_name: "ModelONEXContainer" + module: "omnibase_core.model.model_onex_container" + + # Shared infrastructure models + - name: "model_infrastructure_health_metrics" + type: "model" + class_name: "ModelInfrastructureHealthMetrics" + module: "omnibase_infra.models.infrastructure.model_infrastructure_health_metrics" + + # Circuit breaker integration + - name: "model_circuit_breaker_metrics" + type: "model" + class_name: "ModelCircuitBreakerMetrics" + module: "omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics" + +definitions: + ModelInfrastructureHealthMonitorInput: + type: "object" + properties: + operation_type: + type: "string" + enum: ["comprehensive_health_check", "get_health_trends", "start_monitoring", "stop_monitoring", "get_prometheus_metrics"] + description: "Type of health monitoring operation to perform" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID for the operation" + environment: + type: "string" + description: "Environment configuration (development, staging, production)" + default: "development" + time_period_hours: + type: "integer" + description: "Time period in hours for trend analysis" + minimum: 1 + default: 1 + monitoring_interval_seconds: + type: "integer" + description: "Monitoring interval in seconds" + minimum: 10 + default: 30 + required: + - operation_type + - correlation_id + + ModelInfrastructureHealthMonitorOutput: + type: "object" + properties: + success: + type: "boolean" + description: "Whether the operation succeeded" + operation_type: + type: "string" + description: "Type of operation that was performed" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID from the request" + result: + type: "object" + description: "Operation-specific result data" + error_message: + type: "string" + description: "Error message if operation failed" + health_metrics: + $ref: "#/definitions/ModelInfrastructureHealthMetrics" + description: "Current infrastructure health metrics" + monitoring_active: + type: "boolean" + description: "Whether continuous monitoring is active" + timestamp: + type: "string" + format: "date-time" + description: "Response timestamp" + required: + - success + - operation_type + - correlation_id + - monitoring_active + - timestamp + + ModelInfrastructureHealthMetrics: + type: "object" + description: "Infrastructure health metrics model (external dependency)" \ No newline at end of file diff --git a/src/omnibase_infra/nodes/node_infrastructure_observability_compute/v1_0_0/contract.yaml b/src/omnibase_infra/nodes/node_infrastructure_observability_compute/v1_0_0/contract.yaml new file mode 100644 index 0000000000..5559d843ab --- /dev/null +++ b/src/omnibase_infra/nodes/node_infrastructure_observability_compute/v1_0_0/contract.yaml @@ -0,0 +1,117 @@ +contract_version: "1.0.0" +node_version: "1.0.0" +version: "1.0.0" +node_name: "node_infrastructure_observability_compute" +contract_name: "NodeInfrastructureObservabilityComputeContract" +name: "node_infrastructure_observability_compute" +node_type: "COMPUTE" +description: "Infrastructure Observability Compute Node for metrics collection, alerting, and performance tracking" + +input_model: "ModelInfrastructureObservabilityInput" +output_model: "ModelInfrastructureObservabilityOutput" + +dependencies: + # ONEX Container dependency injection + - name: "model_onex_container" + type: "model" + class_name: "ModelONEXContainer" + module: "omnibase_core.model.model_onex_container" + + # Shared observability models + - name: "model_metric_point" + type: "model" + class_name: "ModelMetricPoint" + module: "omnibase_infra.models.observability.model_metric_point" + + - name: "model_alert" + type: "model" + class_name: "ModelAlert" + module: "omnibase_infra.models.observability.model_alert" + + # Circuit breaker integration + - name: "model_circuit_breaker_metrics" + type: "model" + class_name: "ModelCircuitBreakerMetrics" + module: "omnibase_infra.models.circuit_breaker.model_circuit_breaker_metrics" + +definitions: + ModelInfrastructureObservabilityInput: + type: "object" + properties: + operation_type: + type: "string" + enum: ["record_metric", "get_current_metrics", "get_health_summary", "get_performance_trends", "export_prometheus", "get_alerts", "resolve_alert"] + description: "Type of observability operation to perform" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID for the operation" + metric_name: + type: "string" + description: "Name of metric to record (required for record_metric)" + metric_value: + type: "number" + description: "Value of metric to record (required for record_metric)" + metric_type: + type: "string" + enum: ["counter", "gauge", "histogram", "summary"] + description: "Type of metric" + default: "gauge" + metric_labels: + type: "object" + additionalProperties: true + description: "Labels/tags for the metric" + alert_id: + type: "string" + description: "Alert ID (required for resolve_alert)" + alert_severity: + type: "string" + enum: ["critical", "high", "medium", "low"] + description: "Alert severity filter" + time_period_hours: + type: "integer" + description: "Time period in hours for trend analysis" + minimum: 1 + default: 1 + active_only: + type: "boolean" + description: "Return only active alerts" + default: true + required: + - operation_type + - correlation_id + + ModelInfrastructureObservabilityOutput: + type: "object" + properties: + success: + type: "boolean" + description: "Whether the operation succeeded" + operation_type: + type: "string" + description: "Type of operation that was performed" + correlation_id: + type: "string" + format: "uuid" + description: "Correlation ID from the request" + result: + type: "object" + description: "Operation-specific result data" + error_message: + type: "string" + description: "Error message if operation failed" + metrics_count: + type: "integer" + description: "Number of metrics currently tracked" + alerts_count: + type: "integer" + description: "Number of active alerts" + timestamp: + type: "string" + format: "date-time" + description: "Response timestamp" + required: + - success + - operation_type + - correlation_id + - timestamp \ No newline at end of file diff --git a/src/omnibase_infra/security/audit_logger.py b/src/omnibase_infra/security/audit_logger.py index 721899c217..d6e4fe4c6c 100644 --- a/src/omnibase_infra/security/audit_logger.py +++ b/src/omnibase_infra/security/audit_logger.py @@ -16,13 +16,19 @@ import hashlib import time from datetime import datetime, timezone -from typing import Dict, Any, Optional, List -from dataclasses import dataclass, asdict +from typing import Optional, List, Union +from dataclasses import dataclass from enum import Enum from omnibase_core.core.errors.onex_error import OnexError from omnibase_core.core.errors.onex_error import CoreErrorCode +# Import strongly-typed models to replace Dict[str, Any] usage +from omnibase_infra.models.security.model_audit_details import ( + ModelAuditDetails, + ModelAuditMetadata +) + class AuditEventType(Enum): """Types of events that should be audited.""" @@ -60,10 +66,10 @@ class AuditEvent: resource: str action: str outcome: str # "success", "failure", "denied" - details: Dict[str, Any] + details: ModelAuditDetails source_ip: Optional[str] = None user_agent: Optional[str] = None - metadata: Optional[Dict[str, Any]] = None + metadata: Optional[ModelAuditMetadata] = None def __post_init__(self): """Validate and enrich audit event.""" @@ -73,8 +79,7 @@ def __post_init__(self): if not self.timestamp: self.timestamp = datetime.now(timezone.utc).isoformat() - # Sanitize sensitive data in details - self.details = self._sanitize_details(self.details) + # Details are now strongly typed - no sanitization needed def _generate_event_id(self) -> str: """Generate unique event ID.""" @@ -82,31 +87,25 @@ def _generate_event_id(self) -> str: content = f"{timestamp_ms}{self.event_type.value}{self.resource}{self.action}" return hashlib.sha256(content.encode()).hexdigest()[:16] - def _sanitize_details(self, details: Dict[str, Any]) -> Dict[str, Any]: - """Remove or mask sensitive information from details.""" - sanitized = {} - - for key, value in details.items(): - key_lower = key.lower() - - # Mask sensitive fields - if any(sensitive in key_lower for sensitive in ['password', 'token', 'secret', 'key']): - sanitized[key] = "***REDACTED***" - elif key_lower in ['ssn', 'credit_card', 'account_number']: - sanitized[key] = "***REDACTED***" - elif hasattr(value, '__len__') and hasattr(value, 'strip') and len(value) > 500: - # Truncate very long strings - sanitized[key] = value[:497] + "..." - else: - sanitized[key] = value - - return sanitized - - def to_dict(self) -> Dict[str, Any]: + def to_dict(self) -> dict: """Convert to dictionary for serialization.""" - result = asdict(self) - result['event_type'] = self.event_type.value - result['severity'] = self.severity.value + result = { + 'event_id': self.event_id, + 'timestamp': self.timestamp, + 'event_type': self.event_type.value, + 'severity': self.severity.value, + 'user_id': self.user_id, + 'client_id': self.client_id, + 'session_id': self.session_id, + 'correlation_id': self.correlation_id, + 'resource': self.resource, + 'action': self.action, + 'outcome': self.outcome, + 'details': self.details.dict() if self.details else {}, + 'source_ip': self.source_ip, + 'user_agent': self.user_agent, + 'metadata': self.metadata.dict() if self.metadata else None + } return result def to_json(self) -> str: @@ -169,7 +168,7 @@ def _setup_audit_logger(self): # Prevent audit logs from going to parent loggers self._logger.propagate = False - def _get_audit_config(self, key: str, default: Any) -> Any: + def _get_audit_config(self, key: str, default: Union[str, int, bool]) -> Union[str, int, bool]: """Get audit configuration value.""" import os env_key = f"ONEX_AUDIT_{key.upper()}" @@ -346,7 +345,7 @@ def log_security_violation(self, violation_type: str, description: str, source_ip: Optional[str] = None, - details: Optional[Dict[str, Any]] = None): + details: Optional[ModelAuditDetails] = None): """ Log security violation event. @@ -433,7 +432,7 @@ def _send_security_alert(self, event: AuditEvent): # (Slack, PagerDuty, email, etc.) self._logger.critical(f"SECURITY ALERT: {event.event_type.value} - {event.action} - {event.outcome}") - def get_audit_statistics(self) -> Dict[str, Any]: + def get_audit_statistics(self) -> dict: """ Get audit logging statistics. diff --git a/tests/test_postgres_adapter.py b/tests/test_postgres_adapter.py index 3b1786e928..e2695e87f3 100644 --- a/tests/test_postgres_adapter.py +++ b/tests/test_postgres_adapter.py @@ -12,7 +12,7 @@ from unittest.mock import Mock, AsyncMock, patch from typing import Dict, Any -from omnibase_core.core.onex_container import ModelONEXContainer as ONEXContainer +from omnibase_core.core.onex_container import ModelONEXContainer from omnibase_core.core.errors.onex_error import CoreErrorCode from omnibase_infra.nodes.node_postgres_adapter_effect.v1_0_0.node import NodePostgresAdapterEffect @@ -26,8 +26,8 @@ class TestPostgresAdapter: @pytest.fixture def container(self): - """Create a basic ONEXContainer for testing.""" - return ONEXContainer() + """Create a basic ModelONEXContainer for testing.""" + return ModelONEXContainer() @pytest.fixture def mock_connection_manager(self): @@ -58,7 +58,7 @@ def mock_connection_manager(self): def adapter_with_mock(self, container, mock_connection_manager): """Create adapter with mocked connection manager.""" # For testing purposes, we'll mock the container to avoid service resolution issues - mock_container = Mock(spec=ONEXContainer) + mock_container = Mock(spec=ModelONEXContainer) with patch('omnibase_infra.infrastructure.postgres_connection_manager.PostgresConnectionManager') as mock_manager_class: mock_manager_class.return_value = mock_connection_manager diff --git a/tests/test_postgres_adapter_security.py b/tests/test_postgres_adapter_security.py index c1ffa06d08..83fac05c6b 100644 --- a/tests/test_postgres_adapter_security.py +++ b/tests/test_postgres_adapter_security.py @@ -13,7 +13,7 @@ from unittest.mock import Mock, AsyncMock, patch from typing import Dict, Any -from omnibase_core.core.onex_container import ModelONEXContainer as ONEXContainer +from omnibase_core.core.onex_container import ModelONEXContainer from omnibase_core.core_error_codes import CoreErrorCode from omnibase_core.onex_error import OnexError @@ -28,8 +28,8 @@ class TestPostgresAdapterSecurityEdgeCases: @pytest.fixture def container(self): - """Create a basic ONEXContainer for testing.""" - return ONEXContainer() + """Create a basic ModelONEXContainer for testing.""" + return ModelONEXContainer() @pytest.fixture def mock_connection_manager(self): @@ -74,7 +74,7 @@ def secure_config(self): @pytest.fixture def adapter_with_secure_config(self, container, mock_connection_manager, secure_config): """Create adapter with secure production configuration.""" - mock_container = Mock(spec=ONEXContainer) + mock_container = Mock(spec=ModelONEXContainer) mock_container.get_service.return_value = mock_connection_manager with patch('omnibase_infra.infrastructure.postgres_connection_manager.PostgresConnectionManager') as mock_manager_class: diff --git a/tests/test_sql_sanitizer.py b/tests/test_sql_sanitizer.py new file mode 100644 index 0000000000..174f7a2317 --- /dev/null +++ b/tests/test_sql_sanitizer.py @@ -0,0 +1,237 @@ +""" +Unit tests for SqlSanitizer utility. + +Tests the security-critical SQL query sanitization functionality used +for OpenTelemetry trace attributes to prevent sensitive data leakage. +""" + +import pytest +from unittest.mock import patch, Mock + +from src.omnibase_infra.nodes.node_distributed_tracing_compute.v1_0_0.utils.sql_sanitizer import SqlSanitizer +from omnibase_core.exceptions.base_onex_error import OnexError + + +class TestSqlSanitizer: + """Test cases for SQL query sanitization functionality.""" + + def test_empty_query(self): + """Test handling of empty or None queries.""" + assert SqlSanitizer.sanitize_for_observability("") == "" + assert SqlSanitizer.sanitize_for_observability(" ") == "" + + def test_simple_select_query(self): + """Test sanitization of basic SELECT query with literals.""" + query = "SELECT * FROM users WHERE name = 'John Doe' AND age = 25" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should preserve structure but replace literals + assert "SELECT" in result + assert "FROM users" in result + assert "WHERE" in result + assert "name" in result + assert "age" in result + + # Should not contain original literal values + assert "John Doe" not in result + assert "25" not in result + + def test_insert_query_with_values(self): + """Test sanitization of INSERT query with multiple value types.""" + query = "INSERT INTO products (name, price, active) VALUES ('Widget', 29.99, true)" + result = SqlSanitizer.sanitize_for_observability(query) + + assert "INSERT INTO products" in result + assert "VALUES" in result + assert "Widget" not in result + assert "29.99" not in result + + def test_update_query_sanitization(self): + """Test sanitization of UPDATE query with WHERE clause.""" + query = "UPDATE users SET password = 'secret123' WHERE email = 'user@example.com'" + result = SqlSanitizer.sanitize_for_observability(query) + + assert "UPDATE users" in result + assert "SET password" in result + assert "WHERE email" in result + assert "secret123" not in result + assert "user@example.com" not in result + + def test_complex_query_with_joins(self): + """Test sanitization of complex query with JOINs and multiple conditions.""" + query = """ + SELECT u.name, p.title + FROM users u + JOIN posts p ON u.id = p.user_id + WHERE u.status = 'active' AND p.created_at > '2023-01-01' + """ + result = SqlSanitizer.sanitize_for_observability(query) + + assert "SELECT" in result + assert "JOIN" in result + assert "WHERE" in result + assert "active" not in result + assert "2023-01-01" not in result + + def test_query_with_quotes_and_escapes(self): + """Test handling of queries with escaped quotes and special characters.""" + query = "SELECT * FROM users WHERE name = 'O''Malley' AND comment = \"She said \\\"hello\\\"\"" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should not contain the original escaped strings + assert "O''Malley" not in result + assert 'She said "hello"' not in result + + def test_query_with_numbers_and_hex(self): + """Test sanitization of various number formats.""" + query = "SELECT * FROM data WHERE int_val = 42 AND float_val = 3.14159 AND hex_val = 0xABCD" + result = SqlSanitizer.sanitize_for_observability(query) + + assert "42" not in result + assert "3.14159" not in result + assert "0xABCD" not in result + + def test_query_without_literals(self): + """Test that queries without literals are preserved mostly unchanged.""" + query = "SELECT COUNT(*) FROM users WHERE active IS NOT NULL" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should preserve the structure since no literals to sanitize + assert "SELECT COUNT(*)" in result + assert "FROM users" in result + assert "WHERE active IS NOT NULL" in result + + def test_query_length_limit(self): + """Test truncation of very long queries.""" + long_query = "SELECT * FROM table WHERE " + "column = 'value' AND " * 50 + result = SqlSanitizer.sanitize_for_observability(long_query, max_length=100) + + assert len(result) <= 100 + assert result.endswith("...") + + def test_query_size_limit(self): + """Test handling of queries that exceed size limit.""" + # Create a query larger than 10KB + huge_query = "SELECT * FROM table WHERE " + "x" * 15000 + result = SqlSanitizer.sanitize_for_observability(huge_query) + + assert "QUERY TOO LARGE" in result + assert "15000 chars" in result or "15003 chars" in result # Allow for slight variation + + def test_sql_keywords_preserved(self): + """Test that SQL keywords are properly preserved.""" + query = "SELECT DISTINCT name FROM users ORDER BY created_at LIMIT 10" + result = SqlSanitizer.sanitize_for_observability(query) + + # All keywords should be preserved + keywords = ["SELECT", "DISTINCT", "FROM", "ORDER", "BY", "LIMIT"] + for keyword in keywords: + assert keyword in result.upper() + + def test_multiple_statements(self): + """Test handling of multiple SQL statements.""" + query = "SELECT * FROM users; UPDATE users SET active = true" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should handle multiple statements (sqlparse takes the first one) + assert "SELECT" in result + # The UPDATE might not be included if sqlparse only processes first statement + + @patch('src.omnibase_infra.nodes.node_distributed_tracing_compute.v1_0_0.utils.sql_sanitizer.sqlparse.parse') + def test_sqlparse_failure_handling(self, mock_parse): + """Test handling when sqlparse fails to parse query.""" + mock_parse.side_effect = Exception("Parse error") + + query = "SELECT * FROM users WHERE name = 'test'" + + # Should raise OnexError when sqlparse fails (fail-fast principle) + with pytest.raises(OnexError) as exc_info: + SqlSanitizer.sanitize_for_observability(query) + + assert "SQL query sanitization failed" in str(exc_info.value) + + def test_whitespace_normalization(self): + """Test that excessive whitespace is normalized.""" + query = "SELECT * FROM users WHERE name = 'test'" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should not have excessive whitespace + assert " " not in result # No quadruple spaces + assert "SELECT * FROM users WHERE name" in result + + def test_sensitive_literal_detection(self): + """Test the _is_sensitive_literal helper method.""" + # This is a white-box test to ensure the token detection works + # In practice, this would require mocking sqlparse tokens + pass # Implementation would need sqlparse token mocks + + def test_sql_keywords_set_completeness(self): + """Test that the SQL keywords set contains expected keywords.""" + keywords = SqlSanitizer._get_sql_keywords() + + # Test a few critical keywords + expected_keywords = { + 'SELECT', 'FROM', 'WHERE', 'INSERT', 'UPDATE', 'DELETE', + 'JOIN', 'GROUP', 'ORDER', 'HAVING', 'LIMIT' + } + + assert expected_keywords.issubset(keywords) + assert len(keywords) > 20 # Should have a substantial set of keywords + + def test_clean_whitespace_utility(self): + """Test the whitespace cleaning utility method.""" + test_cases = [ + ("SELECT * FROM users", "SELECT * FROM users"), + (" SELECT * ", "SELECT *"), + ("SELECT\n\t*\r\nFROM users", "SELECT * FROM users"), + ] + + for input_str, expected in test_cases: + result = SqlSanitizer._clean_whitespace(input_str) + assert result == expected + + +class TestSqlSanitizerEdgeCases: + """Test edge cases and error conditions for SqlSanitizer.""" + + def test_none_input_handling(self): + """Test that None input is handled gracefully.""" + # Should handle None input without crashing + try: + result = SqlSanitizer.sanitize_for_observability(None) + assert result == "" + except (TypeError, AttributeError): + # Acceptable to raise error for None input + pass + + def test_special_characters_in_strings(self): + """Test handling of special characters and Unicode in string literals.""" + query = "SELECT * FROM users WHERE name = 'José García' AND emoji = '🚀'" + result = SqlSanitizer.sanitize_for_observability(query) + + # Should not contain the Unicode characters + assert "José García" not in result + assert "🚀" not in result + + def test_nested_quotes_and_backslashes(self): + """Test complex quoting scenarios.""" + query = r"SELECT * FROM data WHERE json_field = '{\"key\": \"value with \\\"quotes\\\"\"}'" + result = SqlSanitizer.sanitize_for_observability(query) + + # The complex JSON string should be sanitized + assert '{"key": "value with \\"quotes\\"}' not in result + + def test_sql_injection_patterns(self): + """Test that common SQL injection patterns are properly sanitized.""" + injection_queries = [ + "SELECT * FROM users WHERE id = 1; DROP TABLE users; --", + "SELECT * FROM users WHERE name = '' OR '1'='1'", + "SELECT * FROM users WHERE id = 1 UNION SELECT password FROM admin", + ] + + for query in injection_queries: + result = SqlSanitizer.sanitize_for_observability(query) + # The literal values that could be injection should be sanitized + # Structure should be preserved for observability + assert "users" in result # Table name should be preserved + assert "SELECT" in result # Keywords should be preserved \ No newline at end of file