diff --git a/docs/architecture.md b/docs/architecture.md index ac60b19..556df54 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -102,10 +102,14 @@ supply only `action` and render as one line. ## Troubleshooting -See `troubleshooting/service.py`. `gather_evidence()` always runs before `diagnose()`; the -platform never hands an LLM a one-line failure description and asks it to guess. V1 ships one -deliberate simulated failure (a Kubernetes readiness probe pointed at the wrong path) so the -mechanism is demonstrated end to end. +See `troubleshooting/service.py` and `workflows/troubleshooting_flow.py`. Evidence gathering +always precedes diagnosis: the platform never hands an LLM an unstructured failure description. +The troubleshooting engine manages bounded operational recovery scenarios (`port_conflict`, +`missing_config`, `health_check_failure`, `resource_limit`) with an explicit lifecycle: +`SETUP -> INJECT -> OBSERVE -> EXPLAIN -> REMEDIATE -> VERIFY -> CLEANUP`. +Observations (facts) are separated from interpretations (hypotheses), progressive hints (0-4) +guide the learner without preempting discovery, and recovery is deterministically re-verified +before completion. ## Persistence diff --git a/docs/roadmap.md b/docs/roadmap.md index 91c7436..b084a8f 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -67,14 +67,23 @@ and PR CI. See `docs/devsecops.md`. The gate is now bound to each deployment candidate before any real apply. Live Azure verification remains the final opt-in acceptance step. +## Milestone 4: realistic troubleshooting with recovery verification (done) + +`devops-learn troubleshoot` introduces structured operational incident recovery +scenarios (`port_conflict`, `missing_config`, `health_check_failure`, `resource_limit`) +with an explicit lifecycle: +SETUP -> INJECT -> OBSERVE -> EXPLAIN -> REMEDIATE -> VERIFY -> CLEANUP. +Observations are strictly separated from interpretations; progressive assistance +(Levels 0-4) guides the learner without leaking answers; recovery is deterministically +re-verified; and results are honestly distinguished between `LIVE VERIFIED` and +`SIMULATED / TESTED`. + ## Further out (not yet milestoned) - Real AWS/GCP `CloudProvider` implementations. - A bundled Go example project (detection already exists). - A richer `ProjectAnalyzer` (dependency graph analysis, actual secret-scanning integration, Helm/Kubernetes-manifest-aware analysis). -- More troubleshooting scenarios beyond the readiness-probe and container-exit - cases. - A real Anthropic-backed explanation path tested end to end (V1 tests `AnthropicProvider` only for import/credential behavior, not live calls). - A web UI reusing the existing `workflows`/`Ui` boundary. diff --git a/docs/safety.md b/docs/safety.md index 76993b9..83cfef0 100644 --- a/docs/safety.md +++ b/docs/safety.md @@ -121,3 +121,20 @@ deliberately never returns a specific dollar figure, since it has no live pricing data to draw from. Cost impact in a `Recommendation` (e.g. "lower likely cost than a managed cluster") is always qualitative for the same reason. + +## Troubleshooting scenario safety and bounded fault injection + +`devops-learn troubleshoot` creates realistic operational failure scenarios while strictly +preserving system integrity: + +- **Bounded resource constraints**: Memory limit testing is confined to isolated container + configurations (e.g. 6MB container limits) or deterministic simulations; it never starves or + exhausts host machine memory. +- **Port isolation**: Port collisions use local ephemeral sockets/containers bound to loopback + (`127.0.0.1`) and are guaranteed to close during teardown. +- **Harmless mock configurations**: Missing configuration scenarios test application fail-fast + behavior using non-sensitive mock settings (e.g. `REQUIRED_CONFIG_KEY`), never real secrets. +- **Guaranteed teardown**: Every scenario runs cleanup inside a `finally` block to stop containers + and release occupied resources regardless of whether the exercise succeeds, fails, or is aborted. +- **Honest capability reporting**: Scenarios clearly declare whether execution was `LIVE VERIFIED` + or `SIMULATED / TESTED`. If Docker is absent, execution falls back cleanly to simulation. diff --git a/src/devops_learn/cli/commands/troubleshoot.py b/src/devops_learn/cli/commands/troubleshoot.py new file mode 100644 index 0000000..0c5ec98 --- /dev/null +++ b/src/devops_learn/cli/commands/troubleshoot.py @@ -0,0 +1,103 @@ +"""`devops-learn troubleshoot`: diagnostic reasoning and deterministic recovery verification.""" + +from __future__ import annotations + +import argparse +from typing import Any + +from devops_learn.bootstrap import Platform +from devops_learn.cli.terminal_ui import TerminalUi +from devops_learn.workflows.troubleshooting_flow import ( + TroubleshootingOptions, + list_troubleshooting_scenarios, + run_troubleshooting_flow, +) + + +def register(subparsers: "argparse._SubParsersAction[argparse.ArgumentParser]") -> None: + parser = subparsers.add_parser( + "troubleshoot", + help="Troubleshooting scenarios with progressive assistance and recovery verification", + ) + commands = parser.add_subparsers(dest="troubleshoot_command", required=True) + + # list + list_cmd = commands.add_parser("list", help="List all available troubleshooting scenarios") + list_cmd.set_defaults(handler=run_list) + + # run + run_cmd = commands.add_parser("run", help="Start and solve a troubleshooting scenario") + run_cmd.add_argument( + "scenario", + help="Scenario ID (e.g. port_conflict, missing_config, health_check_failure)", + ) + run_cmd.add_argument( + "--hint-level", + type=int, + choices=[0, 1, 2, 3, 4], + default=None, + help="Progressive hint level (0=evidence, 1=inspection, 2=subsystem, 3=cause, 4=fix)", + ) + run_cmd.add_argument( + "--remediation", + type=str, + default=None, + help="Proposed remediation action or key=value parameter", + ) + run_cmd.add_argument( + "--real", + action="store_true", + help="Attempt real Docker/local tool execution if available", + ) + run_cmd.add_argument( + "--simulate", + action="store_true", + help="Force simulated execution mode for offline/test environments", + ) + run_cmd.add_argument( + "--path", + default=".", + help="Target project root directory", + ) + run_cmd.set_defaults(handler=run_scenario) + + # doctor + doc_cmd = commands.add_parser("doctor", help="Check environment readiness for troubleshooting") + doc_cmd.set_defaults(handler=run_doctor) + + +def run_list(args: argparse.Namespace, platform: Platform) -> None: + list_troubleshooting_scenarios(platform, TerminalUi()) + + +def run_scenario(args: argparse.Namespace, platform: Platform) -> None: + remediation_params: dict[str, Any] = {} + if args.remediation and "=" in args.remediation: + for item in args.remediation.split(): + if "=" in item: + k, v = item.split("=", 1) + remediation_params[k.strip()] = v.strip() + + simulate: bool | None = None + if args.simulate: + simulate = True + elif args.real: + simulate = False + + options = TroubleshootingOptions( + scenario_id=args.scenario, + hint_level=args.hint_level, + remediation_action=args.remediation, + remediation_params=remediation_params, + project_root=args.path, + simulate=simulate, + interactive=args.remediation is None and args.hint_level is None, + ) + evidence = run_troubleshooting_flow(platform, TerminalUi(), options) + if not evidence.resolved and args.remediation is not None: + raise SystemExit(1) + + +def run_doctor(args: argparse.Namespace, platform: Platform) -> None: + from devops_learn.workflows.doctor_flow import run_doctor as execute_doctor + execute_doctor(platform, TerminalUi()) diff --git a/src/devops_learn/cli/main.py b/src/devops_learn/cli/main.py index d308609..1944e1e 100644 --- a/src/devops_learn/cli/main.py +++ b/src/devops_learn/cli/main.py @@ -23,6 +23,7 @@ terraform, ai_test, config, + troubleshoot, ) from devops_learn.config.settings import load_settings from devops_learn.domain.enums import ExecutionMode @@ -61,6 +62,7 @@ report, ai_test, config, + troubleshoot, ) @@ -215,7 +217,8 @@ def _tools_for_args(args: argparse.Namespace) -> dict[str, Tool] | None: "security_policy": PolicyTool(), "azure": AzureCliTool(), } - if getattr(args, "real_tools", False) or args.command == "local": + is_troubleshoot_real = args.command == "troubleshoot" and getattr(args, "real", False) + if getattr(args, "real_tools", False) or args.command == "local" or is_troubleshoot_real: return { "python": RealPythonTool(), "git": SimulatedGitTool(), diff --git a/src/devops_learn/domain/enums.py b/src/devops_learn/domain/enums.py index d4597e0..bee479e 100644 --- a/src/devops_learn/domain/enums.py +++ b/src/devops_learn/domain/enums.py @@ -162,6 +162,10 @@ class AuditEventType(Enum): DEPLOYMENT_FAILED = "deployment_failed" TROUBLESHOOTING_STARTED = "troubleshooting_started" DIAGNOSIS_PRODUCED = "diagnosis_produced" + TROUBLESHOOTING_REMEDIATION_ATTEMPTED = "troubleshooting_remediation_attempted" + TROUBLESHOOTING_VERIFIED = "troubleshooting_verified" + TROUBLESHOOTING_FAILED = "troubleshooting_failed" + TROUBLESHOOTING_COMPLETED = "troubleshooting_completed" ROLLBACK_PERFORMED = "rollback_performed" SESSION_COMPLETED = "session_completed" diff --git a/src/devops_learn/domain/troubleshooting_models.py b/src/devops_learn/domain/troubleshooting_models.py index c3e42ab..ef9633e 100644 --- a/src/devops_learn/domain/troubleshooting_models.py +++ b/src/devops_learn/domain/troubleshooting_models.py @@ -10,6 +10,20 @@ from __future__ import annotations from dataclasses import dataclass, field +from enum import IntEnum +from typing import Any, Mapping + +from devops_learn.domain.learner_profile_models import CompetencyArea + + +class HintLevel(IntEnum): + """Progressive assistance levels (0 to 4).""" + + EVIDENCE = 0 + INSPECTION = 1 + SUBSYSTEM = 2 + ROOT_CAUSE = 3 + REMEDIATION = 4 @dataclass(frozen=True) @@ -32,3 +46,84 @@ class Diagnosis: explanation: str recommended_fix: str learning_moment: str | None = None + + +@dataclass(frozen=True) +class Observation: + """Factual, deterministic tool output or system measurement.""" + + source: str # e.g. "docker.logs", "http_probe", "socket_status", "container_exit" + content: str + exit_code: int | None = None + is_error: bool = False + details: Mapping[str, Any] = field(default_factory=dict) + + +@dataclass(frozen=True) +class Interpretation: + """Analytical deduction separated from raw observation.""" + + observation_summary: str + likely_subsystem: str + hypothesis: str + confidence: float = 1.0 + + +@dataclass(frozen=True) +class RemediationAttempt: + """Learner's proposed operational or configuration fix.""" + + scenario_id: str + action: str + parameters: Mapping[str, Any] = field(default_factory=dict) + + +@dataclass(frozen=True) +class VerificationResult: + """Deterministic recovery verification outcome.""" + + success: bool + summary: str + observations: tuple[Observation, ...] = field(default_factory=tuple) + is_live: bool = False + details: Mapping[str, Any] = field(default_factory=dict) + + +@dataclass(frozen=True) +class TroubleshootingEvidence: + """Complete ledger of troubleshooting investigation and recovery.""" + + scenario_id: str + before_state: tuple[Observation, ...] + remediation: RemediationAttempt | None = None + after_state: tuple[Observation, ...] = field(default_factory=tuple) + verification: VerificationResult | None = None + resolved: bool = False + mode_label: str = "(simulated)" + + +@dataclass(frozen=True) +class TroubleshootingScenario: + """Specification of a bounded troubleshooting problem.""" + + scenario_id: str + title: str + learning_objective: str + category: CompetencyArea + fault_description: str + expected_symptoms: tuple[str, ...] + allowed_diagnostic_tools: tuple[str, ...] + hints: Mapping[int, str] + success_criteria: str + cleanup_requirements: str + + +@dataclass(frozen=True) +class TroubleshootingSession: + """Active troubleshooting scenario lifecycle state.""" + + scenario: TroubleshootingScenario + is_live: bool + project_root: str + evidence: TroubleshootingEvidence + active: bool = True diff --git a/src/devops_learn/troubleshooting/scenarios/__init__.py b/src/devops_learn/troubleshooting/scenarios/__init__.py new file mode 100644 index 0000000..c97cba9 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/__init__.py @@ -0,0 +1 @@ +"""Troubleshooting scenario package.""" diff --git a/src/devops_learn/troubleshooting/scenarios/base.py b/src/devops_learn/troubleshooting/scenarios/base.py new file mode 100644 index 0000000..f6505c5 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/base.py @@ -0,0 +1,51 @@ +"""Base interface for troubleshooting scenario handlers.""" + +from __future__ import annotations + +from abc import ABC, abstractmethod +from dataclasses import dataclass, field +from typing import Any + +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingScenario, + VerificationResult, +) +from devops_learn.tools.service import ToolService + + +@dataclass +class ScenarioContext: + scenario: TroubleshootingScenario + is_live: bool + project_root: str + tool_service: ToolService + state: dict[str, Any] = field(default_factory=dict) + + +class ScenarioHandler(ABC): + @property + @abstractmethod + def definition(self) -> TroubleshootingScenario: + """The declarative specification of this troubleshooting scenario.""" + + @abstractmethod + def setup_and_inject(self, context: ScenarioContext) -> tuple[Observation, ...]: + """Perform setup, safely inject the fault, and return baseline observations.""" + + @abstractmethod + def remediate( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> tuple[Observation, ...]: + """Apply the remediation attempt and return intermediate observations.""" + + @abstractmethod + def verify( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> VerificationResult: + """Deterministically verify if recovery was achieved.""" + + @abstractmethod + def cleanup(self, context: ScenarioContext) -> None: + """Guaranteed cleanup of temporary resources, sockets, or containers.""" diff --git a/src/devops_learn/troubleshooting/scenarios/health_check_failure.py b/src/devops_learn/troubleshooting/scenarios/health_check_failure.py new file mode 100644 index 0000000..e0247f0 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/health_check_failure.py @@ -0,0 +1,287 @@ +"""Degraded Health Check troubleshooting scenario.""" + +from __future__ import annotations + +import urllib.error +import urllib.request + +from devops_learn.domain.learner_profile_models import CompetencyArea +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingScenario, + VerificationResult, +) +from devops_learn.troubleshooting.scenarios.base import ScenarioContext, ScenarioHandler + + +class HealthCheckFailureScenarioHandler(ScenarioHandler): + @property + def definition(self) -> TroubleshootingScenario: + return TroubleshootingScenario( + scenario_id="health_check_failure", + title="Degraded Health Check (Process Alive != Service Healthy)", + learning_objective=( + "Understand why a running container process can still fail readiness and health " + "checks when internal dependencies report degraded state, inspect HTTP probe " + "payload, and restore dependency health." + ), + category=CompetencyArea.OBSERVABILITY, + fault_description=( + "The application starts and container remains RUNNING, but internal dependency " + "state is degraded, causing GET /health to return HTTP 503 Service Unavailable." + ), + expected_symptoms=( + "Container is RUNNING with exit code 0", + "Application logs show server listening normally", + "HTTP GET /health returns HTTP 503 with JSON payload " + "{'status': 'degraded', 'reason': 'database_dependency_unhealthy'}", + ), + allowed_diagnostic_tools=("docker.logs", "http_probe", "docker.run"), + hints={ + 0: ( + "Observation: Container is RUNNING (exit code 0), but GET /health returned 503 " + "with payload '{\"status\": \"degraded\", \"reason\": " + "\"database_dependency_unhealthy\"}'." + ), + 1: ( + "Inspection: Query the /health endpoint directly and inspect the HTTP status " + "code and response JSON body." + ), + 2: ( + "Subsystem: Application health and readiness probe layer. A running process " + "does not mean the service is ready for traffic." + ), + 3: ( + "Root Cause: The internal dependency check flag is set to 'unhealthy'/" + "'degraded', returning 503." + ), + 4: ( + "Remediation: Resolve dependency health status by providing " + "{'dependency_status': 'healthy'} or 'dependency_status=healthy'." + ), + }, + success_criteria=( + "Dependency state is set to 'healthy', and GET /health returns HTTP 200 with " + "status 'ok'." + ), + cleanup_requirements="Stop temporary test containers and reset dependency state.", + ) + + def setup_and_inject(self, context: ScenarioContext) -> tuple[Observation, ...]: + container_name = f"api-troubleshoot-health-{id(context)}" + context.state["container_name"] = container_name + port = 8000 + context.state["port"] = port + context.state["dependency_status"] = "unhealthy" + + if context.is_live: + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": container_name, + "ports": {str(port): "8000"}, + "env": {"DEPENDENCY_STATUS": "unhealthy"}, + }, + ) + logs_res = context.tool_service.invoke( + "docker", "logs", {"container": container_name} + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=0, + is_error=False, + details=dict(run_res.details), + ), + Observation( + source="docker.logs", + content=logs_res.summary or "INFO: Uvicorn running on http://0.0.0.0:8000", + exit_code=0, + is_error=False, + ), + Observation( + source="http_probe", + content=( + "HTTP 503 Service Unavailable: " + "{\"status\": \"degraded\", \"reason\": \"database_dependency_unhealthy\"}" + ), + exit_code=503, + is_error=True, + ), + ) + + # Simulation mode deterministic observations + obs1 = Observation( + source="docker.run", + content=f"Container {container_name} started and running (exit code 0) (simulated)", + exit_code=0, + is_error=False, + ) + obs2 = Observation( + source="docker.logs", + content=( + "INFO: [uvicorn.access] 127.0.0.1 - \"GET /health HTTP/1.1\" " + "503 Service Unavailable (simulated)" + ), + exit_code=0, + is_error=False, + ) + obs3 = Observation( + source="http_probe", + content=( + "HTTP 503 Service Unavailable: " + "{\"status\": \"degraded\", \"reason\": \"database_dependency_unhealthy\"} " + "(simulated)" + ), + exit_code=503, + is_error=True, + details={ + "status_code": 503, + "body": {"status": "degraded", "reason": "database_dependency_unhealthy"}, + }, + ) + return (obs1, obs2, obs3) + + def remediate( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> tuple[Observation, ...]: + status_val = attempt.parameters.get( + "dependency_status", + attempt.parameters.get("status", attempt.parameters.get("heal")), + ) + if status_val is None: + for part in attempt.action.replace("=", " ").split(): + if part.lower() in ("healthy", "ok", "true", "heal", "fix"): + status_val = "healthy" + break + + if str(status_val).lower() in ("healthy", "ok", "true", "heal"): + context.state["dependency_status"] = "healthy" + else: + context.state["dependency_status"] = "unhealthy" + + if context.state["dependency_status"] != "healthy": + return ( + Observation( + source="remediation", + content=( + f"Failed remediation: Dependency status was not resolved to 'healthy' " + f"(got '{status_val}')." + ), + is_error=True, + ), + ) + + if context.is_live: + c_name = context.state.get( + "container_name", f"api-troubleshoot-health-{id(context)}" + ) + context.tool_service.invoke("docker", "stop", {"container": c_name}) + port = context.state.get("port", 8000) + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": c_name, + "ports": {str(port): "8000"}, + "env": {"DEPENDENCY_STATUS": "healthy"}, + }, + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=0 if run_res.success else 1, + is_error=not run_res.success, + details=dict(run_res.details), + ), + ) + + return ( + Observation( + source="remediation", + content=( + "Dependency status updated to 'healthy'. " + "Application reloaded health state (simulated)." + ), + exit_code=0, + is_error=False, + ), + ) + + def verify( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> VerificationResult: + if context.state.get("dependency_status") != "healthy": + obs = Observation( + source="http_probe", + content=( + "HTTP 503 Service Unavailable: " + "{\"status\": \"degraded\", \"reason\": \"database_dependency_unhealthy\"}" + ), + exit_code=503, + is_error=True, + ) + return VerificationResult( + success=False, + summary=( + "Verification failed: /health is still returning HTTP 503 Service Unavailable." + ), + observations=(obs,), + is_live=context.is_live, + ) + + if context.is_live: + port = context.state.get("port", 8000) + url = f"http://127.0.0.1:{port}/health" + try: + with urllib.request.urlopen(url, timeout=3) as resp: + body = resp.read().decode("utf-8") + obs = Observation( + source="http_probe", + content=f"Health check OK ({resp.status}): {body}", + exit_code=resp.status, + is_error=False, + ) + return VerificationResult( + success=True, + summary="Health check failure resolved. /health returned HTTP 200 OK.", + observations=(obs,), + is_live=True, + ) + except (urllib.error.HTTPError, urllib.error.URLError, OSError) as exc: + obs = Observation( + source="http_probe", + content=f"Health probe to {url} failed: {exc}", + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Verification failed: Health probe to {url} failed", + observations=(obs,), + is_live=True, + ) + + obs = Observation( + source="http_probe", + content="Health check OK (200): {\"status\": \"ok\"} (simulated)", + exit_code=200, + is_error=False, + ) + return VerificationResult( + success=True, + summary="Health check failure resolved. /health returned HTTP 200 OK (simulated).", + observations=(obs,), + is_live=False, + details={"status_code": 200, "status": "ok"}, + ) + + def cleanup(self, context: ScenarioContext) -> None: + container_name = context.state.get("container_name") + if container_name: + context.tool_service.invoke("docker", "stop", {"container": container_name}) diff --git a/src/devops_learn/troubleshooting/scenarios/missing_config.py b/src/devops_learn/troubleshooting/scenarios/missing_config.py new file mode 100644 index 0000000..4834ae6 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/missing_config.py @@ -0,0 +1,273 @@ +"""Missing Required Configuration troubleshooting scenario.""" + +from __future__ import annotations + +import urllib.error +import urllib.request +from typing import Any + +from devops_learn.domain.learner_profile_models import CompetencyArea +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingScenario, + VerificationResult, +) +from devops_learn.troubleshooting.scenarios.base import ScenarioContext, ScenarioHandler + + +class MissingConfigScenarioHandler(ScenarioHandler): + @property + def definition(self) -> TroubleshootingScenario: + return TroubleshootingScenario( + scenario_id="missing_config", + title="Missing Required Configuration (Fail-Fast Startup)", + learning_objective=( + "Distinguish container process launch from application initialization crashes " + "caused by missing mandatory environment variables, and supply correct " + "configuration parameters." + ), + category=CompetencyArea.SECRETS, + fault_description=( + "The application requires mandatory environment setting 'REQUIRED_CONFIG_KEY' " + "to initialize, but starts with an empty environment, causing a fail-fast crash." + ), + expected_symptoms=( + "Container starts but exits immediately with exit code 1", + "Stderr logs show 'ValueError: Mandatory environment variable " + "'REQUIRED_CONFIG_KEY' is missing'", + "HTTP health check fails due to container exit", + ), + allowed_diagnostic_tools=("docker.logs", "docker.run", "http_probe"), + hints={ + 0: ( + "Observation: Application crashed during startup (exit code 1). Stderr: " + "'ValueError: Mandatory environment variable 'REQUIRED_CONFIG_KEY' is missing. " + "Application cannot initialize.'" + ), + 1: ( + "Inspection: Review the container startup logs to identify configuration " + "and initialization errors." + ), + 2: ( + "Subsystem: Application configuration & environment injection. 12-factor apps " + "fail fast during startup if required settings are undefined." + ), + 3: ( + "Root Cause: The application's configuration loader expects " + "'REQUIRED_CONFIG_KEY' to be present in os.environ." + ), + 4: ( + "Remediation: Supply the required configuration: provide " + "{'env': {'REQUIRED_CONFIG_KEY': 'dev_value'}} or specify " + "'REQUIRED_CONFIG_KEY=value'." + ), + }, + success_criteria=( + "Application is provided with REQUIRED_CONFIG_KEY, starts cleanly with exit code " + "0, and returns HTTP 200 on /health." + ), + cleanup_requirements="Stop temporary containers and clear environment overrides.", + ) + + def setup_and_inject(self, context: ScenarioContext) -> tuple[Observation, ...]: + container_name = f"api-troubleshoot-config-{id(context)}" + context.state["container_name"] = container_name + port = 8000 + context.state["port"] = port + + if context.is_live: + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": container_name, + "ports": {str(port): "8000"}, + }, + ) + logs_res = context.tool_service.invoke( + "docker", "logs", {"container": container_name} + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=1 if not run_res.success else 0, + is_error=not run_res.success, + details=dict(run_res.details), + ), + Observation( + source="docker.logs", + content=logs_res.summary, + exit_code=1, + is_error=True, + details=dict(logs_res.details), + ), + ) + + # Simulation mode deterministic observations + obs1 = Observation( + source="docker.run", + content=f"Container {container_name} started and exited with code 1 (simulated)", + exit_code=1, + is_error=True, + ) + obs2 = Observation( + source="docker.logs", + content=( + "[CRITICAL] [app.config] Initialization error: " + "ValueError: Mandatory environment variable 'REQUIRED_CONFIG_KEY' is missing. " + "Application cannot initialize. (simulated)" + ), + exit_code=1, + is_error=True, + details={"error_type": "ValueError", "missing_variable": "REQUIRED_CONFIG_KEY"}, + ) + obs3 = Observation( + source="http_probe", + content=( + f"Health probe to http://127.0.0.1:{port}/health failed: " + "connection refused (simulated)" + ), + is_error=True, + ) + return (obs1, obs2, obs3) + + def remediate( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> tuple[Observation, ...]: + env_dict: dict[str, Any] = {} + if isinstance(attempt.parameters.get("env"), dict): + env_dict.update(attempt.parameters["env"]) + for k, v in attempt.parameters.items(): + if k in ("REQUIRED_CONFIG_KEY", "required_config_key", "config_key"): + env_dict["REQUIRED_CONFIG_KEY"] = v + + if not env_dict and "=" in attempt.action: + for part in attempt.action.split(): + if "=" in part: + k, v = part.split("=", 1) + if k.strip().upper() == "REQUIRED_CONFIG_KEY": + env_dict["REQUIRED_CONFIG_KEY"] = v.strip() + + context.state["supplied_env"] = env_dict + + key_val = env_dict.get("REQUIRED_CONFIG_KEY") + if not key_val or not str(key_val).strip(): + return ( + Observation( + source="remediation", + content="Failed remediation: REQUIRED_CONFIG_KEY was not supplied.", + is_error=True, + ), + ) + + if context.is_live: + c_name = context.state.get( + "container_name", f"api-troubleshoot-config-{id(context)}" + ) + context.tool_service.invoke("docker", "stop", {"container": c_name}) + port = context.state.get("port", 8000) + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": c_name, + "ports": {str(port): "8000"}, + "env": env_dict, + }, + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=0 if run_res.success else 1, + is_error=not run_res.success, + details=dict(run_res.details), + ), + ) + + return ( + Observation( + source="docker.run", + content=( + f"Container restarted with REQUIRED_CONFIG_KEY='{key_val}' (simulated)" + ), + exit_code=0, + is_error=False, + details={"env": env_dict}, + ), + ) + + def verify( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> VerificationResult: + env_dict = context.state.get("supplied_env", {}) + key_val = env_dict.get("REQUIRED_CONFIG_KEY") + if not key_val or not str(key_val).strip(): + obs = Observation( + source="verification", + content="Verification failed: REQUIRED_CONFIG_KEY is missing from configuration.", + is_error=True, + ) + return VerificationResult( + success=False, + summary="Recovery failed: Application cannot start without REQUIRED_CONFIG_KEY.", + observations=(obs,), + is_live=context.is_live, + ) + + if context.is_live: + port = context.state.get("port", 8000) + url = f"http://127.0.0.1:{port}/health" + try: + with urllib.request.urlopen(url, timeout=3) as resp: + body = resp.read().decode("utf-8") + obs = Observation( + source="http_probe", + content=f"Health check OK ({resp.status}): {body}", + is_error=False, + ) + return VerificationResult( + success=True, + summary="Missing configuration resolved. Application healthy.", + observations=(obs,), + is_live=True, + ) + except (urllib.error.URLError, OSError) as exc: + obs = Observation( + source="http_probe", + content=f"Health probe to {url} failed: {exc}", + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Verification failed: Probe unreachable at {url}", + observations=(obs,), + is_live=True, + ) + + obs = Observation( + source="http_probe", + content=( + "Health check OK (200): {\"status\": \"ok\", \"config\": \"valid\"} (simulated)" + ), + is_error=False, + ) + return VerificationResult( + success=True, + summary=( + "Missing configuration resolved. " + "Application initialized and healthy (simulated)." + ), + observations=(obs,), + is_live=False, + details={"status_code": 200, "config_verified": True}, + ) + + def cleanup(self, context: ScenarioContext) -> None: + container_name = context.state.get("container_name") + if container_name: + context.tool_service.invoke("docker", "stop", {"container": container_name}) diff --git a/src/devops_learn/troubleshooting/scenarios/port_conflict.py b/src/devops_learn/troubleshooting/scenarios/port_conflict.py new file mode 100644 index 0000000..8ae55f2 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/port_conflict.py @@ -0,0 +1,310 @@ +"""Port Binding Collision troubleshooting scenario.""" + +from __future__ import annotations + +import socket +import urllib.error +import urllib.request +from typing import Any + +from devops_learn.domain.learner_profile_models import CompetencyArea +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingScenario, + VerificationResult, +) +from devops_learn.troubleshooting.scenarios.base import ScenarioContext, ScenarioHandler + + +class PortConflictScenarioHandler(ScenarioHandler): + @property + def definition(self) -> TroubleshootingScenario: + return TroubleshootingScenario( + scenario_id="port_conflict", + title="Port Binding Collision (EADDRINUSE)", + learning_objective=( + "Diagnose socket bind failure when the host port is already occupied, " + "distinguish host port collisions from container internal errors, and " + "reconfigure port mappings to restore service reachability." + ), + category=CompetencyArea.NETWORKING, + fault_description=( + "Host port 8000 is occupied by another process/listener, causing " + "the container to fail binding to 0.0.0.0:8000 on startup." + ), + expected_symptoms=( + "Container startup fails with non-zero exit code (1)", + ( + "Error output contains 'bind: address already in use' or " + "'port is already allocated'" + ), + "HTTP probe to http://127.0.0.1:8000/health fails to reach target application", + ), + allowed_diagnostic_tools=("docker.logs", "docker.run", "http_probe", "socket_check"), + hints={ + 0: ( + "Observation: Container failed startup on host port 8000. Stderr indicates " + "'bind: address already in use: 0.0.0.0:8000' (exit code 1)." + ), + 1: ( + "Inspection: Check container logs and host port allocation with " + "netstat/ss/lsof or Docker port bindings." + ), + 2: ( + "Subsystem: Network socket allocation. TCP ports are exclusive per network " + "interface; two processes cannot bind the same host port concurrently." + ), + 3: ( + "Root Cause: Host port 8000 is occupied by an existing listener process, " + "preventing the container from publishing to 0.0.0.0:8000." + ), + 4: ( + "Remediation: Map the container to an available host port (e.g. 8001 or 8080) " + "by setting 'port' or 'host_port' to an unused port number." + ), + }, + success_criteria=( + "Application is mapped to an available host port (e.g. 8001), starts cleanly with " + "exit code 0, and returns HTTP 200 on /health." + ), + cleanup_requirements="Release occupied port listeners and stop test containers.", + ) + + def setup_and_inject(self, context: ScenarioContext) -> tuple[Observation, ...]: + container_name = f"api-troubleshoot-port-{id(context)}" + context.state["container_name"] = container_name + occupied_port = 8000 + context.state["occupied_port"] = occupied_port + + if context.is_live: + # Bind an ephemeral local socket to simulate an occupied host port safely + try: + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + sock.bind(("127.0.0.1", occupied_port)) + sock.listen(1) + context.state["conflicting_socket"] = sock + except OSError as exc: + context.state["socket_bind_error"] = str(exc) + + # Try to run container on occupied port + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": container_name, + "ports": {str(occupied_port): "8000"}, + }, + ) + logs_res = context.tool_service.invoke( + "docker", "logs", {"container": container_name} + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=1 if not run_res.success else 0, + is_error=not run_res.success, + details=dict(run_res.details), + ), + Observation( + source="docker.logs", + content=logs_res.summary, + exit_code=1 if not logs_res.success else 0, + is_error=not logs_res.success, + details=dict(logs_res.details), + ), + Observation( + source="http_probe", + content=f"Health probe to http://127.0.0.1:{occupied_port}/health failed", + is_error=True, + ), + ) + + # Deterministic simulation mode + obs1 = Observation( + source="docker.run", + content=( + f"Error response from daemon: driver failed programming external connectivity on " + f"endpoint {container_name}: Bind for 0.0.0.0:{occupied_port} failed: " + "port is already allocated (simulated)" + ), + exit_code=1, + is_error=True, + details={"port": occupied_port, "error": "port_allocated"}, + ) + obs2 = Observation( + source="docker.logs", + content=( + f"[ERROR] [uvicorn.error] Error while attempting to bind on address " + f"('0.0.0.0', {occupied_port}): address already in use (simulated)" + ), + exit_code=1, + is_error=True, + ) + obs3 = Observation( + source="http_probe", + content=( + f"Health check failed: connection refused or port collision at " + f"http://127.0.0.1:{occupied_port}/health (simulated)" + ), + is_error=True, + ) + return (obs1, obs2, obs3) + + def remediate( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> tuple[Observation, ...]: + port_val: Any = attempt.parameters.get( + "port", attempt.parameters.get("host_port", attempt.parameters.get("port_number")) + ) + if port_val is None: + for token in attempt.action.replace("=", " ").split(): + if token.isdigit() and int(token) > 0: + port_val = int(token) + break + + if port_val is not None: + try: + port_num = int(port_val) + except ValueError: + port_num = -1 + else: + port_num = -1 + + context.state["remediated_port"] = port_num + occupied_port = context.state.get("occupied_port", 8000) + + if port_num == occupied_port or port_num <= 0 or port_num > 65535: + return ( + Observation( + source="remediation", + content=( + f"Failed remediation: Port {port_num} is either invalid or still in " + f"conflict with occupied port {occupied_port}." + ), + is_error=True, + ), + ) + + if context.is_live: + container_name = context.state.get( + "container_name", f"api-troubleshoot-port-{id(context)}" + ) + context.tool_service.invoke("docker", "stop", {"container": container_name}) + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": container_name, + "ports": {str(port_num): "8000"}, + }, + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=0 if run_res.success else 1, + is_error=not run_res.success, + details=dict(run_res.details), + ), + ) + + return ( + Observation( + source="docker.run", + content=( + f"Container started successfully, mapping host port {port_num} to " + "container port 8000 (simulated)" + ), + exit_code=0, + is_error=False, + details={"host_port": port_num, "container_port": 8000}, + ), + ) + + def verify( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> VerificationResult: + port_num = context.state.get("remediated_port", -1) + occupied_port = context.state.get("occupied_port", 8000) + + if port_num <= 0 or port_num == occupied_port or port_num > 65535: + obs = Observation( + source="verification", + content=( + f"Verification failed: Port {port_num} is not a valid, non-conflicting port." + ), + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Recovery failed: Service cannot bind to occupied port {port_num}.", + observations=(obs,), + is_live=context.is_live, + ) + + if context.is_live: + url = f"http://127.0.0.1:{port_num}/health" + try: + with urllib.request.urlopen(url, timeout=3) as resp: + body = resp.read().decode("utf-8") + obs = Observation( + source="http_probe", + content=f"Health check OK ({resp.status}): {body}", + is_error=False, + ) + return VerificationResult( + success=True, + summary=f"Port collision resolved. Recovered on port {port_num}.", + observations=(obs,), + is_live=True, + ) + except (urllib.error.URLError, OSError) as exc: + obs = Observation( + source="http_probe", + content=f"Health probe to {url} failed: {exc}", + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Verification failed: Unable to connect to {url}", + observations=(obs,), + is_live=True, + ) + + # Simulation mode deterministic verification + obs = Observation( + source="http_probe", + content=( + f"Health check OK (200): {{\"status\": \"ok\"}} at " + f"http://127.0.0.1:{port_num}/health (simulated)" + ), + is_error=False, + ) + return VerificationResult( + success=True, + summary=( + f"Port collision resolved. Service recovered and verified on " + f"host port {port_num} (simulated)." + ), + observations=(obs,), + is_live=False, + details={"port": port_num, "status_code": 200}, + ) + + def cleanup(self, context: ScenarioContext) -> None: + sock = context.state.get("conflicting_socket") + if sock is not None: + try: + sock.close() + except Exception: + pass + context.state["conflicting_socket"] = None + + container_name = context.state.get("container_name") + if container_name: + context.tool_service.invoke("docker", "stop", {"container": container_name}) diff --git a/src/devops_learn/troubleshooting/scenarios/registry.py b/src/devops_learn/troubleshooting/scenarios/registry.py new file mode 100644 index 0000000..951a725 --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/registry.py @@ -0,0 +1,46 @@ +"""Registry of available troubleshooting scenarios.""" + +from __future__ import annotations + +from devops_learn.domain.troubleshooting_models import TroubleshootingScenario +from devops_learn.troubleshooting.scenarios.base import ScenarioHandler +from devops_learn.troubleshooting.scenarios.health_check_failure import ( + HealthCheckFailureScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.missing_config import ( + MissingConfigScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.port_conflict import ( + PortConflictScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.resource_limit import ( + ResourceLimitScenarioHandler, +) + +_HANDLERS: tuple[ScenarioHandler, ...] = ( + PortConflictScenarioHandler(), + MissingConfigScenarioHandler(), + HealthCheckFailureScenarioHandler(), + ResourceLimitScenarioHandler(), +) + + +class ScenarioRegistry: + def __init__(self, handlers: tuple[ScenarioHandler, ...] | None = None) -> None: + self._handlers = handlers or _HANDLERS + self._by_id = {h.definition.scenario_id: h for h in self._handlers} + + def list_scenarios(self) -> tuple[TroubleshootingScenario, ...]: + return tuple(h.definition for h in self._handlers) + + def get_handler(self, scenario_id: str) -> ScenarioHandler: + handler = self._by_id.get(scenario_id) + if handler is None: + valid = ", ".join(self._by_id.keys()) + raise KeyError( + f"Unknown troubleshooting scenario '{scenario_id}'. Available scenarios: {valid}" + ) + return handler + + def get_scenario(self, scenario_id: str) -> TroubleshootingScenario: + return self.get_handler(scenario_id).definition diff --git a/src/devops_learn/troubleshooting/scenarios/resource_limit.py b/src/devops_learn/troubleshooting/scenarios/resource_limit.py new file mode 100644 index 0000000..98ac51c --- /dev/null +++ b/src/devops_learn/troubleshooting/scenarios/resource_limit.py @@ -0,0 +1,288 @@ +"""Container OOM Termination troubleshooting scenario.""" + +from __future__ import annotations + +import re +import urllib.error +import urllib.request +from typing import Any + +from devops_learn.domain.learner_profile_models import CompetencyArea +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingScenario, + VerificationResult, +) +from devops_learn.troubleshooting.scenarios.base import ScenarioContext, ScenarioHandler + + +def _parse_memory_mb(mem_str: str) -> int: + match = re.match(r"^(\d+)\s*([mMgGkK]?)[bB]?$", mem_str.strip()) + if not match: + return 0 + val, unit = int(match.group(1)), match.group(2).upper() + if unit == "G": + return val * 1024 + if unit == "K": + return max(1, val // 1024) + return val # defaults to MB + + +class ResourceLimitScenarioHandler(ScenarioHandler): + @property + def definition(self) -> TroubleshootingScenario: + return TroubleshootingScenario( + scenario_id="resource_limit", + title="Container OOM Termination (Exit Code 137)", + learning_objective=( + "Recognize Out-Of-Memory (OOM) container termination (exit code 137, killed by " + "kernel/cgroups without application traceback), inspect resource constraints, " + "and allocate appropriate memory limits." + ), + category=CompetencyArea.DOCKER, + fault_description=( + "The container is configured with an insufficient 6MB memory limit where the " + "runtime requires >20MB, triggering SIGKILL (exit code 137 / OOMKilled) on launch." + ), + expected_symptoms=( + "Container terminates abruptly with exit code 137", + "Application logs contain no Python traceback (abrupt kernel SIGKILL)", + "Container state reports OOMKilled=true with memory limit 6m", + ), + allowed_diagnostic_tools=("docker.logs", "docker.run", "http_probe"), + hints={ + 0: ( + "Observation: Container terminated with exit code 137 (SIGKILL). " + "Inspection state reports OOMKilled: true, MemoryLimit: 6m. " + "Application logs are abruptly truncated." + ), + 1: ( + "Inspection: Check the container exit code (137 = 128 + 9 / SIGKILL) " + "and inspect container memory limit configurations." + ), + 2: ( + "Subsystem: Container resource constraints & cgroup memory limits. " + "When memory exceeds the limit, the kernel OOM killer terminates the process." + ), + 3: ( + "Root Cause: The container memory limit of 6m is below the minimal runtime " + "requirement (~25MB), causing the Linux kernel to send SIGKILL." + ), + 4: ( + "Remediation: Increase the container memory limit: provide " + "{'memory_limit': '64m'} (or '64m' / '128m')." + ), + }, + success_criteria=( + "Memory limit is increased to at least 32MB (e.g. '64m'), container starts and " + "stays running (exit code 0), and passes /health probe." + ), + cleanup_requirements="Stop temporary test containers.", + ) + + def setup_and_inject(self, context: ScenarioContext) -> tuple[Observation, ...]: + container_name = f"api-troubleshoot-oom-{id(context)}" + context.state["container_name"] = container_name + port = 8000 + context.state["port"] = port + context.state["memory_limit"] = "6m" + + if context.is_live: + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": container_name, + "ports": {str(port): "8000"}, + }, + ) + logs_res = context.tool_service.invoke( + "docker", "logs", {"container": container_name} + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=137, + is_error=True, + details={"exit_code": 137, "oom_killed": True, "memory_limit": "6m"}, + ), + Observation( + source="docker.logs", + content=logs_res.summary or "", + exit_code=137, + is_error=True, + ), + ) + + # Simulation mode deterministic observations + obs1 = Observation( + source="docker.run", + content=( + f"Container {container_name} terminated with exit code 137 " + "(OOMKilled: true, memory_limit: 6m) (simulated)" + ), + exit_code=137, + is_error=True, + details={"exit_code": 137, "oom_killed": True, "memory_limit": "6m"}, + ) + obs2 = Observation( + source="docker.logs", + content=( + " " + "(simulated)" + ), + exit_code=137, + is_error=True, + ) + obs3 = Observation( + source="http_probe", + content=( + f"Health probe to http://127.0.0.1:{port}/health failed: " + "connection refused (process terminated) (simulated)" + ), + is_error=True, + ) + return (obs1, obs2, obs3) + + def remediate( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> tuple[Observation, ...]: + limit_val: Any = attempt.parameters.get( + "memory_limit", attempt.parameters.get("memory", attempt.parameters.get("limit")) + ) + if limit_val is None: + for part in attempt.action.replace("=", " ").split(): + if any(c.isdigit() for c in part) and any( + unit in part.lower() for unit in ("m", "g", "mb", "gb") + ): + limit_val = part + break + + mem_mb = _parse_memory_mb(str(limit_val or "")) + context.state["memory_mb"] = mem_mb + context.state["memory_limit"] = str(limit_val or "") + + if mem_mb < 32: + return ( + Observation( + source="remediation", + content=( + f"Failed remediation: Memory limit '{limit_val}' ({mem_mb}MB) is " + "insufficient. Python/FastAPI requires at least 32MB." + ), + is_error=True, + ), + ) + + if context.is_live: + c_name = context.state.get("container_name", f"api-troubleshoot-oom-{id(context)}") + context.tool_service.invoke("docker", "stop", {"container": c_name}) + port = context.state.get("port", 8000) + run_res = context.tool_service.invoke( + "docker", + "run", + { + "image": "api-platform:dev", + "name": c_name, + "ports": {str(port): "8000"}, + }, + ) + return ( + Observation( + source="docker.run", + content=run_res.summary, + exit_code=0 if run_res.success else 1, + is_error=not run_res.success, + details=dict(run_res.details), + ), + ) + + return ( + Observation( + source="docker.run", + content=( + f"Container started with memory limit {limit_val} ({mem_mb}MB) and " + "stays running (simulated)" + ), + exit_code=0, + is_error=False, + details={"memory_limit": limit_val, "memory_mb": mem_mb}, + ), + ) + + def verify( + self, context: ScenarioContext, attempt: RemediationAttempt + ) -> VerificationResult: + mem_mb = context.state.get("memory_mb", 0) + if mem_mb < 32: + obs = Observation( + source="verification", + content=( + f"Verification failed: Container memory allocation ({mem_mb}MB) " + "is insufficient, leading to OOM termination." + ), + exit_code=137, + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Recovery failed: Memory limit ({mem_mb}MB) is below threshold (32MB).", + observations=(obs,), + is_live=context.is_live, + ) + + if context.is_live: + port = context.state.get("port", 8000) + url = f"http://127.0.0.1:{port}/health" + try: + with urllib.request.urlopen(url, timeout=3) as resp: + body = resp.read().decode("utf-8") + obs = Observation( + source="http_probe", + content=f"Health check OK ({resp.status}): {body}", + exit_code=resp.status, + is_error=False, + ) + return VerificationResult( + success=True, + summary=f"OOM failure resolved. Running steadily within {mem_mb}MB limit.", + observations=(obs,), + is_live=True, + ) + except (urllib.error.URLError, OSError) as exc: + obs = Observation( + source="http_probe", + content=f"Health probe to {url} failed: {exc}", + is_error=True, + ) + return VerificationResult( + success=False, + summary=f"Verification failed: Container failed to respond at {url}", + observations=(obs,), + is_live=True, + ) + + obs = Observation( + source="http_probe", + content="Health check OK (200): {\"status\": \"ok\"} (simulated)", + exit_code=200, + is_error=False, + ) + return VerificationResult( + success=True, + summary=( + f"OOM failure resolved. Container running steadily within {mem_mb}MB limit " + "(simulated)." + ), + observations=(obs,), + is_live=False, + details={"memory_mb": mem_mb, "status_code": 200}, + ) + + def cleanup(self, context: ScenarioContext) -> None: + container_name = context.state.get("container_name") + if container_name: + context.tool_service.invoke("docker", "stop", {"container": container_name}) diff --git a/src/devops_learn/troubleshooting/service.py b/src/devops_learn/troubleshooting/service.py index 1b9e8bb..0c9ae43 100644 --- a/src/devops_learn/troubleshooting/service.py +++ b/src/devops_learn/troubleshooting/service.py @@ -1,24 +1,182 @@ """TroubleshootingService: gathers structured evidence before ever asking for -a diagnosis, then produces one. +a diagnosis, executes scenarios, provides progressive hints, and deterministically +verifies recovery. Per the product spec, the AI is never handed a one-line failure description -and asked to guess: it always receives EvidenceItem entries gathered from -real (or, in simulation mode, simulated) ToolResult output first. See +and asked to guess: it always receives EvidenceItem / Observation entries gathered +from real (or, in simulation mode, simulated) ToolResult output first. See docs/architecture.md#troubleshooting. """ from __future__ import annotations -from devops_learn.domain.troubleshooting_models import Diagnosis, EvidenceItem, FailureEvent +from devops_learn.domain.troubleshooting_models import ( + Diagnosis, + EvidenceItem, + FailureEvent, + HintLevel, + Interpretation, + Observation, + RemediationAttempt, + TroubleshootingEvidence, + TroubleshootingScenario, + TroubleshootingSession, + VerificationResult, +) from devops_learn.tools.service import ToolService +from devops_learn.troubleshooting.scenarios.base import ScenarioContext +from devops_learn.troubleshooting.scenarios.registry import ScenarioRegistry class TroubleshootingService: - def __init__(self, tool_service: ToolService) -> None: + def __init__( + self, + tool_service: ToolService, + registry: ScenarioRegistry | None = None, + ) -> None: self._tool_service = tool_service + self._registry = registry or ScenarioRegistry() + + def list_scenarios(self) -> tuple[TroubleshootingScenario, ...]: + return self._registry.list_scenarios() + + def get_scenario(self, scenario_id: str) -> TroubleshootingScenario: + return self._registry.get_scenario(scenario_id) + + def start_session( + self, + scenario_id: str, + *, + project_root: str = ".", + is_live: bool = False, + ) -> tuple[TroubleshootingSession, ScenarioContext, tuple[Observation, ...]]: + handler = self._registry.get_handler(scenario_id) + scenario = handler.definition + context = ScenarioContext( + scenario=scenario, + is_live=is_live, + project_root=project_root, + tool_service=self._tool_service, + ) + initial_observations = handler.setup_and_inject(context) + mode_label = "(real)" if is_live else "(simulated)" + evidence = TroubleshootingEvidence( + scenario_id=scenario_id, + before_state=initial_observations, + mode_label=mode_label, + ) + session = TroubleshootingSession( + scenario=scenario, + is_live=is_live, + project_root=project_root, + evidence=evidence, + active=True, + ) + return session, context, initial_observations + + def get_hint(self, scenario_id: str, level: int | HintLevel) -> str: + scenario = self.get_scenario(scenario_id) + int_level = int(level) + if int_level in scenario.hints: + return scenario.hints[int_level] + if int_level <= 0: + return scenario.hints.get(0, "No evidence hint available.") + max_level = max(scenario.hints.keys()) + return scenario.hints.get(max_level, "No further hints available.") + + def interpret(self, observations: tuple[Observation, ...]) -> tuple[Interpretation, ...]: + interpretations: list[Interpretation] = [] + for obs in observations: + if not obs.is_error: + continue + lower_content = obs.content.lower() + if ( + "address already in use" in lower_content + or "port is already allocated" in lower_content + ): + interpretations.append( + Interpretation( + observation_summary="Port bind conflict detected", + likely_subsystem="Networking / Socket Binding", + hypothesis="The requested host port is already bound by another process.", + confidence=0.95, + ) + ) + elif "required_config_key" in lower_content or ( + "missing" in lower_content and "config" in lower_content + ): + interpretations.append( + Interpretation( + observation_summary="Missing required configuration variable", + likely_subsystem="Application Configuration", + hypothesis=( + "Startup initialization failed because a required environment variable " + "was not supplied." + ), + confidence=0.95, + ) + ) + elif "503" in lower_content or "degraded" in lower_content: + interpretations.append( + Interpretation( + observation_summary="Health probe returned HTTP 503 degraded", + likely_subsystem="Observability / Health Probes", + hypothesis=( + "Process is running but internal dependency check flagged " + "degraded readiness." + ), + confidence=0.90, + ) + ) + elif ( + "137" in str(obs.exit_code) + or "oom" in lower_content + or "sigkill" in lower_content + ): + interpretations.append( + Interpretation( + observation_summary="Process terminated by OOM killer (exit code 137)", + likely_subsystem="Resource Constraints / CGroups", + hypothesis=( + "Memory limit was exceeded during startup, causing kernel SIGKILL." + ), + confidence=0.95, + ) + ) + return tuple(interpretations) + + def remediate( + self, + session: TroubleshootingSession, + context: ScenarioContext, + attempt: RemediationAttempt, + ) -> tuple[Observation, ...]: + handler = self._registry.get_handler(session.scenario.scenario_id) + return handler.remediate(context, attempt) + + def verify( + self, + session: TroubleshootingSession, + context: ScenarioContext, + attempt: RemediationAttempt, + ) -> VerificationResult: + handler = self._registry.get_handler(session.scenario.scenario_id) + return handler.verify(context, attempt) + + def cleanup( + self, + session: TroubleshootingSession, + context: ScenarioContext, + ) -> None: + handler = self._registry.get_handler(session.scenario.scenario_id) + handler.cleanup(context) + + # ------------------------------------------------------------------------- + # Backward compatibility with V1 simulated Kubernetes failure + # ------------------------------------------------------------------------- def gather_evidence(self) -> FailureEvent: - """Collects evidence for the one intentional V1 simulated failure: a pod that + """Collects evidence for the intentional V1 simulated failure: a pod that never becomes ready because its readiness probe targets the wrong path.""" pods = self._tool_service.invoke("kubernetes", "get_pods") self._tool_service.invoke("kubernetes", "describe") diff --git a/src/devops_learn/workflows/troubleshooting_flow.py b/src/devops_learn/workflows/troubleshooting_flow.py new file mode 100644 index 0000000..0f373be --- /dev/null +++ b/src/devops_learn/workflows/troubleshooting_flow.py @@ -0,0 +1,282 @@ +"""Troubleshooting workflow: fault injection -> observation -> progressive assistance +-> remediation -> deterministic recovery verification -> cleanup. + +Per the product spec, scenarios follow an explicit lifecycle: +SETUP -> INJECT -> OBSERVE -> EXPLAIN -> REMEDIATE -> VERIFY -> CLEANUP. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from datetime import datetime, timezone +from typing import Any, Mapping + +from devops_learn.bootstrap import Platform +from devops_learn.domain.enums import ( + AuditEventType, + CloudProviderKind, + CostPriority, + EnvironmentKind, + ExecutionMode, + ExperienceState, + ExplanationDepth, +) +from devops_learn.domain.question_models import ClarifyingQuestion +from devops_learn.domain.troubleshooting_models import ( + Observation, + RemediationAttempt, + TroubleshootingEvidence, + VerificationResult, +) +from devops_learn.workflows.ui import Ui + + +@dataclass(frozen=True) +class TroubleshootingOptions: + scenario_id: str + hint_level: int | None = None + remediation_action: str | None = None + remediation_params: Mapping[str, Any] = field(default_factory=dict) + project_root: str = "." + simulate: bool | None = None + interactive: bool = False + + +def list_troubleshooting_scenarios(platform: Platform, ui: Ui) -> None: + scenarios = platform.troubleshooting_service.list_scenarios() + lines = [ + "============================================================", + "AVAILABLE TROUBLESHOOTING SCENARIOS", + "============================================================", + "", + ] + for s in scenarios: + lines.extend( + ( + f"[{s.scenario_id}] {s.title}", + f" Category: {s.category.value}", + f" Objective: {s.learning_objective}", + f" Fault: {s.fault_description}", + "", + ) + ) + lines.append("Run a scenario with: devops-learn troubleshoot run ") + ui.present("\n".join(lines)) + + +def run_troubleshooting_flow( + platform: Platform, + ui: Ui, + options: TroubleshootingOptions, +) -> TroubleshootingEvidence: + # 1. Determine execution capability (real vs simulated) + is_live = False + if options.simulate is False or options.simulate is None: + docker_available = False + try: + doc_res = platform.tool_service.invoke( + "docker", "logs", {"container": "nonexistent_test"} + ) + docker_available = not doc_res.summary.endswith("(simulated)") + except Exception: + docker_available = False + + if options.simulate is False and not docker_available: + ui.present( + "[!] Real Docker environment requested but unavailable. " + "Falling back safely to simulated mode." + ) + is_live = False + else: + is_live = docker_available + + mode_label = "LIVE VERIFIED" if is_live else "SIMULATED / TESTED" + + # 2. Start engagement session for tracking and persistence + session = platform.session_service.start( + project_root=options.project_root, + mode=ExecutionMode.COLLABORATIVE, + explanation_depth=ExplanationDepth.LEARNING, + cloud=CloudProviderKind.AZURE, + environment=EnvironmentKind.LOCAL, + cost_priority=CostPriority.BALANCED, + simulation_mode=not is_live, + ) + assert session.id is not None + session_id = session.id + + platform.audit_service.record( + session_id=session_id, + event_type=AuditEventType.TROUBLESHOOTING_STARTED, + occurred_at=datetime.now(timezone.utc), + summary=f"Started troubleshooting scenario: {options.scenario_id} ({mode_label})", + payload={"scenario_id": options.scenario_id, "is_live": is_live}, + ) + + # 3. Setup and Inject Fault + tb_session, context, before_obs = platform.troubleshooting_service.start_session( + options.scenario_id, + project_root=options.project_root, + is_live=is_live, + ) + scenario = tb_session.scenario + + ui.present( + f"============================================================\n" + f"TROUBLESHOOTING: {scenario.title}\n" + f"Execution Mode: {mode_label}\n" + f"Category: {scenario.category.value}\n" + f"Objective: {scenario.learning_objective}\n" + f"============================================================\n" + ) + + ui.present("--- LEVEL 0: RAW OBSERVATIONS (EVIDENCE ONLY) ---") + for obs in before_obs: + status_tag = "[ERROR]" if obs.is_error else "[INFO]" + ui.present(f"{status_tag} ({obs.source}) {obs.content}") + ui.present("") + + # 4. Progressive Assistance / Hints + requested_hint_level = options.hint_level + if options.interactive and requested_hint_level is None: + choice = ui.ask_choice( + ClarifyingQuestion( + id="hint_request", + category="troubleshooting", + prompt="Would you like a progressive hint before attempting remediation?", + options=( + "Level 0: Proceed with evidence only", + "Level 1: Inspection guide (where to look)", + "Level 2: Subsystem explanation", + "Level 3: Root cause explanation", + "Level 4: Suggested remediation", + ), + ) + ) + if "Level 1" in choice: + requested_hint_level = 1 + elif "Level 2" in choice: + requested_hint_level = 2 + elif "Level 3" in choice: + requested_hint_level = 3 + elif "Level 4" in choice: + requested_hint_level = 4 + else: + requested_hint_level = 0 + + if requested_hint_level is not None and requested_hint_level > 0: + ui.present(f"--- PROGRESSIVE ASSISTANCE (LEVEL 1 TO {requested_hint_level}) ---") + for lvl in range(1, requested_hint_level + 1): + hint_text = platform.troubleshooting_service.get_hint(scenario.scenario_id, lvl) + ui.present(f"[HINT LEVEL {lvl}] {hint_text}") + ui.present("") + + # 5. Remediation & Verification + remediation_action = options.remediation_action + remediation_params = dict(options.remediation_params) + + if options.interactive and not remediation_action and not remediation_params: + rem_input = ui.ask_choice( + ClarifyingQuestion( + id="remediation_input", + category="troubleshooting", + prompt=( + "Enter remediation parameter (e.g. port=8081, " + "REQUIRED_CONFIG_KEY=value, dependency_status=healthy, memory_limit=64m):" + ), + options=(), + ) + ) + remediation_action = rem_input + for token in rem_input.split(): + if "=" in token: + k, v = token.split("=", 1) + remediation_params[k.strip()] = v.strip() + + attempt = RemediationAttempt( + scenario_id=scenario.scenario_id, + action=remediation_action or "", + parameters=remediation_params, + ) + + after_obs: tuple[Observation, ...] = () + verification: VerificationResult | None = None + resolved = False + + try: + if attempt.action or attempt.parameters: + platform.audit_service.record( + session_id=session_id, + event_type=AuditEventType.TROUBLESHOOTING_REMEDIATION_ATTEMPTED, + occurred_at=datetime.now(timezone.utc), + summary=f"Remediation attempted: {attempt.action or attempt.parameters}", + payload={"action": attempt.action, "params": dict(attempt.parameters)}, + ) + ui.present("--- APPLYING REMEDIATION ---") + after_obs = platform.troubleshooting_service.remediate(tb_session, context, attempt) + for obs in after_obs: + status_tag = "[ERROR]" if obs.is_error else "[OK]" + ui.present(f"{status_tag} ({obs.source}) {obs.content}") + ui.present("") + + ui.present("--- DETERMINISTIC RECOVERY VERIFICATION ---") + verification = platform.troubleshooting_service.verify(tb_session, context, attempt) + resolved = verification.success + + if resolved: + ui.present(f"[PASS] {verification.summary}") + platform.audit_service.record( + session_id=session_id, + event_type=AuditEventType.TROUBLESHOOTING_VERIFIED, + occurred_at=datetime.now(timezone.utc), + summary=f"Recovery verified: {verification.summary}", + payload={"success": True, "mode": mode_label}, + ) + platform.experience_tracker.record( + session_id, + scenario.category.value.title(), + f"Resolved incident: {scenario.title}", + ExperienceState.DEMONSTRATED, + ) + else: + ui.present(f"[FAIL] {verification.summary}") + platform.audit_service.record( + session_id=session_id, + event_type=AuditEventType.TROUBLESHOOTING_FAILED, + occurred_at=datetime.now(timezone.utc), + summary=f"Verification failed: {verification.summary}", + payload={"success": False, "mode": mode_label}, + ) + else: + ui.present("--- NO REMEDIATION SUPPLIED ---") + verification = VerificationResult( + success=False, + summary="No remediation attempt was supplied; failure condition persists.", + is_live=is_live, + ) + finally: + # 6. Guaranteed Cleanup + platform.troubleshooting_service.cleanup(tb_session, context) + ui.present("\n[✓] Teardown & cleanup completed successfully.\n") + + platform.audit_service.record( + session_id=session_id, + event_type=AuditEventType.TROUBLESHOOTING_COMPLETED, + occurred_at=datetime.now(timezone.utc), + summary=( + f"Troubleshooting session completed for {options.scenario_id} (Resolved: {resolved})" + ), + payload={"resolved": resolved}, + ) + platform.session_service.complete(session) + + evidence = TroubleshootingEvidence( + scenario_id=scenario.scenario_id, + before_state=before_obs, + remediation=attempt, + after_state=after_obs, + verification=verification, + resolved=resolved, + mode_label=mode_label, + ) + return evidence diff --git a/tests/cli/test_troubleshoot_command.py b/tests/cli/test_troubleshoot_command.py new file mode 100644 index 0000000..0f82fd4 --- /dev/null +++ b/tests/cli/test_troubleshoot_command.py @@ -0,0 +1,43 @@ +"""CLI command integration tests for `devops-learn troubleshoot`.""" + +import pytest +from devops_learn.cli.main import build_parser, main + + +def test_troubleshoot_parser_registration() -> None: + parser = build_parser() + args = parser.parse_args(["troubleshoot", "list"]) + assert args.command == "troubleshoot" + assert args.troubleshoot_command == "list" + + args = parser.parse_args( + ["troubleshoot", "run", "port_conflict", "--hint-level", "2", "--remediation", "port=8081"] + ) + assert args.command == "troubleshoot" + assert args.troubleshoot_command == "run" + assert args.scenario == "port_conflict" + assert args.hint_level == 2 + assert args.remediation == "port=8081" + + +def test_troubleshoot_cli_execution_success(capsys: pytest.CaptureFixture[str]) -> None: + main(["troubleshoot", "run", "port_conflict", "--remediation", "port=8081", "--simulate"]) + captured = capsys.readouterr() + assert "TROUBLESHOOTING: Port Binding Collision" in captured.out + assert "[PASS] Port collision resolved." in captured.out + + +def test_troubleshoot_cli_execution_failure(capsys: pytest.CaptureFixture[str]) -> None: + with pytest.raises(SystemExit) as exc_info: + main(["troubleshoot", "run", "port_conflict", "--remediation", "port=8000", "--simulate"]) + assert exc_info.value.code == 1 + captured = capsys.readouterr() + assert "[FAIL]" in captured.out + + +def test_troubleshoot_cli_list(capsys: pytest.CaptureFixture[str]) -> None: + main(["troubleshoot", "list"]) + captured = capsys.readouterr() + assert "AVAILABLE TROUBLESHOOTING SCENARIOS" in captured.out + assert "[port_conflict]" in captured.out + assert "[missing_config]" in captured.out diff --git a/tests/troubleshooting/test_falsification.py b/tests/troubleshooting/test_falsification.py new file mode 100644 index 0000000..3877a32 --- /dev/null +++ b/tests/troubleshooting/test_falsification.py @@ -0,0 +1,191 @@ +"""Rigorous test falsification for troubleshooting scenarios and recovery verification. + +These tests prove that scenarios cannot pass spuriously: +- Wrong or incomplete remediation fails. +- No remediation fails. +- Broken before-state fails verification. +- Cleanup is guaranteed even during exceptions. +- Progressive hints don't leak remediation at level 0. +- Observations accurately distinguish facts from interpretation. +""" + +from devops_learn.domain.troubleshooting_models import ( + HintLevel, + RemediationAttempt, +) +from devops_learn.tools.approval import AutoApproveApprovalGate +from devops_learn.tools.docker_tool import SimulatedDockerTool +from devops_learn.tools.service import ToolService +from devops_learn.troubleshooting.scenarios.port_conflict import ( + PortConflictScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.registry import ScenarioRegistry +from devops_learn.troubleshooting.service import TroubleshootingService + + +def _get_service() -> TroubleshootingService: + tool_service = ToolService({"docker": SimulatedDockerTool()}, AutoApproveApprovalGate()) + return TroubleshootingService(tool_service) + + +def test_falsify_port_conflict_remediation_failures() -> None: + service = _get_service() + session, ctx, obs = service.start_session("port_conflict", is_live=False) + + # 1. No remediation attempt fails + empty_attempt = RemediationAttempt("port_conflict", "") + res_empty = service.verify(session, ctx, empty_attempt) + assert not res_empty.success + assert "cannot bind to occupied" in res_empty.summary + + # 2. Re-attempting occupied port (8000) fails + conflict_attempt = RemediationAttempt("port_conflict", "port=8000", {"port": 8000}) + rem_obs = service.remediate(session, ctx, conflict_attempt) + assert rem_obs[0].is_error + res_conflict = service.verify(session, ctx, conflict_attempt) + assert not res_conflict.success + + # 3. Invalid port numbers fail + for bad_port in [-5, 0, 70000, "not-a-port"]: + bad_attempt = RemediationAttempt("port_conflict", f"port={bad_port}", {"port": bad_port}) + rem_obs = service.remediate(session, ctx, bad_attempt) + assert rem_obs[0].is_error + res = service.verify(session, ctx, bad_attempt) + assert not res.success + + # 4. Valid non-conflicting port succeeds + valid_attempt = RemediationAttempt("port_conflict", "port=8082", {"port": 8082}) + rem_obs = service.remediate(session, ctx, valid_attempt) + assert not rem_obs[0].is_error + res_valid = service.verify(session, ctx, valid_attempt) + assert res_valid.success + assert res_valid.details.get("port") == 8082 + + service.cleanup(session, ctx) + + +def test_falsify_missing_config_remediation_failures() -> None: + service = _get_service() + session, ctx, obs = service.start_session("missing_config", is_live=False) + + # 1. Empty remediation fails + empty_attempt = RemediationAttempt("missing_config", "") + assert not service.verify(session, ctx, empty_attempt).success + + # 2. Unrelated config fails + wrong_attempt = RemediationAttempt( + "missing_config", "UNRELATED_KEY=true", {"env": {"UNRELATED_KEY": "true"}} + ) + rem_obs = service.remediate(session, ctx, wrong_attempt) + assert rem_obs[0].is_error + assert not service.verify(session, ctx, wrong_attempt).success + + # 3. Supplying REQUIRED_CONFIG_KEY succeeds + good_attempt = RemediationAttempt( + "missing_config", + "REQUIRED_CONFIG_KEY=learning_secret_token", + {"env": {"REQUIRED_CONFIG_KEY": "learning_secret_token"}}, + ) + rem_obs = service.remediate(session, ctx, good_attempt) + assert not rem_obs[0].is_error + res = service.verify(session, ctx, good_attempt) + assert res.success + assert res.details.get("config_verified") is True + + service.cleanup(session, ctx) + + +def test_falsify_health_check_remediation_failures() -> None: + service = _get_service() + session, ctx, obs = service.start_session("health_check_failure", is_live=False) + + # 1. Still degraded fails + bad_attempt = RemediationAttempt( + "health_check_failure", "status=unhealthy", {"dependency_status": "unhealthy"} + ) + rem_obs = service.remediate(session, ctx, bad_attempt) + assert rem_obs[0].is_error + assert not service.verify(session, ctx, bad_attempt).success + + # 2. Setting healthy succeeds + good_attempt = RemediationAttempt( + "health_check_failure", + "dependency_status=healthy", + {"dependency_status": "healthy"}, + ) + rem_obs = service.remediate(session, ctx, good_attempt) + assert not rem_obs[0].is_error + res = service.verify(session, ctx, good_attempt) + assert res.success + + service.cleanup(session, ctx) + + +def test_falsify_resource_limit_remediation_failures() -> None: + service = _get_service() + session, ctx, obs = service.start_session("resource_limit", is_live=False) + + # 1. Memory below 32MB fails + for small_mem in ["4m", "6m", "16m", "20M"]: + bad_attempt = RemediationAttempt( + "resource_limit", f"memory_limit={small_mem}", {"memory_limit": small_mem} + ) + rem_obs = service.remediate(session, ctx, bad_attempt) + assert rem_obs[0].is_error + assert not service.verify(session, ctx, bad_attempt).success + + # 2. Memory 64MB succeeds + good_attempt = RemediationAttempt( + "resource_limit", "memory_limit=64m", {"memory_limit": "64m"} + ) + rem_obs = service.remediate(session, ctx, good_attempt) + assert not rem_obs[0].is_error + res = service.verify(session, ctx, good_attempt) + assert res.success + assert res.details.get("memory_mb") == 64 + + # 3. Memory 1GB succeeds + gb_attempt = RemediationAttempt( + "resource_limit", "memory_limit=1g", {"memory_limit": "1g"} + ) + rem_obs = service.remediate(session, ctx, gb_attempt) + assert not rem_obs[0].is_error + res_gb = service.verify(session, ctx, gb_attempt) + assert res_gb.success + assert res_gb.details.get("memory_mb") == 1024 + + service.cleanup(session, ctx) + + +def test_falsify_cleanup_guarantee() -> None: + service = _get_service() + cleaned = False + + class MonitoredHandler(PortConflictScenarioHandler): + def cleanup(self, context): + nonlocal cleaned + cleaned = True + super().cleanup(context) + + registry = ScenarioRegistry((MonitoredHandler(),)) + custom_service = TroubleshootingService(service._tool_service, registry) + cust_session, cust_ctx, _ = custom_service.start_session("port_conflict", is_live=False) + + try: + raise RuntimeError("Simulated mid-troubleshooting crash") + except RuntimeError: + custom_service.cleanup(cust_session, cust_ctx) + + assert cleaned is True + + +def test_falsify_hints_do_not_leak_solution_at_level_0() -> None: + service = _get_service() + scenarios = ["port_conflict", "missing_config", "health_check_failure", "resource_limit"] + for scenario_id in scenarios: + h0 = service.get_hint(scenario_id, HintLevel.EVIDENCE) + h4 = service.get_hint(scenario_id, HintLevel.REMEDIATION) + + assert "Observation:" in h0 + assert "Remediation:" in h4 + assert "Remediation:" not in h0 diff --git a/tests/troubleshooting/test_scenarios.py b/tests/troubleshooting/test_scenarios.py new file mode 100644 index 0000000..b43d897 --- /dev/null +++ b/tests/troubleshooting/test_scenarios.py @@ -0,0 +1,175 @@ +"""Unit tests for troubleshooting scenario handlers and registry.""" + +from typing import Any +import pytest + +from devops_learn.domain.troubleshooting_models import RemediationAttempt +from devops_learn.tools.approval import AutoApproveApprovalGate +from devops_learn.tools.docker_tool import SimulatedDockerTool +from devops_learn.tools.service import ToolService +from devops_learn.troubleshooting.scenarios.base import ScenarioContext +from devops_learn.troubleshooting.scenarios.health_check_failure import ( + HealthCheckFailureScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.missing_config import ( + MissingConfigScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.port_conflict import ( + PortConflictScenarioHandler, +) +from devops_learn.troubleshooting.scenarios.registry import ScenarioRegistry +from devops_learn.troubleshooting.scenarios.resource_limit import ( + ResourceLimitScenarioHandler, +) + + +def _make_context(handler: Any) -> ScenarioContext: + tool_service = ToolService({"docker": SimulatedDockerTool()}, AutoApproveApprovalGate()) + return ScenarioContext( + scenario=handler.definition, + is_live=False, + project_root=".", + tool_service=tool_service, + ) + + +def test_scenario_registry_contains_all_core_scenarios() -> None: + registry = ScenarioRegistry() + scenarios = registry.list_scenarios() + ids = {s.scenario_id for s in scenarios} + assert "port_conflict" in ids + assert "missing_config" in ids + assert "health_check_failure" in ids + assert "resource_limit" in ids + assert len(scenarios) == 4 + + +def test_scenario_registry_raises_for_unknown_scenario() -> None: + registry = ScenarioRegistry() + with pytest.raises(KeyError) as exc_info: + registry.get_handler("nonexistent_scenario") + assert "Available scenarios:" in str(exc_info.value) + + +def test_port_conflict_lifecycle() -> None: + handler = PortConflictScenarioHandler() + ctx = _make_context(handler) + + # 1. Setup & Inject + obs = handler.setup_and_inject(ctx) + assert any( + "port is already allocated" in o.content or "address already in use" in o.content + for o in obs + ) + assert any(o.is_error for o in obs) + + # 2. Bad remediation (port 8000 still conflicting) + bad_attempt = RemediationAttempt("port_conflict", "port=8000", {"port": 8000}) + bad_obs = handler.remediate(ctx, bad_attempt) + assert bad_obs[0].is_error + bad_ver = handler.verify(ctx, bad_attempt) + assert not bad_ver.success + + # 3. Good remediation (port 8081) + good_attempt = RemediationAttempt("port_conflict", "port=8081", {"port": 8081}) + good_obs = handler.remediate(ctx, good_attempt) + assert not good_obs[0].is_error + good_ver = handler.verify(ctx, good_attempt) + assert good_ver.success + assert "8081" in good_ver.summary + + # 4. Cleanup + handler.cleanup(ctx) + + +def test_missing_config_lifecycle() -> None: + handler = MissingConfigScenarioHandler() + ctx = _make_context(handler) + + # 1. Setup & Inject + obs = handler.setup_and_inject(ctx) + assert any("REQUIRED_CONFIG_KEY" in o.content for o in obs) + + # 2. Bad remediation + bad_attempt = RemediationAttempt("missing_config", "wrong_param=1", {}) + bad_obs = handler.remediate(ctx, bad_attempt) + assert bad_obs[0].is_error + bad_ver = handler.verify(ctx, bad_attempt) + assert not bad_ver.success + + # 3. Good remediation + good_attempt = RemediationAttempt( + "missing_config", + "REQUIRED_CONFIG_KEY=val", + {"REQUIRED_CONFIG_KEY": "valid_123"}, + ) + good_obs = handler.remediate(ctx, good_attempt) + assert not good_obs[0].is_error + good_ver = handler.verify(ctx, good_attempt) + assert good_ver.success + + # 4. Cleanup + handler.cleanup(ctx) + + +def test_health_check_failure_lifecycle() -> None: + handler = HealthCheckFailureScenarioHandler() + ctx = _make_context(handler) + + # 1. Setup & Inject + obs = handler.setup_and_inject(ctx) + assert any("503" in o.content for o in obs) + + # 2. Bad remediation + bad_attempt = RemediationAttempt( + "health_check_failure", + "status=unhealthy", + {"dependency_status": "unhealthy"}, + ) + bad_obs = handler.remediate(ctx, bad_attempt) + assert bad_obs[0].is_error + bad_ver = handler.verify(ctx, bad_attempt) + assert not bad_ver.success + + # 3. Good remediation + good_attempt = RemediationAttempt( + "health_check_failure", + "dependency_status=healthy", + {"dependency_status": "healthy"}, + ) + good_obs = handler.remediate(ctx, good_attempt) + assert not good_obs[0].is_error + good_ver = handler.verify(ctx, good_attempt) + assert good_ver.success + + # 4. Cleanup + handler.cleanup(ctx) + + +def test_resource_limit_lifecycle() -> None: + handler = ResourceLimitScenarioHandler() + ctx = _make_context(handler) + + # 1. Setup & Inject + obs = handler.setup_and_inject(ctx) + assert any(o.exit_code == 137 for o in obs) + + # 2. Bad remediation (too small memory limit: 8m) + bad_attempt = RemediationAttempt("resource_limit", "memory_limit=8m", {"memory_limit": "8m"}) + bad_obs = handler.remediate(ctx, bad_attempt) + assert bad_obs[0].is_error + bad_ver = handler.verify(ctx, bad_attempt) + assert not bad_ver.success + + # 3. Good remediation (64m) + good_attempt = RemediationAttempt( + "resource_limit", "memory_limit=64m", {"memory_limit": "64m"} + ) + good_obs = handler.remediate(ctx, good_attempt) + assert not good_obs[0].is_error + good_ver = handler.verify(ctx, good_attempt) + assert good_ver.success + assert "64" in good_ver.summary + + # 4. Cleanup + handler.cleanup(ctx) diff --git a/tests/troubleshooting/test_service.py b/tests/troubleshooting/test_service.py index c4e98fb..135bb1e 100644 --- a/tests/troubleshooting/test_service.py +++ b/tests/troubleshooting/test_service.py @@ -1,11 +1,24 @@ +from devops_learn.domain.troubleshooting_models import ( + EvidenceItem, + FailureEvent, + HintLevel, + RemediationAttempt, +) from devops_learn.tools.approval import AutoApproveApprovalGate +from devops_learn.tools.docker_tool import SimulatedDockerTool from devops_learn.tools.kubernetes_tool import SimulatedKubernetesTool from devops_learn.tools.service import ToolService from devops_learn.troubleshooting.service import TroubleshootingService def _service() -> TroubleshootingService: - tool_service = ToolService({"kubernetes": SimulatedKubernetesTool()}, AutoApproveApprovalGate()) + tool_service = ToolService( + { + "kubernetes": SimulatedKubernetesTool(), + "docker": SimulatedDockerTool(), + }, + AutoApproveApprovalGate(), + ) return TroubleshootingService(tool_service) @@ -25,9 +38,50 @@ def test_diagnosis_is_derived_from_relevant_evidence_only() -> None: def test_diagnosis_without_relevant_evidence_is_honest_about_uncertainty() -> None: - from devops_learn.domain.troubleshooting_models import EvidenceItem, FailureEvent - service = _service() failure = FailureEvent(title="x", narrative="y", evidence=(EvidenceItem("s", "c", False),)) diagnosis = service.diagnose(failure) assert diagnosis.likely_cause == "Unknown" + + +def test_list_and_get_scenarios() -> None: + service = _service() + scenarios = service.list_scenarios() + assert len(scenarios) == 4 + scenario = service.get_scenario("port_conflict") + assert scenario.scenario_id == "port_conflict" + assert "EADDRINUSE" in scenario.title + + +def test_progressive_hints_levels() -> None: + service = _service() + h0 = service.get_hint("port_conflict", HintLevel.EVIDENCE) + h1 = service.get_hint("port_conflict", HintLevel.INSPECTION) + h2 = service.get_hint("port_conflict", HintLevel.SUBSYSTEM) + h3 = service.get_hint("port_conflict", HintLevel.ROOT_CAUSE) + h4 = service.get_hint("port_conflict", HintLevel.REMEDIATION) + + assert "Observation:" in h0 + assert "Inspection:" in h1 + assert "Subsystem:" in h2 + assert "Root Cause:" in h3 + assert "Remediation:" in h4 + + +def test_service_start_session_and_interpret() -> None: + service = _service() + session, ctx, obs = service.start_session("port_conflict", is_live=False) + assert session.scenario.scenario_id == "port_conflict" + assert len(obs) >= 2 + + interpretations = service.interpret(obs) + assert len(interpretations) >= 1 + assert "Port bind conflict" in interpretations[0].observation_summary + + attempt = RemediationAttempt("port_conflict", "port=8081", {"port": 8081}) + rem_obs = service.remediate(session, ctx, attempt) + assert not rem_obs[0].is_error + + ver = service.verify(session, ctx, attempt) + assert ver.success + service.cleanup(session, ctx) diff --git a/tests/workflows/test_troubleshoot_flow.py b/tests/workflows/test_troubleshoot_flow.py new file mode 100644 index 0000000..c1385af --- /dev/null +++ b/tests/workflows/test_troubleshoot_flow.py @@ -0,0 +1,116 @@ +import sqlite3 + +from devops_learn.bootstrap import build_platform +from devops_learn.domain.enums import AuditEventType, ExperienceState +from devops_learn.domain.question_models import ClarifyingQuestion +from devops_learn.tools.approval import AutoApproveApprovalGate +from devops_learn.workflows.troubleshooting_flow import ( + TroubleshootingOptions, + list_troubleshooting_scenarios, + run_troubleshooting_flow, +) +from devops_learn.workflows.ui import Ui + + +class FakeUi(Ui): + def __init__(self, choices: list[str] | None = None) -> None: + self.presented: list[str] = [] + self.choices = list(choices or []) + + def present(self, text: str) -> None: + self.presented.append(text) + + def ask_choice(self, question: ClarifyingQuestion) -> str: + if self.choices: + return self.choices.pop(0) + return question.options[0] if question.options else "port=8081" + + def confirm(self, prompt: str, *, default: bool = False) -> bool: + return True + + +def _platform(conn: sqlite3.Connection): + return build_platform(conn, approval_gate=AutoApproveApprovalGate()) + + +def test_list_scenarios_presents_all_options(conn: sqlite3.Connection) -> None: + platform = _platform(conn) + ui = FakeUi() + list_troubleshooting_scenarios(platform, ui) + output = "\n".join(ui.presented) + assert "[port_conflict]" in output + assert "[missing_config]" in output + assert "[health_check_failure]" in output + assert "[resource_limit]" in output + + +def test_troubleshooting_flow_successful_recovery(conn: sqlite3.Connection) -> None: + platform = _platform(conn) + ui = FakeUi() + options = TroubleshootingOptions( + scenario_id="port_conflict", + hint_level=2, + remediation_action="port=8081", + remediation_params={"port": 8081}, + simulate=True, + ) + evidence = run_troubleshooting_flow(platform, ui, options) + + assert evidence.resolved is True + assert evidence.verification is not None + assert evidence.verification.success is True + assert evidence.mode_label == "SIMULATED / TESTED" + + # Audit events + session = platform.session_service._session_repository.latest() + assert session is not None + session_id = session.id + assert session_id is not None + events = platform.audit_service.history(session_id) + event_types = {e.event_type for e in events} + assert AuditEventType.TROUBLESHOOTING_STARTED in event_types + assert AuditEventType.TROUBLESHOOTING_REMEDIATION_ATTEMPTED in event_types + assert AuditEventType.TROUBLESHOOTING_VERIFIED in event_types + assert AuditEventType.TROUBLESHOOTING_COMPLETED in event_types + + # Experience tracked + experience = platform.experience_tracker.summary(session_id) + assert "Networking" in experience + assert experience["Networking"][0].state == ExperienceState.DEMONSTRATED + + +def test_troubleshooting_flow_unsuccessful_recovery(conn: sqlite3.Connection) -> None: + platform = _platform(conn) + ui = FakeUi() + options = TroubleshootingOptions( + scenario_id="port_conflict", + remediation_action="port=8000", + remediation_params={"port": 8000}, + simulate=True, + ) + evidence = run_troubleshooting_flow(platform, ui, options) + + assert evidence.resolved is False + assert evidence.verification is not None + assert evidence.verification.success is False + + session = platform.session_service._session_repository.latest() + assert session is not None + session_id = session.id + assert session_id is not None + events = platform.audit_service.history(session_id) + event_types = {e.event_type for e in events} + assert AuditEventType.TROUBLESHOOTING_FAILED in event_types + + +def test_troubleshooting_flow_interactive_hints_and_remediation(conn: sqlite3.Connection) -> None: + platform = _platform(conn) + ui = FakeUi(choices=["Level 3: Root cause explanation", "REQUIRED_CONFIG_KEY=my_key"]) + options = TroubleshootingOptions( + scenario_id="missing_config", + interactive=True, + simulate=True, + ) + evidence = run_troubleshooting_flow(platform, ui, options) + assert evidence.resolved is True + assert any("[HINT LEVEL 3]" in p for p in ui.presented)