commit b07562921b3cd2147af00f376ba60d2ef13f886b Author: hexdev Date: Sat Jul 11 23:54:36 2026 +0700 first commit diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..8722fed --- /dev/null +++ b/.gitignore @@ -0,0 +1,177 @@ +# ---> Python +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ +cover/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +.pybuilder/ +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +# For a library or package, you might want to ignore these files since the code is +# intended to run in multiple environments; otherwise, check them in: +# .python-version + +# pipenv +# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. +# However, in case of collaboration, if having platform-specific dependencies or dependencies +# having no cross-platform support, pipenv may install dependencies that don't work, or not +# install all needed dependencies. +#Pipfile.lock + +# UV +# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control. +# This is especially recommended for binary packages to ensure reproducibility, and is more +# commonly ignored for libraries. +#uv.lock + +# poetry +# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control. +# This is especially recommended for binary packages to ensure reproducibility, and is more +# commonly ignored for libraries. +# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control +#poetry.lock + +# pdm +# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control. +#pdm.lock +# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it +# in version control. +# https://pdm.fming.dev/latest/usage/project/#working-with-version-control +.pdm.toml +.pdm-python +.pdm-build/ + +# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +*.env +dev.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# pytype static type analyzer +.pytype/ + +# Cython debug symbols +cython_debug/ + +# PyCharm +# JetBrains specific template is maintained in a separate JetBrains.gitignore that can +# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore +# and can be added to the global gitignore or merged into this file. For a more nuclear +# option (not recommended) you can uncomment the following to ignore the entire idea folder. +#.idea/ + +# Ruff stuff: +.ruff_cache/ + +# PyPI configuration file +.pypirc \ No newline at end of file diff --git a/HYDROGEN_CONFIGURATION.md b/HYDROGEN_CONFIGURATION.md new file mode 100644 index 0000000..7dc925a --- /dev/null +++ b/HYDROGEN_CONFIGURATION.md @@ -0,0 +1,606 @@ +# Hydrogen Configuration Guide + +This document explains how configuration works in Hydrogen, which settings are actually used at runtime, and how configuration connects to module loading, plugin discovery, and report delivery. + +## Configuration Entry Points + +Hydrogen starts in `main.py`. + +- `--config` (`-c`) defaults to `config.yaml` in the current working directory. +- `--path` (`-p`) defaults to the `modules/` directory in the current working directory. +- `--package-prefix` defaults to `modules`. + +The configuration file is loaded by `config.py`: + +```python +def load_config(path: Path) -> config.Config: + with path.open(encoding="utf-8") as f: + content = yaml.safe_load(f) + + return config.Config.model_validate(content) +``` + +Hydrogen validates the YAML against the Pydantic model `core.schemas.config.Config` before the audit starts. + +## Full Configuration Shape + +Hydrogen currently expects this top-level structure: + +```yaml +exclude_categories: [] +fail_fast: false +dry_run: false +max_concurrency: 4 + +logging: + level: INFO + output: stdout + +plugin_packages: + renderers: + - reporting.exporters + transports: + - reporting.transports + modules: [] + +strict_mode: true +allow_failures_below: medium + +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/latest + append_extension: true + +modules: + ssh: + enabled: true +``` + +The schema is defined in `core/schemas/config.py`. + +## Top-Level Fields + +### `exclude_categories` + +Type: `list[str]` + +Default: `[]` + +This field is actively used in `core/module_loader.py`. + +Hydrogen compares each module manifest's `category` value against this list: + +```python +if module.manifest.category in config.exclude_categories: + logger.info("module %s was skipped due to config exclusion", module.manifest.identifier) + continue +``` + +Important details: + +- Exclusion is based on `manifest.category`, not on the directory name. +- Exclusion happens before `build_worker()` is called. +- If multiple modules share the same category, one category entry disables all of them. + +Example: + +```yaml +exclude_categories: + - ssh + - tls +``` + +### `strict_mode` + +Type: `bool` + +This field is used in `core/evaluation.py` to determine the final process exit code. + +Behavior: + +- If a finding reaches the configured severity threshold and `strict_mode` is `true`, Hydrogen returns exit code `1`. +- Otherwise, Hydrogen returns exit code `0`. + +Relevant code: + +```python +if threshold_crossed and config.strict_mode: + return 1 +return 0 +``` + +This is the main switch controlling whether Hydrogen behaves as a CI gate. + +### `allow_failures_below` + +Type: `AuditSeverity` + +Allowed values: + +- `low` +- `medium` +- `high` +- `critical` + +This field is also used in `core/evaluation.py`. + +Hydrogen maps severities to numeric weights: + +- `low` -> `1` +- `medium` -> `2` +- `high` -> `3` +- `critical` -> `4` + +Then it fails the run when any finding has a weight greater than or equal to the configured threshold. + +How to read the field name correctly: + +- `allow_failures_below: medium` means Hydrogen tolerates only findings below `medium`. +- In practice, `medium`, `high`, and `critical` findings can trigger exit code `1` when `strict_mode: true`. +- Only `low` findings are below that threshold. + +Examples: + +```yaml +allow_failures_below: high +``` + +Meaning: + +- `low` and `medium` findings are tolerated. +- `high` and `critical` findings can fail the run when `strict_mode` is enabled. + +```yaml +allow_failures_below: critical +``` + +Meaning: + +- Only `critical` findings can fail the run. + +### `fail_fast` + +Type: `bool` + +Default: `false` + +When `true`, Hydrogen stops submitting new worker tasks as soon as any module returns a `fail` status. Already-running workers are allowed to finish. + +This is used in `core/runner.py`: + +```python +if config.fail_fast and result.result.status is AuditStatus.FAIL: + stop_submitting = True +``` + +### `dry_run` + +Type: `bool` + +Default: `false` + +When `true`, Hydrogen runs all security modules and evaluates exit codes but skips report publishing. Useful for testing module behavior without side effects. + +```python +if resolved_config.dry_run: + logger.info("dry_run enabled, skipping report publish") +``` + +### `max_concurrency` + +Type: `int` + +Default: `4` + +Controls the maximum number of worker threads in the `ThreadPoolExecutor`. Hydrogen runs module workers concurrently up to this limit. + +### `logging` + +Type: object + +Controls logging behavior. + +```yaml +logging: + level: INFO + output: stdout +``` + +Fields: + +- `level`: any standard Python logging level name (`DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`). +- `output`: `stdout` or `stderr`. + +### `plugin_packages` + +Type: `PluginPackagesConfig` + +This section controls where Hydrogen discovers renderers, transports, and modules. + +Schema: + +```yaml +plugin_packages: + renderers: + - reporting.exporters + transports: + - reporting.transports + modules: + - mycompany.hydrogen_modules +``` + +Fields: + +- `renderers`: list of Python package names to scan for `BaseRenderer` subclasses. +- `transports`: list of Python package names to scan for `BaseTransport` subclasses. +- `modules`: list of Python package names to scan for modules (in addition to filesystem discovery). + +Hydrogen uses `reporting/bootstrap.py` and `core/plugin_types.py` to discover plugins: + +```python +def discover_plugins(group_config: PluginGroupConfig, base_class: type[TPlugin]): + for package_name in group_config.package_names: + for plugin_class in _discover_plugin_classes(package_name, base_class): + descriptors.append(PluginDescriptor(plugin=plugin_class(), source=...)) +``` + +Additionally, Hydrogen discovers plugins via Python `entry_points` under the following groups: + +- `hydrogen.renderers` for renderers. +- `hydrogen.transports` for transports. +- `hydrogen.modules` for modules. + +This allows third-party packages installed via pip to register plugins without manual configuration. + +### `reports` + +Type: `ReportsConfig` + +This section controls report rendering and delivery. Unlike the old single-output schema, Hydrogen now supports **multiple report outputs**. + +Schema: + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/audit + append_extension: true + - renderer: + type: json + transport: + type: webhook + url: https://collector.example.com/hydrogen + method: POST + payload_mode: envelope +``` + +Each output contains: + +- `renderer`: a `PluginConfigRef` with a `type` field and optional extra fields. +- `transport`: a `PluginConfigRef` with a `type` field and optional extra fields. + +Hydrogen uses it in `reporting/service.py`: + +```python +def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]: + return [self.publish(report, output) for output in outputs] +``` + +The `PluginConfigRef` model: + +```python +class PluginConfigRef(BaseModel): + model_config = ConfigDict(extra="allow") + type: str = Field(min_length=1) + + def payload(self) -> dict[str, Any]: + return dict(self.model_extra or {}) +``` + +This means: + +- The `type` field selects which renderer/transport is used. +- All other fields (the "payload") are passed to the plugin's `config_model` for validation. +- Hydrogen validates the payload using each plugin's own `config_model` Pydantic model. + +### `modules` + +Type: `dict[str, dict[str, Any]]` + +Default: `{}` + +This section stores per-module configuration. + +Hydrogen looks up module settings by manifest identifier and validates them against the module's `CONFIG_MODEL`: + +```python +module_config = config.modules.get(module.manifest.identifier, {}) +validated_config = module.config_model.model_validate(module_config) +worker = build_worker(module, validated_config) +``` + +Important implications: + +- The key must match `MANIFEST.identifier`, not the human-readable module name. +- If the key is missing, your module receives an empty dict, which is then validated by the module's `config_model`. +- Module configuration is validated at runtime by the module's own Pydantic config model. +- If validation fails, Hydrogen reports a `PluginRuntimeError` and continues with other modules. + +Example: + +```yaml +modules: + ssh: + enabled: true + test-failure: false +``` + +## Report Output Configuration + +Each output in `reports.outputs` defines one renderer + transport pair. Hydrogen sends the rendered report through each transport. + +### Renderer Config Ref + +```yaml +outputs: + - renderer: + type: json +``` + +Fields: + +- `type`: must match a registered renderer's `content_type` or one of its `aliases`. + +Extra fields are passed to the renderer's `config_model` for validation. + +### Transport Config Ref + +```yaml +outputs: + - transport: + type: file + path: reports/security-report + append_extension: true +``` + +Fields: + +- `type`: must match a registered transport's `transport_type`. +- Additional fields are plugin-specific (e.g., `path`, `url`, `method`, `headers`). + +Extra fields are validated against the transport's `config_model`. + +### Example: File Transport + +```yaml +outputs: + - renderer: + type: json + transport: + type: file + path: reports/security-report + append_extension: true +``` + +Fields: + +- `type`: must be `file` +- `path`: output file path +- `append_extension`: when `true`, Hydrogen appends the renderer's extension if the path has no suffix + +Example result with the JSON renderer: + +- `reports/security-report` becomes `reports/security-report.json` + +### Example: Webhook Transport + +```yaml +outputs: + - renderer: + type: json + transport: + type: webhook + url: https://example.internal/security-ingest + method: POST + headers: + Authorization: Bearer secret-token + timeout_seconds: 10 + payload_mode: envelope +``` + +Fields: + +- `type`: must be `webhook` +- `url`: validated as an HTTP or HTTPS URL by Pydantic +- `method`: `POST`, `PUT`, or `PATCH` +- `headers`: arbitrary string-to-string HTTP headers +- `timeout_seconds`: positive float +- `payload_mode`: `rendered` or `envelope` + +Payload behavior: + +- `rendered`: Hydrogen sends the renderer output directly with the renderer's `media_type` as `Content-Type`. +- `envelope`: Hydrogen sends a JSON object containing format metadata and the rendered content with `Content-Type: application/json`. + +## Plugin Discovery via Entry Points + +Hydrogen supports discovering plugins via Python package entry points. This is the recommended way to distribute third-party plugins. + +Entry point groups: + +- `hydrogen.renderers` — for custom renderers +- `hydrogen.transports` — for custom transports +- `hydrogen.modules` — for custom security modules + +Example `pyproject.toml` for a third-party renderer: + +```toml +[project.entry-points."hydrogen.renderers"] +markdown = "my_package.renderers:MarkdownRenderer" +``` + +Entry points can point to either a class (which Hydrogen instantiates) or a pre-built instance. Hydrogen validates that the result is a non-abstract subclass or instance of the expected base class. + +## Example Production-Oriented Configurations + +### Save a JSON Report to Disk + +```yaml +fail_fast: false +dry_run: false + +exclude_categories: [] + +logging: + level: INFO + output: stdout + +strict_mode: true +allow_failures_below: high + +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/hydrogen-audit + append_extension: true + +modules: + ssh: + test-failure: false +``` + +### Send the Report to Multiple Destinations + +```yaml +fail_fast: false +dry_run: false + +exclude_categories: + - experimental + +logging: + level: DEBUG + output: stdout + +strict_mode: true +allow_failures_below: medium + +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/hydrogen-audit + append_extension: true + - renderer: + type: json + transport: + type: webhook + url: https://collector.example.com/hydrogen/report + method: POST + headers: + X-Source: hydrogen + timeout_seconds: 5 + payload_mode: envelope + +modules: + ssh: + test-failure: false +``` + +## Validation and Failure Modes + +Because Hydrogen validates configuration with Pydantic before execution, the following failures happen early: + +- invalid enum values such as `allow_failures_below: severe` +- invalid logging output values +- missing required fields such as `reports.outputs` + +Other failures happen later at runtime: + +- unknown renderer names in `renderer.type` raise an error when `RendererRegistry` cannot find that name. +- unknown transport type names raise an error when `TransportRegistry` cannot find that name. +- plugin config validation errors (when a plugin's `config_model` rejects the provided payload) raise errors during config resolution in `core/runtime.py`. +- module import failures are not swallowed and can stop execution. + +## Multi-Output Architecture + +Hydrogen supports multiple report outputs per run. Each output defines an independent renderer + transport pair. + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/audit.json + - renderer: + type: markdown + transport: + type: webhook + url: https://wiki.internal/ingest + method: POST +``` + +Hydrogen iterates through all outputs and publishes each one: + +```python +def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]: + return [self.publish(report, output) for output in outputs] +``` + +If an output fails, Hydrogen collects the error in `plugin_errors` and continues with the next output. + +## Relationship Between `--path` and `--package-prefix` + +This is not part of YAML, but it matters when you organize Hydrogen modules. + +Hydrogen discovers modules from the filesystem path passed to `--path`, but imports them by Python package name using `--package-prefix`. + +Example: + +```bash +python main.py --path custom_modules --package-prefix custom_modules +``` + +For this to work: + +- `custom_modules/` must exist. +- Each module must be a package directory with its own `__init__.py`. +- Python must be able to import `custom_modules.`. + +Additionally, modules can come from `plugin_packages.modules` in YAML and from `hydrogen.modules` entry points. All sources are combined in `discover_modules()` with deduplication. + +## Internal Runtime Config + +After loading and validation, Hydrogen builds a `ResolvedConfig` dataclass: + +```python +@dataclass(frozen=True) +class ResolvedConfig: + exclude_categories: list[str] + allow_failures_below: AuditSeverity + strict_mode: bool + fail_fast: bool + dry_run: bool + max_concurrency: int + logging: LoggingConfig + plugin_packages: PluginPackagesConfig + reports: list[ResolvedReportOutput] + modules: dict[str, BaseModel] +``` + +This is the resolved internal config that modules and reporting use at runtime. Each module's config is already validated into its typed `config_model`. diff --git a/HYDROGEN_RENDERERS.md b/HYDROGEN_RENDERERS.md new file mode 100644 index 0000000..2c9695f --- /dev/null +++ b/HYDROGEN_RENDERERS.md @@ -0,0 +1,427 @@ +# Hydrogen Report Rendering and Custom Renderers + +This document explains how report rendering works in Hydrogen, how the built-in JSON renderer behaves, and how to add your own renderer. + +Renderers transform an `AuditReport` into a transport-ready `RenderedReport`. They are the formatting layer between raw audit data and delivery. + +## Where Rendering Happens in Hydrogen + +Hydrogen builds reporting output through `ReportService` in `reporting/service.py`. + +The render path is: + +```python +def render(self, report: AuditReport, report_config: ResolvedReportOutput) -> RenderedReport: + renderer = self._renderer_registry.get(report_config.renderer.type) + return renderer.render(report, report_config.renderer.config) +``` + +This means: + +- `reports.outputs[].renderer.type` selects the renderer by its `content_type` or alias. +- The renderer receives the full `AuditReport` and a validated config `BaseModel`. +- The renderer returns a `RenderedReport`. +- The chosen transport publishes that rendered output. + +## Core Renderer Contract + +The base class lives in `core/base.py`. + +```python +class BaseRenderer(ABC): + content_type: str + media_type: str + file_extension: str + aliases: tuple[str, ...] = () + config_model: type[BaseModel] = EmptyPluginConfig + + @abstractmethod + def render(self, report: AuditReport, config: BaseModel) -> RenderedReport: ... +``` + +Every Hydrogen renderer must define: + +- `content_type` — primary configuration key used in `config.yaml`. +- `media_type` — MIME type used by transports (e.g., `application/json`). +- `file_extension` — default extension for file-based delivery (e.g., `.json`). +- `aliases` — optional alternative lookup names (e.g., `("application/json",)`). +- `config_model` — optional Pydantic model for renderer-specific configuration. Defaults to `EmptyPluginConfig`. +- `render(report, config)` — the rendering implementation. + +## The `RenderedReport` Shape + +Renderers must return `reporting.models.RenderedReport`: + +```python +class RenderedReport(BaseModel): + format_name: str + media_type: str + file_extension: str + content: str +``` + +Field meanings: + +- `format_name` — logical renderer name such as `json` or `markdown`. +- `media_type` — content type such as `application/json`. +- `file_extension` — extension such as `.json`. +- `content` — the final text payload. + +This object is the handoff contract between Hydrogen renderers and Hydrogen transports. + +## Renderer Configuration + +Renderers can accept configuration via `config_model`. Options are passed as extra fields alongside `type` in the YAML configuration. + +```yaml +reports: + outputs: + - renderer: + type: my_renderer + pretty: true + include_passed: false +``` + +Hydrogen resolves this in `core/runtime.py`: + +```python +def _resolve_plugin_config(plugin_type, payload, plugin): + config_model = ensure_plugin_config_model(plugin) + return ResolvedPluginConfig( + type=plugin_type, + config=config_model.model_validate(payload) + ) +``` + +If your renderer uses `EmptyPluginConfig` (the default), any extra fields in YAML will cause a validation error because `EmptyPluginConfig` forbids extras. To accept options, define a custom config model. + +## Built-In Renderer: JSON + +Implementation: `reporting/exporters/json.py` + +```python +class JsonRenderer(BaseRenderer): + content_type = "json" + media_type = "application/json" + file_extension = ".json" + aliases = ("application/json",) + config_model = EmptyPluginConfig + + def render(self, report: AuditReport, config: EmptyPluginConfig) -> RenderedReport: + return RenderedReport( + format_name=self.content_type, + media_type=self.media_type, + file_extension=self.file_extension, + content=json.dumps(report.model_dump(mode="json"), ensure_ascii=True, indent=2), + ) +``` + +Behavior: + +- Serializes `AuditReport` with `report.model_dump(mode="json")`. +- Output is formatted with `json.dumps(..., ensure_ascii=True, indent=2)`. +- The result is UTF-8 safe ASCII JSON text. + +Two renderer lookup keys work out of the box: + +- `json` +- `application/json` + +This is because `RendererRegistry.register()` stores both the primary `content_type` and every alias. + +## How Hydrogen Discovers Renderers + +Renderer discovery is implemented in `reporting/bootstrap.py` and `core/plugin_types.py`. + +Hydrogen discovers renderers from two sources: + +1. **Package scan** — Python packages listed in `plugin_packages.renderers` (default: `["reporting.exporters"]`). +2. **Entry points** — `hydrogen.renderers` entry points registered by installed packages. + +```python +def bootstrap_reporting(plugin_packages: PluginPackagesConfig) -> ReportingBootstrapResult: + renderer_descriptors, renderer_errors = discover_plugins( + PluginGroupConfig( + package_names=plugin_packages.renderers, + entry_point_group="hydrogen.renderers", + plugin_kind="renderer", + ), + BaseRenderer, + ) + ... +``` + +The discovery process: + +- Imports the package and all its submodules via `pkgutil.walk_packages`. +- Inspects each module for non-abstract subclasses of `BaseRenderer`. +- Filters out classes that are only imported (not defined) in that module. +- Instantiates each discovered class with no constructor arguments. +- Registers them in `RendererRegistry`. + +Important implications for custom renderers: + +- Your renderer class must be defined in a module under one of the configured packages. +- It must be a concrete subclass of `BaseRenderer`. +- It must have a zero-argument constructor. +- Discovery ignores classes that are only re-exported from another module. +- Names and aliases are registered in a single dictionary — collisions raise `ValueError`. + +## How Renderer Selection Works + +Hydrogen selects the renderer by `reports.outputs[].renderer.type` from `config.yaml`. + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/latest +``` + +The lookup: + +```python +renderer = self._renderer_registry.get(report_config.renderer.type) +``` + +If there is no renderer for that key, `RendererRegistry.get()` raises: + +```python +ValueError("Renderer for content type '' is not registered") +``` + +This is a runtime failure, not a configuration-schema failure. `renderer.type` is a plain string in `PluginConfigRef`. + +## Adding a Custom Renderer + +To add a renderer in Hydrogen, create a module under one of the configured plugin packages and implement a `BaseRenderer` subclass. + +### Via Package (In-Repository) + +Create a file under `reporting/exporters/`: + +File: `reporting/exporters/markdown.py` + +```python +from pydantic import BaseModel, Field + +from core.base import BaseRenderer +from reporting.models import AuditReport, RenderedReport + + +class MarkdownRendererConfig(BaseModel): + include_passed: bool = Field(True, description="Include modules with no findings") + + +class MarkdownRenderer(BaseRenderer): + content_type = "markdown" + media_type = "text/markdown" + file_extension = ".md" + aliases = ("md", "text/markdown") + config_model = MarkdownRendererConfig + + def render(self, report: AuditReport, config: MarkdownRendererConfig) -> RenderedReport: + lines = [ + "# Hydrogen Security Report", + "", + f"Generated at: {report.generated_at.isoformat()}", + f"Exit code: {report.exit_code}", + "", + ] + + for module_result in report.results: + if not config.include_passed and not module_result.result.findings: + continue + + lines.append(f"## {module_result.module.name}") + lines.append(f"Status: {module_result.result.status}") + lines.append(f"Risk level: {module_result.result.risk_level}") + lines.append("") + + if not module_result.result.findings: + lines.append("No findings.") + lines.append("") + continue + + for finding in module_result.result.findings: + lines.append(f"- [{finding.severity}] {finding.name}: {finding.description}") + lines.append("") + + return RenderedReport( + format_name=self.content_type, + media_type=self.media_type, + file_extension=self.file_extension, + content="\n".join(lines), + ) +``` + +Then use it in configuration: + +```yaml +reports: + outputs: + - renderer: + type: markdown + include_passed: false + transport: + type: file + path: reports/security + append_extension: true +``` + +With the built-in file transport, that produces `reports/security.md`. + +### Via Entry Points (Third-Party Package) + +If you distribute your renderer as a pip-installable package, register it via entry points: + +```toml +[project.entry-points."hydrogen.renderers"] +markdown = "my_package.renderers:MarkdownRenderer" +``` + +The entry point value can be: + +- A class reference (Hydrogen instantiates it). +- A pre-built instance. + +Hydrogen validates that the result is a non-abstract subclass or instance of `BaseRenderer`. + +### Via `plugin_packages` Configuration + +You can add additional packages for renderer discovery: + +```yaml +plugin_packages: + renderers: + - reporting.exporters + - mycompany.reporting_extras +``` + +Hydrogen scans all listed packages and their submodules for `BaseRenderer` subclasses. + +## Renderer Aliases + +Aliases are optional, but useful. + +```python +aliases = ("md", "text/markdown") +``` + +This allows the following Hydrogen config values to point to the same renderer: + +- `markdown` +- `md` +- `text/markdown` + +Registry behavior: + +```python +self._renderers[renderer.content_type] = renderer +for alias in renderer.aliases: + self._renderers[alias] = renderer +``` + +## Renderer Name Collisions + +Hydrogen stores renderers in a dictionary. If two renderers register the same `content_type` or alias, `register()` raises a `ValueError` with a descriptive message. + +In `bootstrap.py`, this error is caught and converted to a `PluginLoadError`: + +```python +try: + renderer_registry.register(descriptor.plugin) +except ValueError as exc: + errors.append(PluginLoadError(plugin_kind="renderer", source=descriptor.source, message=str(exc))) +``` + +Choose unique names and aliases for custom renderers. + +## Configurable Renderer Behavior + +Unlike the old architecture, Hydrogen now supports renderer-specific configuration via `config_model`. + +Define a Pydantic model for your options: + +```python +class MyRendererConfig(BaseModel): + pretty: bool = Field(True) + max_findings: int = Field(50, ge=1) +``` + +Set it on the renderer class: + +```python +class MyRenderer(BaseRenderer): + content_type = "my_format" + config_model = MyRendererConfig + + def render(self, report: AuditReport, config: MyRendererConfig) -> RenderedReport: + if config.pretty: + ... +``` + +Configure it in YAML: + +```yaml +outputs: + - renderer: + type: my_format + pretty: false + max_findings: 100 +``` + +The extra fields (`pretty`, `max_findings`) are extracted from `PluginConfigRef.payload()` and validated against `MyRendererConfig`. + +## The `__init__.py` Export + +Renderers do not need to be re-exported from the package's `__init__.py` for discovery to work. Hydrogen discovers submodules directly with `pkgutil.walk_packages`. + +An explicit export like this is optional: + +```python +from reporting.exporters.markdown import MarkdownRenderer +``` + +It can still be useful for developer ergonomics and explicit imports. + +## Troubleshooting Custom Renderers + +If Hydrogen does not pick up your renderer, check the following: + +1. The file is inside one of the configured plugin packages. +2. The class subclasses `BaseRenderer`. +3. The class is not abstract. +4. The class is defined in that module, not only imported into it. +5. The class can be instantiated with no arguments. +6. `config_model` is a valid Pydantic `BaseModel` subclass (if defined). +7. `reports.outputs[].renderer.type` matches either `content_type` or one of the aliases. +8. The renderer returns a proper `RenderedReport`. +9. The `render()` method accepts two arguments: `report` and `config`. + +## Design Guidance for Hydrogen Renderers + +Good custom renderers usually follow these rules: + +1. Keep rendering pure and deterministic. +2. Treat the renderer as a formatting layer, not a transport. +3. Use `media_type` and `file_extension` consistently so transports behave correctly. +4. Keep `content_type` short and stable because it becomes part of configuration. +5. Return text content only, because `RenderedReport.content` is currently a string. +6. If you need configurable behavior, define a `config_model` and use the `config` parameter. +7. Do not rely on global state — all required inputs come through `render(report, config)`. + +## Summary + +Hydrogen renderers are straightforward extension points: + +- They live under configurable plugin packages (default: `reporting/exporters/`). +- They convert `AuditReport` into `RenderedReport`. +- They are selected by `reports.outputs[].renderer.type`. +- They receive a typed config object validated by their `config_model`. +- They are auto-discovered from packages and entry points. + +The main difference from the old architecture is that renderers now receive a config parameter and support per-renderer options through `config_model`. diff --git a/HYDROGEN_SECURITY_MODULES.md b/HYDROGEN_SECURITY_MODULES.md new file mode 100644 index 0000000..b077b7b --- /dev/null +++ b/HYDROGEN_SECURITY_MODULES.md @@ -0,0 +1,520 @@ +# Hydrogen Custom Security Modules + +This document explains how to add your own security checking modules to Hydrogen. It is based on the actual module loading flow implemented in `core/module_loader.py`, `core/base.py`, `core/schemas/modules.py`, and the example SSH module under `modules/ssh`. + +## How Hydrogen Loads Modules + +Hydrogen starts module execution in `core/runner.py`: + +```python +loaded_modules, module_discovery_errors = load_module_descriptors(modules_path, config, package_prefix) +``` + +The module loader then discovers modules from three sources (combined and deduplicated): + +1. **Filesystem directory** — scanned via `--path` (default: `modules/`). +2. **`plugin_packages.modules`** — Python packages listed in `config.yaml`. +3. **`hydrogen.modules` entry points** — registered by installed Python packages. + +```python +def discover_modules(path: Path, config: Config, package_prefix: str = "modules") -> list[str]: + # 1. Scan filesystem + for entry in path.iterdir(): + if entry.is_dir() and not entry.name.startswith("__") and (entry / "__init__.py").exists(): + package_names.append(f"{package_prefix}.{entry.name}") + + # 2. From config + package_names.extend(config.plugin_packages.modules) + + # 3. From entry points + for ep in entry_points(group="hydrogen.modules"): + package_names.append(ep.value.partition(":")[0]) + + return list(dict.fromkeys(package_names)) # deduplicate +``` + +Filesystem rules: + +- Hydrogen only inspects direct children of the modules directory. +- Entries must be directories. +- Directory names starting with `__` are skipped. +- The directory must contain `__init__.py`. +- The package is imported as `.`. + +Hydrogen does **not** recursively discover module packages inside nested directories. A module must be a first-level package under the selected modules root. + +## Required Module Contract + +Each Hydrogen module package must provide three things: + +- `MANIFEST` — a `ModuleManifest` instance. +- `build_worker(config)` — a callable that returns a `BaseWorker`. +- `CONFIG_MODEL` (optional, recommended) — a Pydantic `BaseModel` subclass for config validation. + +If either `MANIFEST` or `build_worker` is missing, the loader logs a warning and skips the module. + +### `MANIFEST` + +`MANIFEST` is validated against `core.schemas.manifest.ModuleManifest`. + +```python +class ModuleManifest(BaseModel): + model_config = ConfigDict(extra="forbid", frozen=True) + + identifier: str = Field(pattern=r"^[a-z][a-z0-9_-]*$") + name: str = Field(min_length=1) + category: str = Field(min_length=1) + version: str = Field(pattern=r"^\d+\.\d+\.\d+$") + api_version: str = Field("1", pattern=r"^\d+$") + description: str = "" +``` + +Required fields: + +- `identifier` — must match `^[a-z][a-z0-9_-]*$` +- `name` — human-readable name, non-empty +- `category` — group name used for exclusion, non-empty +- `version` — semantic version `X.Y.Z` +- `api_version` — must match Hydrogen's `HYDROGEN_API_VERSION` (currently `"1"`) + +`description` is optional and defaults to an empty string. + +Example from the built-in SSH module: + +```python +MANIFEST = ModuleManifest( + identifier="ssh", + name="SSH Security Audit", + category="ssh", + version="0.1.0", + api_version="1", + description="Audits OpenSSH server configuration.", +) +``` + +**Important:** `api_version` must match Hydrogen's API version. If it does not match, the module is rejected with a `ValueError`. + +### `CONFIG_MODEL` + +Modules can export a `CONFIG_MODEL` — a Pydantic `BaseModel` subclass that defines the expected shape of the module's configuration section from `config.yaml`. + +```python +# modules/my_check/config.py +from pydantic import BaseModel, Field + +class MyCheckConfig(BaseModel): + enabled: bool = Field(True) + threshold: int = Field(5, ge=1) +``` + +```python +# modules/my_check/__init__.py +from .config import MyCheckConfig + +CONFIG_MODEL = MyCheckConfig +``` + +When `CONFIG_MODEL` is defined, Hydrogen validates the module's config section from YAML against this model before passing it to `build_worker()`: + +```python +module_config = config.modules.get(module.manifest.identifier, {}) +validated_config = module.config_model.model_validate(module_config) +worker = build_worker(module, validated_config) +``` + +If `CONFIG_MODEL` is not defined, Hydrogen passes an `EmptyPluginConfig` instance (an empty Pydantic model). + +### `build_worker(config)` + +The loader expects a callable named `build_worker`. + +Signature: + +```python +def build_worker(config: MyCheckConfig) -> BaseWorker: + ... +``` + +Rules: + +- It must be callable. +- It receives the **validated** module-specific config (a `BaseModel` instance from `CONFIG_MODEL`). +- It must return an instance of `BaseWorker`. + +If it returns anything else, Hydrogen logs a warning and skips the module. + +## Worker Contract + +All custom workers must inherit `core.base.BaseWorker`. + +```python +class BaseWorker(ABC): + config_model: type[BaseModel] = EmptyPluginConfig + + def __init__(self, config: BaseModel | dict[str, Any] | None = None) -> None: + self.raw_config = config + self.config: BaseModel | dict[str, Any] + if isinstance(config, BaseModel): + self.config = config.model_dump(mode="python") + else: + self.config = config or {} + + @abstractmethod + def run(self) -> AuditResults: + ... +``` + +Key points: + +- `__init__` accepts either a `BaseModel` (preferred) or a raw `dict`. +- The config is stored as both `self.raw_config` (original form) and `self.config` (dict form for easy access). +- `config_model` is a class attribute that can be set on the worker for reference. +- `run()` must return `AuditResults`. + +The built-in SSH module shows the recommended pattern: + +```python +class SSHSecurityWorker(BaseWorker): + config_model = SSHModuleConfig + + def __init__(self, config: SSHModuleConfig | None = None) -> None: + super().__init__(config) + self.module_config = config or SSHModuleConfig() + + def run(self) -> AuditResults: + if not self.module_config.enabled: + return AuditResults(status=AuditStatus.SKIPPED, findings=[], risk_level=0.0) + ... +``` + +## Required Result Shape + +Your worker's `run()` method must return `core.schemas.results.AuditResults`. + +```python +class AuditResults(BaseModel): + status: AuditStatus + findings: list[AuditFindings] + risk_level: float +``` + +### `status` + +Allowed values from `AuditStatus`: + +- `pass` — the audit ran and found no issues. +- `fail` — the audit ran and produced one or more findings. +- `skipped` — the audit was intentionally not applicable on the current system. + +### `findings` + +Each finding is an `AuditFindings` object: + +```python +class AuditFindings(BaseModel): + name: str + description: str = "" + severity: AuditSeverity +``` + +Severity values: + +- `low` +- `medium` +- `high` +- `critical` + +These severities feed directly into Hydrogen's exit code calculation. + +### `risk_level` + +`risk_level` is a float from `0.0` to `1.0` indicating how dangerous the results are. + +Practical guidance: + +- Return `0.0` for clean results. +- Return a bounded value up to `1.0`. +- Keep the calculation deterministic so reports remain comparable over time. + +## Full Module Layout + +Here is the full structure Hydrogen expects: + +```text +modules/ + my_check/ + __init__.py + config.py # optional, recommended + worker.py +``` + +### `modules/my_check/config.py` + +```python +from pydantic import BaseModel, Field + + +class MyCheckConfig(BaseModel): + enabled: bool = Field(True) + must_be_enabled: bool = Field(False) +``` + +### `modules/my_check/__init__.py` + +```python +from core.base import BaseWorker +from core.schemas import ModuleManifest + +from .config import MyCheckConfig +from .worker import MyCheckWorker + +MANIFEST = ModuleManifest( + identifier="my_check", + name="My Custom Check", + category="custom", + version="1.0.0", + api_version="1", + description="Checks a custom hardening rule.", +) + +CONFIG_MODEL = MyCheckConfig + + +def build_worker(config: MyCheckConfig) -> BaseWorker: + return MyCheckWorker(config) +``` + +### `modules/my_check/worker.py` + +```python +from core.base import BaseWorker +from core.schemas.results import AuditFindings, AuditResults +from core.schemas.status import AuditSeverity, AuditStatus + +from .config import MyCheckConfig + + +class MyCheckWorker(BaseWorker): + config_model = MyCheckConfig + + def __init__(self, config: MyCheckConfig | None = None) -> None: + super().__init__(config) + self.module_config = config or MyCheckConfig() + + def run(self) -> AuditResults: + findings: list[AuditFindings] = [] + + if self.module_config.must_be_enabled is not True: + findings.append( + AuditFindings( + name="custom_rule_disabled", + description="The custom rule is not enabled.", + severity=AuditSeverity.HIGH, + ) + ) + + return AuditResults( + status=AuditStatus.PASS if not findings else AuditStatus.FAIL, + findings=findings, + risk_level=0.0 if not findings else 0.3, + ) +``` + +## Wiring Module Configuration + +Hydrogen validates and passes module config by manifest identifier. + +Given this manifest: + +```python +MANIFEST = ModuleManifest(identifier="my_check", ...) +``` + +The YAML section must be: + +```yaml +modules: + my_check: + must_be_enabled: true +``` + +Hydrogen does this: + +```python +module_config = config.modules.get(module.manifest.identifier, {}) +validated_config = module.config_model.model_validate(module_config) +worker = build_worker(module, validated_config) +``` + +Important details: + +- The key is `my_check`, not the directory label shown to users in a report title. +- If the key is missing, your module receives an empty dict, which is validated against your `CONFIG_MODEL` (using defaults). +- Config validation is strict: if you define `CONFIG_MODEL`, extra fields in YAML that are not in the model will raise a validation error (unless `extra="allow"` is set). +- If config validation fails, Hydrogen reports a `PluginRuntimeError` and continues with other modules — your worker is not created. + +## Category-Based Exclusion + +Hydrogen can disable modules by manifest category. + +Example: + +```yaml +exclude_categories: + - custom +``` + +If your module's manifest has `category="custom"`, Hydrogen skips it before worker creation. + +This is useful for: + +- Experimental checks. +- Platform-specific checks. +- Expensive checks that should be disabled in some environments. + +## Module Discovery via Entry Points + +Third-party modules can register themselves via Python entry points in the `hydrogen.modules` group. + +Example `pyproject.toml`: + +```toml +[project.entry-points."hydrogen.modules"] +my_check = "mycompany.hydrogen_modules.my_check" +``` + +The entry point value must be a Python package path (module or package). Hydrogen imports it and looks for `MANIFEST` and `build_worker` inside. + +## Import Mechanics and Packaging Rules + +Hydrogen imports modules using Python package names, not only filesystem paths. + +For a module directory named `my_check` and default package prefix `modules`, Hydrogen imports: + +```python +import modules.my_check +``` + +Therefore: + +- The module directory must be importable as a Python package. +- Its `__init__.py` is the integration entry point. +- Import-time exceptions are not swallowed — they propagate and fail the module. + +For third-party packages installed via pip, set `plugin_packages.modules` in config: + +```yaml +plugin_packages: + modules: + - mycompany.hydrogen_modules +``` + +Or use entry points as described above. + +## Behavior on Loader Errors + +Hydrogen is tolerant of contract violations, but only up to a point. + +**Handled by warning and skip:** + +- Missing `MANIFEST`. +- Missing `build_worker`. +- Non-callable `build_worker`. +- `build_worker()` returning something that is not a `BaseWorker`. + +**Handled by error reporting (module skipped, run continues):** + +- Import failures while importing the package (caught in `load_modules()`). +- Config validation errors in `resolve_config()`. +- Runtime exceptions inside `worker.run()` (caught in `_run_single_worker()`). + +**Not handled silently (can fail the entire run):** + +- Manifest validation failures (e.g., `api_version` mismatch). +- Missing `CONFIG_MODEL` export with invalid type. + +## Duplicate Module Detection + +Hydrogen detects duplicate module identifiers during config resolution: + +```python +if module.manifest.identifier in seen_module_ids: + runtime_errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=module.manifest.identifier, + stage="discovery", + message="duplicate module identifier", + ) + ) + continue +``` + +If two sources provide the same module identifier, the second one is rejected with an error. + +## Recommended Development Pattern + +For custom Hydrogen modules, the safest pattern is: + +1. Define a `CONFIG_MODEL` Pydantic model for type-safe config validation. +2. Keep `__init__.py` small — just `MANIFEST`, `CONFIG_MODEL`, and `build_worker()`. +3. Put most logic in a separate worker module. +4. Set `config_model` on your worker class for consistency. +5. Use stable, machine-friendly finding names. +6. Return `skipped` when the check is not applicable on the current platform. +7. Use `api_version="1"` (match Hydrogen's `HYDROGEN_API_VERSION`). + +## Example: Running a Custom Module Tree + +If your module packages live outside the default `modules/` directory, you must align the filesystem path and package prefix. + +Example: + +```bash +python main.py --path custom_checks --package-prefix custom_checks +``` + +Then Hydrogen expects packages like: + +```text +custom_checks/ + __init__.py + my_check/ + __init__.py + config.py + worker.py +``` + +Alternatively, add your package to `plugin_packages.modules`: + +```yaml +plugin_packages: + modules: + - custom_checks.my_check +``` + +## Testing Checklist for a New Module + +Before relying on a new Hydrogen module, verify all of the following: + +1. The package imports cleanly. +2. `MANIFEST` validates successfully (especially `api_version`). +3. `CONFIG_MODEL` (if defined) validates correctly. +4. `build_worker()` returns a real `BaseWorker` instance. +5. `run()` always returns `AuditResults`. +6. All findings use valid `AuditSeverity` values. +7. The module behaves correctly when its config section is missing. +8. The module behaves correctly when it is excluded by category. +9. The module handles runtime errors gracefully (Hydrogen wraps exceptions). + +## Summary + +Hydrogen custom security modules are intentionally simple: + +- One package per module. +- One manifest with `api_version` matching Hydrogen's. +- Optional `CONFIG_MODEL` for validated config. +- One worker builder receiving a validated config model. +- One worker object that returns structured results. + +The most important implementation details are that module discovery supports three sources (filesystem, config packages, entry points), module config is validated by `CONFIG_MODEL` and keyed by `MANIFEST.identifier`, and import-time failures are reported but do not stop the entire run. diff --git a/HYDROGEN_TRANSPORTS.md b/HYDROGEN_TRANSPORTS.md new file mode 100644 index 0000000..eb0a00e --- /dev/null +++ b/HYDROGEN_TRANSPORTS.md @@ -0,0 +1,403 @@ +# Hydrogen Custom Report Transports + +This document explains how report transports work in Hydrogen, how the built-in transports behave, and what you need to do to add your own transport implementation. + +## Where Transports Fit in the Hydrogen Pipeline + +Hydrogen builds a report after all security modules finish. + +The reporting flow in `reporting/service.py` is: + +1. Build an `AuditReport`. +2. Render it into a `RenderedReport`. +3. Publish the rendered report through a transport. + +The relevant method is: + +```python +def publish(self, report: AuditReport, report_config: ResolvedReportOutput) -> str: + rendered_report = self.render(report, report_config) + transport = self._transport_registry.get(report_config.transport.type) + return transport.publish(rendered_report, report_config.transport) +``` + +This means transports operate on already-rendered content. They do not receive raw findings directly. + +## Core Transport Contract + +The base transport contract lives in `core/base.py`. + +```python +class BaseTransport(ABC): + transport_type: str + config_model: type[BaseModel] = EmptyPluginConfig + + @abstractmethod + def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: ... +``` + +Every transport must define: + +- `transport_type` — the name used in `config.yaml`. +- `config_model` — optional Pydantic model for transport-specific configuration. +- `publish(...)` — the delivery implementation. + +The `transport` parameter is a `ResolvedPluginConfig`: + +```python +@dataclass(frozen=True) +class ResolvedPluginConfig: + type: str + config: BaseModel +``` + +- `transport.type` — the transport type string. +- `transport.config` — the validated config model instance. + +The return value of `publish()` is a string. The built-in transports return values such as the output path or an HTTP status string. Hydrogen currently does not use that return value elsewhere, but your transport still needs to return one. + +## Typed Transport Pattern + +Hydrogen provides `TypedTransport` as a convenient base class for transports with their own config model: + +```python +class TypedTransport(BaseTransport, Generic[TReportTransport], ABC): + config_model: type[TReportTransport] + + def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: + if not isinstance(transport.config, self.config_model): + raise TypeError( + f"Transport '{self.transport_type}' expected {self.config_model.__name__}, " + f"got {type(transport.config).__name__}" + ) + return self.publish_typed(rendered_report, cast(TReportTransport, transport.config)) + + @abstractmethod + def publish_typed(self, rendered_report: RenderedReport, transport: TReportTransport) -> str: ... +``` + +This is the recommended pattern when your transport has its own configuration model. + +Benefits: + +- Runtime type checking for the config object. +- Cleaner transport code — `publish_typed()` receives the typed config directly. +- No manual casting inside `publish()`. + +## How Hydrogen Discovers Transports + +Transport discovery is implemented in `reporting/bootstrap.py` and `core/plugin_types.py`. + +Hydrogen discovers transports from two sources: + +1. **Package scan** — Python packages listed in `plugin_packages.transports` (default: `["reporting.transports"]`). +2. **Entry points** — `hydrogen.transports` entry points registered by installed packages. + +```python +transport_descriptors, transport_errors = discover_plugins( + PluginGroupConfig( + package_names=plugin_packages.transports, + entry_point_group="hydrogen.transports", + plugin_kind="transport", + ), + BaseTransport, +) +``` + +The discovery process: + +- Imports the package and all its submodules via `pkgutil.walk_packages`. +- Inspects each module for non-abstract subclasses of `BaseTransport`. +- Filters out classes that are only imported (not defined) in that module. +- Instantiates each discovered class with no constructor arguments. +- Registers them in `TransportRegistry` by `transport_type`. + +Important consequences for custom transports: + +- Your class must be a non-abstract subclass of `BaseTransport`. +- The class must be defined in the module itself, not only imported from another module. +- The class constructor must work with no arguments. +- Registration is based on `transport_type` — names must be unique. + +## Built-In Transport: `file` + +Implementation: `reporting/transports/file.py` + +Config model: `FileTransportConfig` + +```python +class FileTransportConfig(BaseModel): + path: str = Field() + append_extension: bool = Field(True) +``` + +Configuration: + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: file + path: reports/latest + append_extension: true +``` + +Behavior: + +- `path` is converted to `Path`. +- If `append_extension` is `true` and the destination has no suffix, Hydrogen appends the renderer's file extension. +- Parent directories are created automatically. +- The report is written as UTF-8 text. + +```python +destination = Path(transport.path) +if transport.append_extension and not destination.suffix: + destination = destination.with_suffix(rendered_report.file_extension) + +destination.parent.mkdir(parents=True, exist_ok=True) +destination.write_text(rendered_report.content, encoding="utf-8") +``` + +Practical example: + +- `path: reports/audit` with the JSON renderer becomes `reports/audit.json`. +- `path: reports/audit.txt` stays `reports/audit.txt`. + +## Built-In Transport: `webhook` + +Implementation: `reporting/transports/webhook.py` + +Config model: `WebhookTransportConfig` + +```python +class WebhookTransportConfig(BaseModel): + url: AnyHttpUrl = Field() + method: WebhookMethod = Field(WebhookMethod.POST) + headers: dict[str, str] = Field(default_factory=dict) + timeout_seconds: float = Field(10.0, gt=0) + payload_mode: WebhookPayloadMode = Field(WebhookPayloadMode.RENDERED) +``` + +Configuration: + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: webhook + url: https://example.internal/ingest + method: POST + headers: + Authorization: Bearer token + timeout_seconds: 10 + payload_mode: rendered +``` + +Supported fields: + +- `url` — validated as HTTP or HTTPS URL by Pydantic. +- `method` — `POST`, `PUT`, or `PATCH`. +- `headers` — arbitrary string-to-string HTTP headers. +- `timeout_seconds` — positive float. +- `payload_mode` — `rendered` or `envelope`. + +### `payload_mode: rendered` + +Hydrogen sends the renderer output as the raw HTTP body. + +Header behavior: + +- Existing custom headers are preserved. +- `Content-Type` defaults to the renderer media type if you did not set it explicitly. + +For the built-in JSON renderer, that means `Content-Type: application/json` unless overridden. + +### `payload_mode: envelope` + +Hydrogen sends a JSON object with report metadata and content: + +```json +{ + "format": "json", + "content_type": "application/json", + "file_extension": ".json", + "content": "...rendered report..." +} +``` + +Header behavior: + +- `Content-Type` defaults to `application/json`. + +## Adding a Custom Transport + +To add a custom transport, you need two things: + +1. A transport config model. +2. A transport class implementing `BaseTransport` or `TypedTransport`. + +### Via Package (In-Repository) + +Create a file under `reporting/transports/`: + +`reporting/transports/stdout.py`: + +```python +from pydantic import BaseModel, Field + +from core.base import TypedTransport +from reporting.models import RenderedReport + + +class StdoutTransportConfig(BaseModel): + prefix: str = Field("[Hydrogen]") + + +class StdoutTransport(TypedTransport[StdoutTransportConfig]): + transport_type = "stdout" + config_model = StdoutTransportConfig + + def publish_typed(self, rendered_report: RenderedReport, transport: StdoutTransportConfig) -> str: + print(f"{transport.prefix}\n{rendered_report.content}") + return "stdout" +``` + +Use it in `config.yaml`: + +```yaml +reports: + outputs: + - renderer: + type: json + transport: + type: stdout + prefix: "[Hydrogen Report]" +``` + +### Via Entry Points (Third-Party Package) + +If you distribute your transport as a pip-installable package, register it via entry points: + +```toml +[project.entry-points."hydrogen.transports"] +stdout = "my_package.transports:StdoutTransport" +``` + +The entry point value can be: + +- A class reference (Hydrogen instantiates it). +- A pre-built instance. + +Hydrogen validates that the result is a non-abstract subclass or instance of `BaseTransport`. + +### Via `plugin_packages` Configuration + +You can add additional packages for transport discovery: + +```yaml +plugin_packages: + transports: + - reporting.transports + - mycompany.transport_extras +``` + +Hydrogen scans all listed packages and their submodules for `BaseTransport` subclasses. + +## Configuration Schema for Custom Transports + +Unlike the old architecture (which required a Pydantic discriminated union), transports no longer need to be added to a central type union. The new `PluginConfigRef` model uses `extra="allow"`: + +```python +class PluginConfigRef(BaseModel): + model_config = ConfigDict(extra="allow") + type: str = Field(min_length=1) + + def payload(self) -> dict[str, Any]: + return dict(self.model_extra or {}) +``` + +This means: + +- The `type` field selects the transport by name. +- Any additional YAML fields are captured as the "payload". +- Hydrogen validates the payload against your transport's `config_model` at runtime. + +There is **no central transport union** to modify. As long as your transport is discovered and registered, and its `config_model` matches the provided payload, it will work. + +## The `__init__.py` Export + +You may export your custom transport from `reporting/transports/__init__.py`, but Hydrogen discovery does not depend on that export. + +Why: + +- Hydrogen walks the package tree and imports submodules directly. +- Discovery inspects those imported submodules, not just the `__init__.py` exports. + +So this file is optional for plugin loading: + +```python +from reporting.transports.stdout import StdoutTransport +``` + +It is still useful if you want a cleaner package API for developers. + +## Transport Name Collisions + +`TransportRegistry.register()` stores transports in a dictionary by `transport_type`: + +```python +self._transports[transport.transport_type] = transport +``` + +This means: + +- Names must be unique. +- If two transports declare the same `transport_type`, `register()` raises a `ValueError`. + +In `bootstrap.py`, collisions are caught and converted to `PluginLoadError`: + +```python +try: + transport_registry.register(descriptor.plugin) +except ValueError as exc: + errors.append(PluginLoadError(plugin_kind="transport", source=descriptor.source, message=str(exc))) +``` + +Choose stable, explicit names for custom Hydrogen transports. + +## Limits of the Current Hydrogen Transport Design + +Before designing a transport extension, be aware of the current limits: + +- Transports are instantiated with no constructor arguments. +- There is no dependency injection container. +- There is no built-in retry, queueing, or backoff abstraction. +- The transport type string must match `transport_type` exactly. + +These limits do not make custom transports impossible. They just mean the extension point is code-first rather than fully pluggable from YAML alone. + +## Troubleshooting Custom Transports + +If a transport does not work in Hydrogen, check these points first: + +1. The class subclasses `BaseTransport` or `TypedTransport`. +2. The class is defined inside a module under one of the configured transport packages. +3. The class is not abstract. +4. The class can be instantiated with no arguments. +5. `transport_type` matches the YAML `transport.type` value. +6. The transport's `config_model` validates the YAML payload correctly. +7. The renderer selected by `renderer.type` exists and runs successfully. +8. If using extra YAML fields, they are validated by the transport's `config_model` (not by `PluginConfigRef`). + +## Summary + +In Hydrogen, a transport is the final delivery mechanism for a rendered report. The implementation is straightforward: + +- A transport class discoverable from configured packages or entry points. +- An optional config model validated at runtime via `PluginConfigRef`. + +If you remember that discovery and configuration are separate concerns, extending Hydrogen transports becomes predictable and low-risk. Unlike the old architecture, there is no central transport union to modify — any discovered transport with a valid `config_model` is immediately usable. diff --git a/README.md b/README.md new file mode 100644 index 0000000..e69de29 diff --git a/config.py b/config.py new file mode 100644 index 0000000..c68b76a --- /dev/null +++ b/config.py @@ -0,0 +1,12 @@ +from pathlib import Path + +import yaml + +from core.schemas import config + + +def load_config(path: Path) -> config.Config: + with path.open(encoding="utf-8") as f: + content = yaml.safe_load(f) + + return config.Config.model_validate(content) diff --git a/config.yaml b/config.yaml new file mode 100644 index 0000000..0555d15 --- /dev/null +++ b/config.yaml @@ -0,0 +1,37 @@ +## Script Behavior +exclude_categories: [] +fail_fast: false +dry_run: false +max_concurrency: 4 + +logging: + level: INFO + output: stdout + +plugin_packages: + renderers: + - reporting.exporters + transports: + - reporting.transports + modules: [] + +## Compliance +strict_mode: true # True - exits with code 1, triggering the CI response, False - exits with code: 0. +allow_failures_below: medium + +## Reports +reports: + outputs: + - renderer: + type: json + transport: + type: webhook + url: http://localhost:5000/webhook + method: POST + payload_mode: rendered + +## Module Specific Configuration +modules: + ssh: + enabled: true + test-failure: true diff --git a/core/__init__.py b/core/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/core/base.py b/core/base.py new file mode 100644 index 0000000..e6c724a --- /dev/null +++ b/core/base.py @@ -0,0 +1,76 @@ +from __future__ import annotations + +from abc import ABC, abstractmethod +from typing import TYPE_CHECKING, Any, Generic, TypeVar, cast + +from pydantic import BaseModel + +from core.plugin_types import EmptyPluginConfig +from core.schemas.results import AuditResults + +if TYPE_CHECKING: + from core.schemas.config import ResolvedPluginConfig + from reporting.models import AuditReport, RenderedReport + +TReportTransport = TypeVar("TReportTransport", bound=BaseModel) +TWorkerConfig = TypeVar("TWorkerConfig", bound=BaseModel) + + +class BaseWorker(ABC): + config_model: type[BaseModel] = EmptyPluginConfig + + def __init__(self, config: BaseModel | dict[str, Any] | None = None) -> None: + """ + Config dict will be provided on initialization, providing settings, parsed from config.yaml + """ + self.raw_config = config + self.config: BaseModel | dict[str, Any] + if isinstance(config, BaseModel): + self.config = config.model_dump(mode="python") + else: + self.config = config or {} + + @abstractmethod + def run(self) -> AuditResults: + """ + Executes the scanning. + Method should return AuditResults object, otherwise the output won't be processed. + """ + ... + + +class BaseRenderer(ABC): + content_type: str + media_type: str + file_extension: str + aliases: tuple[str, ...] = () + config_model: type[BaseModel] = EmptyPluginConfig + + @abstractmethod + def render(self, report: AuditReport, config: BaseModel) -> RenderedReport: ... + + +class BaseTransport(ABC): + transport_type: str + config_model: type[BaseModel] = EmptyPluginConfig + + @abstractmethod + def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: ... + + +class TypedTransport(BaseTransport, Generic[TReportTransport], ABC): + config_model: type[TReportTransport] + + def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: + if not isinstance(transport.config, self.config_model): + raise TypeError( + f"Transport '{self.transport_type}' expected {self.config_model.__name__}, " + f"got {type(transport.config).__name__}" + ) + + return self.publish_typed(rendered_report, cast(TReportTransport, transport.config)) + + @abstractmethod + def publish_typed( + self, rendered_report: RenderedReport, transport: TReportTransport + ) -> str: ... diff --git a/core/evaluation.py b/core/evaluation.py new file mode 100644 index 0000000..5cfa50b --- /dev/null +++ b/core/evaluation.py @@ -0,0 +1,26 @@ +from core.schemas.config import Config, ResolvedConfig +from core.schemas.results import AuditResults +from core.schemas.status import AuditSeverity + +SEVERITY_WEIGHTS = { + AuditSeverity.LOW: 1, + AuditSeverity.MEDIUM: 2, + AuditSeverity.HIGH: 3, + AuditSeverity.CRITICAL: 4, +} + + +def evaluate_exit_code(results: list[AuditResults], config: Config | ResolvedConfig) -> int: + severity_threshold = SEVERITY_WEIGHTS.get( + config.allow_failures_below, SEVERITY_WEIGHTS[AuditSeverity.HIGH] + ) + + threshold_crossed = any( + SEVERITY_WEIGHTS.get(finding.severity, 1) >= severity_threshold + for result in results + for finding in result.findings + ) + + if threshold_crossed and config.strict_mode: + return 1 + return 0 diff --git a/core/module_loader.py b/core/module_loader.py new file mode 100644 index 0000000..2c81729 --- /dev/null +++ b/core/module_loader.py @@ -0,0 +1,199 @@ +from __future__ import annotations + +import importlib +import logging +from importlib.metadata import entry_points +from pathlib import Path +from typing import Any + +from pydantic import BaseModel + +from core.base import BaseWorker +from core.plugin_types import HYDROGEN_API_VERSION, EmptyPluginConfig +from core.schemas import ModuleManifest +from core.schemas.config import Config, ResolvedConfig +from core.schemas.modules import LoadedModule, LoadedWorker, ModuleLoadResult +from reporting.models import PluginRuntimeError + +logger = logging.getLogger(__name__) + + +def load_module(package_name: str) -> LoadedModule | None: + module = importlib.import_module(package_name) + raw_manifest = getattr(module, "MANIFEST", None) + + if raw_manifest is None: + logger.warning("module package %s does not export MANIFEST", package_name) + return None + + manifest = ModuleManifest.model_validate(raw_manifest) + if manifest.api_version != HYDROGEN_API_VERSION: + raise ValueError( + f"module package {package_name} targets api_version={manifest.api_version}, " + f"expected {HYDROGEN_API_VERSION}" + ) + + builder = getattr(module, "build_worker", None) + + if builder is None: + logger.warning("module package %s does not export build_worker", package_name) + return None + + if not callable(builder): + logger.warning("build_worker in package %s is not callable", package_name) + return None + + config_model = getattr(module, "CONFIG_MODEL", None) + if config_model is None: + config_model = getattr(getattr(builder, "__self__", None), "config_model", None) + if config_model is None: + config_model = EmptyPluginConfig + if not isinstance(config_model, type) or not issubclass(config_model, BaseModel): + raise TypeError(f"module package {package_name} exports invalid CONFIG_MODEL") + + return LoadedModule( + package_name=package_name, + manifest=manifest, + build_worker=builder, + config_model=config_model, + ) + + +def build_worker( + loaded_module: LoadedModule, config: BaseModel | dict[str, Any] +) -> LoadedWorker | None: + worker = loaded_module.build_worker(config) + if not isinstance(worker, BaseWorker): + logger.warning( + "build_worker in package %s did not return BaseWorker", loaded_module.package_name + ) + return None + + return LoadedWorker(worker=worker, manifest=loaded_module.manifest, config=config) + + +def discover_modules(path: Path, config: Config, package_prefix: str = "modules") -> list[str]: + package_names: list[str] = [] + + if path.exists(): + for entry in path.iterdir(): + if not entry.is_dir() or entry.name.startswith("__"): + continue + if not (entry / "__init__.py").exists(): + continue + package_names.append(f"{package_prefix}.{entry.name}") + + package_names.extend(config.plugin_packages.modules) + + try: + module_entry_points = entry_points(group="hydrogen.modules") + except TypeError: + module_entry_points = entry_points().select(group="hydrogen.modules") + + for module_entry_point in module_entry_points: + package_names.append(module_entry_point.value.partition(":")[0]) + + return list(dict.fromkeys(package_names)) + + +def load_modules(path: Path, config: Config, package_prefix: str = "modules") -> ModuleLoadResult: + workers: list[LoadedWorker] = [] + errors: list[PluginRuntimeError] = [] + + for package_name in discover_modules(path, config, package_prefix): + try: + module = load_module(package_name) + if module is None: + continue + + if module.manifest.category in config.exclude_categories: + logger.info( + "module %s was skipped due to config exclusion", module.manifest.identifier + ) + continue + + module_config = config.modules.get(module.manifest.identifier, {}) + validated_config = module.config_model.model_validate(module_config) + worker = build_worker(module, validated_config) + if worker is not None: + workers.append(worker) + logger.info("module %s has been successfully loaded!", module.manifest.identifier) + except Exception as exc: + logger.exception("failed to load module package %s", package_name) + errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=package_name, + stage="load", + message=str(exc), + ) + ) + + return ModuleLoadResult(loaded_workers=workers, errors=errors) + + +def load_prevalidated_workers( + loaded_modules: list[LoadedModule], + resolved_config: ResolvedConfig, +) -> ModuleLoadResult: + workers: list[LoadedWorker] = [] + errors: list[PluginRuntimeError] = [] + + for loaded_module in loaded_modules: + if loaded_module.manifest.category in resolved_config.exclude_categories: + continue + + module_config = resolved_config.modules.get(loaded_module.manifest.identifier) + if module_config is None: + continue + + try: + worker = build_worker(loaded_module, module_config) + if worker is None: + errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=loaded_module.manifest.identifier, + stage="build_worker", + message="build_worker returned an invalid worker instance", + ) + ) + continue + workers.append(worker) + except Exception as exc: + logger.exception("failed to build worker for module %s", loaded_module.package_name) + errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=loaded_module.manifest.identifier, + stage="build_worker", + message=str(exc), + ) + ) + + return ModuleLoadResult(loaded_workers=workers, errors=errors) + + +def load_module_descriptors( + path: Path, config: Config, package_prefix: str = "modules" +) -> tuple[list[LoadedModule], list[PluginRuntimeError]]: + modules: list[LoadedModule] = [] + errors: list[PluginRuntimeError] = [] + + for package_name in discover_modules(path, config, package_prefix): + try: + module = load_module(package_name) + if module is not None: + modules.append(module) + except Exception as exc: + logger.exception("failed to discover module package %s", package_name) + errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=package_name, + stage="discovery", + message=str(exc), + ) + ) + + return modules, errors diff --git a/core/plugin_types.py b/core/plugin_types.py new file mode 100644 index 0000000..a4359bd --- /dev/null +++ b/core/plugin_types.py @@ -0,0 +1,155 @@ +from __future__ import annotations + +import inspect +import logging +import pkgutil +from dataclasses import dataclass +from importlib import import_module +from importlib.metadata import entry_points +from types import ModuleType +from typing import Any, TypeVar + +from pydantic import BaseModel + +logger = logging.getLogger(__name__) + +TPlugin = TypeVar("TPlugin", bound=object) + +HYDROGEN_API_VERSION = "1" + + +class EmptyPluginConfig(BaseModel): + pass + + +@dataclass(frozen=True) +class PluginLoadError: + plugin_kind: str + source: str + message: str + + +@dataclass(frozen=True) +class PluginDescriptor: + plugin: object + source: str + + +@dataclass(frozen=True) +class PluginGroupConfig: + package_names: list[str] + entry_point_group: str + plugin_kind: str + + +def discover_plugins( + group_config: PluginGroupConfig, + base_class: type[TPlugin], +) -> tuple[list[PluginDescriptor], list[PluginLoadError]]: + descriptors: list[PluginDescriptor] = [] + errors: list[PluginLoadError] = [] + + for package_name in group_config.package_names: + try: + for plugin_class in _discover_plugin_classes(package_name, base_class): + descriptors.append( + PluginDescriptor(plugin=plugin_class(), source=f"package:{package_name}") + ) + except Exception as exc: + logger.exception( + "failed to discover %s package %s", group_config.plugin_kind, package_name + ) + errors.append( + PluginLoadError( + plugin_kind=group_config.plugin_kind, + source=f"package:{package_name}", + message=str(exc), + ) + ) + + try: + available_entry_points = entry_points(group=group_config.entry_point_group) + except TypeError: + available_entry_points = entry_points().select(group=group_config.entry_point_group) + + for entry_point in available_entry_points: + try: + candidate = entry_point.load() + plugin = _instantiate_entry_point_plugin(candidate, base_class) + descriptors.append( + PluginDescriptor( + plugin=plugin, + source=f"entry-point:{group_config.entry_point_group}:{entry_point.name}", + ) + ) + except Exception as exc: + logger.exception( + "failed to load %s entry point %s", + group_config.plugin_kind, + entry_point.name, + ) + errors.append( + PluginLoadError( + plugin_kind=group_config.plugin_kind, + source=f"entry-point:{group_config.entry_point_group}:{entry_point.name}", + message=str(exc), + ) + ) + + return descriptors, errors + + +def ensure_plugin_config_model(plugin: object) -> type[BaseModel]: + config_model = getattr(plugin, "config_model", EmptyPluginConfig) + if not inspect.isclass(config_model) or not issubclass(config_model, BaseModel): + raise TypeError(f"plugin {type(plugin).__name__} has invalid config_model") + return config_model + + +def _instantiate_entry_point_plugin(candidate: Any, base_class: type[TPlugin]) -> TPlugin: + if inspect.isclass(candidate): + if not issubclass(candidate, base_class): + raise TypeError(f"{candidate.__name__} is not a {base_class.__name__}") + if inspect.isabstract(candidate): + raise TypeError(f"{candidate.__name__} is abstract") + return candidate() + + if not isinstance(candidate, base_class): + raise TypeError(f"entry point did not return {base_class.__name__}") + return candidate + + +def _discover_plugin_classes( + package_name: str, base_class: type[TPlugin] +) -> tuple[type[TPlugin], ...]: + package = import_module(package_name) + modules = [package, *(_load_modules(package))] + discovered: list[type[TPlugin]] = [] + seen: set[type[object]] = set() + + for module in modules: + for _, candidate in inspect.getmembers(module, inspect.isclass): + if candidate in seen: + continue + if candidate is base_class or not issubclass(candidate, base_class): + continue + if inspect.isabstract(candidate): + continue + if candidate.__module__ != module.__name__: + continue + + seen.add(candidate) + discovered.append(candidate) + + return tuple(discovered) + + +def _load_modules(package: ModuleType) -> list[ModuleType]: + if not hasattr(package, "__path__"): + return [] + + modules: list[ModuleType] = [] + for module_info in pkgutil.walk_packages(package.__path__, prefix=f"{package.__name__}."): + modules.append(import_module(module_info.name)) + + return modules diff --git a/core/runner.py b/core/runner.py new file mode 100644 index 0000000..ce4e185 --- /dev/null +++ b/core/runner.py @@ -0,0 +1,171 @@ +from __future__ import annotations + +import logging +import sys +from collections import Counter +from concurrent.futures import FIRST_COMPLETED, Future, ThreadPoolExecutor, wait +from pathlib import Path +from time import perf_counter + +from core.evaluation import evaluate_exit_code +from core.module_loader import load_module_descriptors, load_prevalidated_workers +from core.runtime import resolve_config, validate_reporting_plugins +from core.schemas.config import Config +from core.schemas.results import AuditResults +from core.schemas.status import AuditStatus +from reporting.bootstrap import bootstrap_reporting +from reporting.models import ModuleAuditResult, ModuleExecutionStats, PluginRuntimeError +from reporting.service import ReportService + +logger = logging.getLogger(__name__) + + +def run_audit(modules_path: Path, config: Config, package_prefix: str = "modules") -> int: + _configure_logging(config) + + loaded_modules, module_discovery_errors = load_module_descriptors( + modules_path, config, package_prefix + ) + reporting_bootstrap = bootstrap_reporting(config.plugin_packages) + reporting_validation_errors = validate_reporting_plugins( + reporting_bootstrap.renderer_registry, + reporting_bootstrap.transport_registry, + ) + resolved_config, config_errors = resolve_config( + config, + loaded_modules, + reporting_bootstrap.renderer_registry, + reporting_bootstrap.transport_registry, + reporting_bootstrap.errors, + ) + + worker_load_result = load_prevalidated_workers(loaded_modules, resolved_config) + plugin_errors = [ + *module_discovery_errors, + *reporting_validation_errors, + *config_errors, + *worker_load_result.errors, + ] + + results = _run_workers(worker_load_result.loaded_workers, resolved_config) + exit_code = evaluate_exit_code([r.result for r in results], resolved_config) + report_service = ReportService( + renderer_registry=reporting_bootstrap.renderer_registry, + transport_registry=reporting_bootstrap.transport_registry, + ) + report = report_service.build_report(results, exit_code, plugin_errors) + + if resolved_config.dry_run: + logger.info("dry_run enabled, skipping report publish") + else: + publish_errors = _publish_reports(report_service, report, resolved_config.reports) + if publish_errors: + report.plugin_errors.extend(publish_errors) + + return report.exit_code + + +def _run_workers(workers: list, config) -> list[ModuleAuditResult]: + if not workers: + return [] + + indexed_workers = list(enumerate(workers)) + results_by_index: dict[int, ModuleAuditResult] = {} + + with ThreadPoolExecutor(max_workers=config.max_concurrency) as executor: + pending: dict[Future[ModuleAuditResult], int] = {} + next_index = 0 + + while next_index < len(indexed_workers) and len(pending) < config.max_concurrency: + index, worker = indexed_workers[next_index] + pending[executor.submit(_run_single_worker, worker)] = index + next_index += 1 + + stop_submitting = False + while pending: + done, _ = wait(tuple(pending), return_when=FIRST_COMPLETED) + for future in done: + index = pending.pop(future) + result = future.result() + results_by_index[index] = result + + if config.fail_fast and result.result.status is AuditStatus.FAIL: + stop_submitting = True + + while ( + not stop_submitting + and next_index < len(indexed_workers) + and len(pending) < config.max_concurrency + ): + index, worker = indexed_workers[next_index] + pending[executor.submit(_run_single_worker, worker)] = index + next_index += 1 + + return [results_by_index[index] for index in sorted(results_by_index)] + + +def _run_single_worker(worker) -> ModuleAuditResult: + started_at = perf_counter() + try: + result = worker.worker.run() + duration_seconds = perf_counter() - started_at + counts = Counter(finding.severity for finding in result.findings) + return ModuleAuditResult( + module=worker.manifest, + result=result, + stats=ModuleExecutionStats( + duration_seconds=duration_seconds, finding_counts=dict(counts) + ), + ) + except Exception as exc: + logger.exception("worker %s failed during execution", worker.manifest.identifier) + duration_seconds = perf_counter() - started_at + failed_result = AuditResults(status=AuditStatus.FAIL, findings=[], risk_level=1.0) + return ModuleAuditResult( + module=worker.manifest, + result=failed_result, + stats=ModuleExecutionStats(duration_seconds=duration_seconds, finding_counts={}), + error=PluginRuntimeError( + plugin_kind="module", + plugin_name=worker.manifest.identifier, + stage="run", + message=str(exc), + ), + ) + + +def _configure_logging(config: Config) -> None: + level_name = config.logging.level.upper() + level = getattr(logging, level_name, logging.INFO) + root_logger = logging.getLogger() + stream = sys.stdout if config.logging.output.value == "stdout" else sys.stderr + + handler = logging.StreamHandler(stream) + handler.setLevel(level) + handler.setFormatter(logging.Formatter("%(levelname)s:%(name)s:%(message)s")) + + root_logger.handlers.clear() + root_logger.addHandler(handler) + root_logger.setLevel(level) + + +def _publish_reports(report_service, report, outputs) -> list[PluginRuntimeError]: + errors: list[PluginRuntimeError] = [] + for output in outputs: + try: + report_service.publish(report, output) + except Exception as exc: + logger.exception( + "failed to publish output %s via %s", + output.renderer.type, + output.transport.type, + ) + errors.append( + PluginRuntimeError( + plugin_kind="report_output", + plugin_name=f"{output.renderer.type}->{output.transport.type}", + stage="publish", + message=str(exc), + ) + ) + return errors diff --git a/core/runtime.py b/core/runtime.py new file mode 100644 index 0000000..58d7774 --- /dev/null +++ b/core/runtime.py @@ -0,0 +1,148 @@ +from __future__ import annotations + +import logging + +from pydantic import BaseModel, ValidationError + +from core.base import BaseRenderer, BaseTransport +from core.plugin_types import PluginLoadError, ensure_plugin_config_model +from core.schemas.config import Config, ResolvedConfig, ResolvedPluginConfig, ResolvedReportOutput +from core.schemas.modules import LoadedModule +from reporting.models import PluginRuntimeError +from reporting.registry import RendererRegistry +from reporting.transport_registry import TransportRegistry + +logger = logging.getLogger(__name__) + + +def resolve_config( + config: Config, + loaded_modules: list[LoadedModule], + renderer_registry: RendererRegistry, + transport_registry: TransportRegistry, + plugin_errors: list[PluginLoadError] | None = None, +) -> tuple[ResolvedConfig, list[PluginRuntimeError]]: + runtime_errors = [_plugin_load_error_to_runtime_error(error) for error in plugin_errors or ()] + resolved_modules: dict[str, BaseModel] = {} + seen_module_ids: set[str] = set() + + for module in loaded_modules: + if module.manifest.identifier in seen_module_ids: + runtime_errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=module.manifest.identifier, + stage="discovery", + message="duplicate module identifier", + ) + ) + continue + seen_module_ids.add(module.manifest.identifier) + + raw_module_config = config.modules.get(module.manifest.identifier, {}) + try: + resolved_modules[module.manifest.identifier] = module.config_model.model_validate( + raw_module_config + ) + except ValidationError as exc: + runtime_errors.append( + PluginRuntimeError( + plugin_kind="module", + plugin_name=module.manifest.identifier, + stage="config_validation", + message=str(exc), + ) + ) + + resolved_outputs: list[ResolvedReportOutput] = [] + for output in config.reports.outputs: + try: + renderer = renderer_registry.get(output.renderer.type) + transport = transport_registry.get(output.transport.type) + resolved_outputs.append( + ResolvedReportOutput( + renderer=_resolve_plugin_config( + output.renderer.type, output.renderer.payload(), renderer + ), + transport=_resolve_plugin_config( + output.transport.type, output.transport.payload(), transport + ), + ) + ) + except (ValidationError, ValueError, TypeError) as exc: + runtime_errors.append( + PluginRuntimeError( + plugin_kind="report_output", + plugin_name=f"{output.renderer.type}->{output.transport.type}", + stage="config_validation", + message=str(exc), + ) + ) + + resolved_config = ResolvedConfig( + exclude_categories=config.exclude_categories, + allow_failures_below=config.allow_failures_below, + strict_mode=config.strict_mode, + fail_fast=config.fail_fast, + dry_run=config.dry_run, + max_concurrency=config.max_concurrency, + logging=config.logging, + plugin_packages=config.plugin_packages, + reports=resolved_outputs, + modules=resolved_modules, + ) + return resolved_config, runtime_errors + + +def validate_reporting_plugins( + renderer_registry: RendererRegistry, + transport_registry: TransportRegistry, +) -> list[PluginRuntimeError]: + errors: list[PluginRuntimeError] = [] + + errors.extend(_validate_registry_plugins("renderer", renderer_registry._renderers.values())) + errors.extend(_validate_registry_plugins("transport", transport_registry._transports.values())) + + return errors + + +def _validate_registry_plugins(plugin_kind: str, plugins: object) -> list[PluginRuntimeError]: + seen: set[type[object]] = set() + errors: list[PluginRuntimeError] = [] + + for plugin in plugins: + plugin_type = type(plugin) + if plugin_type in seen: + continue + seen.add(plugin_type) + try: + ensure_plugin_config_model(plugin) + except TypeError as exc: + errors.append( + PluginRuntimeError( + plugin_kind=plugin_kind, + plugin_name=plugin_type.__name__, + stage="plugin_validation", + message=str(exc), + ) + ) + + return errors + + +def _resolve_plugin_config( + plugin_type: str, + payload: dict[str, object], + plugin: BaseRenderer | BaseTransport, +) -> ResolvedPluginConfig: + config_model = ensure_plugin_config_model(plugin) + return ResolvedPluginConfig(type=plugin_type, config=config_model.model_validate(payload)) + + +def _plugin_load_error_to_runtime_error(error: PluginLoadError) -> PluginRuntimeError: + return PluginRuntimeError( + plugin_kind=error.plugin_kind, + plugin_name=error.source, + stage="discovery", + message=error.message, + ) diff --git a/core/schemas/__init__.py b/core/schemas/__init__.py new file mode 100644 index 0000000..5c5fefd --- /dev/null +++ b/core/schemas/__init__.py @@ -0,0 +1,5 @@ +from .manifest import ModuleManifest +from .results import AuditFindings, AuditResults +from .status import AuditSeverity, AuditStatus + +__all__ = ["AuditFindings", "AuditResults", "AuditSeverity", "AuditStatus", "ModuleManifest"] diff --git a/core/schemas/config.py b/core/schemas/config.py new file mode 100644 index 0000000..9185553 --- /dev/null +++ b/core/schemas/config.py @@ -0,0 +1,83 @@ +from dataclasses import dataclass +from enum import StrEnum +from typing import Any + +from pydantic import BaseModel, ConfigDict, Field + +from core.schemas.status import AuditSeverity + + +class LoggingOutput(StrEnum): + STDOUT = "stdout" + STDERR = "stderr" + + +class LoggingConfig(BaseModel): + level: str = Field("INFO", min_length=1) + output: LoggingOutput = Field(LoggingOutput.STDOUT) + + +class PluginPackagesConfig(BaseModel): + renderers: list[str] = Field(default_factory=lambda: ["reporting.exporters"]) + transports: list[str] = Field(default_factory=lambda: ["reporting.transports"]) + modules: list[str] = Field(default_factory=list) + + +class PluginConfigRef(BaseModel): + model_config = ConfigDict(extra="allow") + + type: str = Field(min_length=1) + + def payload(self) -> dict[str, Any]: + return dict(self.model_extra or {}) + + +class ReportOutputConfig(BaseModel): + renderer: PluginConfigRef = Field() + transport: PluginConfigRef = Field() + + +class ReportsConfig(BaseModel): + outputs: list[ReportOutputConfig] = Field(min_length=1) + + +class Config(BaseModel): + exclude_categories: list[str] = Field(default_factory=list) + + allow_failures_below: AuditSeverity = Field() + strict_mode: bool = Field() + fail_fast: bool = Field(False) + dry_run: bool = Field(False) + max_concurrency: int = Field(4, ge=1) + logging: LoggingConfig = Field(default_factory=LoggingConfig) + + plugin_packages: PluginPackagesConfig = Field(default_factory=PluginPackagesConfig) + reports: ReportsConfig = Field() + + modules: dict[str, dict[str, Any]] = Field(default_factory=dict) + + +@dataclass(frozen=True) +class ResolvedPluginConfig: + type: str + config: BaseModel + + +@dataclass(frozen=True) +class ResolvedReportOutput: + renderer: ResolvedPluginConfig + transport: ResolvedPluginConfig + + +@dataclass(frozen=True) +class ResolvedConfig: + exclude_categories: list[str] + allow_failures_below: AuditSeverity + strict_mode: bool + fail_fast: bool + dry_run: bool + max_concurrency: int + logging: LoggingConfig + plugin_packages: PluginPackagesConfig + reports: list[ResolvedReportOutput] + modules: dict[str, BaseModel] diff --git a/core/schemas/manifest.py b/core/schemas/manifest.py new file mode 100644 index 0000000..3f33543 --- /dev/null +++ b/core/schemas/manifest.py @@ -0,0 +1,15 @@ +from pydantic import BaseModel, ConfigDict, Field + + +class ModuleManifest(BaseModel): + model_config = ConfigDict( + extra="forbid", + frozen=True, + ) + + identifier: str = Field(pattern=r"^[a-z][a-z0-9_-]*$") + name: str = Field(min_length=1) + category: str = Field(min_length=1) + version: str = Field(pattern=r"^\d+\.\d+\.\d+$") + api_version: str = Field("1", pattern=r"^\d+$") + description: str = "" diff --git a/core/schemas/modules.py b/core/schemas/modules.py new file mode 100644 index 0000000..a4f5427 --- /dev/null +++ b/core/schemas/modules.py @@ -0,0 +1,32 @@ +from collections.abc import Callable +from dataclasses import dataclass +from typing import Any + +from pydantic import BaseModel + +from core.base import BaseWorker +from core.schemas import ModuleManifest +from reporting.models import PluginRuntimeError + +WorkerBuilder = Callable[[BaseModel | dict[str, Any]], BaseWorker] + + +@dataclass(frozen=True) +class LoadedModule: + package_name: str + manifest: ModuleManifest + build_worker: WorkerBuilder + config_model: type[BaseModel] + + +@dataclass(frozen=True) +class LoadedWorker: + worker: BaseWorker + manifest: ModuleManifest + config: BaseModel | dict[str, Any] + + +@dataclass(frozen=True) +class ModuleLoadResult: + loaded_workers: list[LoadedWorker] + errors: list[PluginRuntimeError] diff --git a/core/schemas/results.py b/core/schemas/results.py new file mode 100644 index 0000000..2e4112f --- /dev/null +++ b/core/schemas/results.py @@ -0,0 +1,17 @@ +from pydantic import BaseModel, Field + +from core.schemas.status import AuditSeverity, AuditStatus + + +class AuditFindings(BaseModel): + name: str = Field() + description: str = Field("") + severity: AuditSeverity = Field() + + +class AuditResults(BaseModel): + status: AuditStatus = Field() + findings: list[AuditFindings] = Field() + risk_level: float = Field( + description="value from 0.0 to 1.0, indicating how dangerous the auditresults are." + ) diff --git a/core/schemas/status.py b/core/schemas/status.py new file mode 100644 index 0000000..93441f7 --- /dev/null +++ b/core/schemas/status.py @@ -0,0 +1,14 @@ +from enum import StrEnum + + +class AuditStatus(StrEnum): + PASS = "pass" + SKIPPED = "skipped" + FAIL = "fail" + + +class AuditSeverity(StrEnum): + CRITICAL = "critical" + HIGH = "high" + MEDIUM = "medium" + LOW = "low" diff --git a/loader.py b/loader.py new file mode 100644 index 0000000..b50dccc --- /dev/null +++ b/loader.py @@ -0,0 +1,3 @@ +from core.module_loader import LoadedModule, build_worker, load_module, load_modules + +__all__ = ["LoadedModule", "build_worker", "load_module", "load_modules"] diff --git a/main.py b/main.py new file mode 100644 index 0000000..ac85b2d --- /dev/null +++ b/main.py @@ -0,0 +1,30 @@ +import argparse +import sys +from pathlib import Path + +from config import load_config +from core.runner import run_audit + + +def parse_args() -> argparse.Namespace: + cwd = Path.cwd() + parser = argparse.ArgumentParser( + "hydrogen", + description="Lightweight highly extensible security framework for your server.", + ) + + parser.add_argument("-c", "--config", type=Path, default=cwd / "config.yaml") + parser.add_argument("-p", "--path", type=Path, default=cwd / "modules") + parser.add_argument("--package-prefix", default="modules") + + return parser.parse_args() + + +def main() -> int: + args = parse_args() + config = load_config(args.config) + return run_audit(args.path, config, args.package_prefix) + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/modules/__init__.py b/modules/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/modules/ssh/__init__.py b/modules/ssh/__init__.py new file mode 100644 index 0000000..41a56c3 --- /dev/null +++ b/modules/ssh/__init__.py @@ -0,0 +1,20 @@ +from core.base import BaseWorker +from core.schemas import ModuleManifest + +from .config import SSHModuleConfig +from .ssh_security import SSHSecurityWorker + +MANIFEST = ModuleManifest( + identifier="ssh", + name="SSH Security Audit", + category="ssh", + version="0.1.0", + api_version="1", + description="Audits OpenSSH server configuration.", +) + +CONFIG_MODEL = SSHModuleConfig + + +def build_worker(config: SSHModuleConfig) -> BaseWorker: + return SSHSecurityWorker(config) diff --git a/modules/ssh/config.py b/modules/ssh/config.py new file mode 100644 index 0000000..13ebc80 --- /dev/null +++ b/modules/ssh/config.py @@ -0,0 +1,6 @@ +from pydantic import BaseModel, Field + + +class SSHModuleConfig(BaseModel): + enabled: bool = Field(True) + test_failure: bool = Field(False, alias="test-failure") diff --git a/modules/ssh/ssh_security.py b/modules/ssh/ssh_security.py new file mode 100644 index 0000000..a1483a6 --- /dev/null +++ b/modules/ssh/ssh_security.py @@ -0,0 +1,229 @@ +import os +import shlex +import stat +from pathlib import Path + +from core.base import BaseWorker +from core.schemas.results import AuditFindings, AuditResults +from core.schemas.status import AuditSeverity, AuditStatus + +from .config import SSHModuleConfig + + +class SSHSecurityWorker(BaseWorker): + config_model = SSHModuleConfig + + def __init__(self, config: SSHModuleConfig | None = None) -> None: + super().__init__(config) + self.module_config = config or SSHModuleConfig() + + def run(self) -> AuditResults: + if not self.module_config.enabled: + return AuditResults( + status=AuditStatus.SKIPPED, + findings=[], + risk_level=0.0, + ) + + if self.module_config.test_failure: + findings = [ + AuditFindings( + name="im fucking dumbass rragh", + description="im testing!.", + severity=AuditSeverity.CRITICAL, + ) + ] + return AuditResults( + status=AuditStatus.PASS if not findings else AuditStatus.FAIL, + findings=findings, + risk_level=self._calculate_risk(findings), + ) + + if os.name != "posix": + return AuditResults( + status=AuditStatus.SKIPPED, + findings=[], + risk_level=0.0, + ) + + sshd_config = Path("/etc/ssh/sshd_config") + if not sshd_config.exists(): + return AuditResults( + status=AuditStatus.SKIPPED, + findings=[ + AuditFindings( + name="sshd_config_not_found", + description="SSH server config /etc/ssh/sshd_config was not found.", + severity=AuditSeverity.LOW, + ) + ], + risk_level=0.1, + ) + + findings: list[AuditFindings] = [] + resolved_config = self._read_merged_config(sshd_config) + + findings.extend(self._check_config_directives(resolved_config)) + findings.extend(self._check_config_permissions(sshd_config)) + findings.extend(self._check_host_key_permissions(Path("/etc/ssh"))) + + return AuditResults( + status=AuditStatus.PASS if not findings else AuditStatus.FAIL, + findings=findings, + risk_level=self._calculate_risk(findings), + ) + + def _read_merged_config(self, root_config: Path) -> dict[str, str]: + config_files = [root_config] + include_dir = root_config.parent / "sshd_config.d" + if include_dir.exists() and include_dir.is_dir(): + config_files.extend(sorted(include_dir.glob("*.conf"))) + + directives: dict[str, str] = {} + for config_file in config_files: + for raw_line in config_file.read_text(encoding="utf-8").splitlines(): + line = raw_line.strip() + if not line or line.startswith("#"): + continue + + line = line.split("#", 1)[0].strip() + if not line: + continue + + parts = shlex.split(line, comments=False, posix=True) + if len(parts) < 2: # noqa: PLR2004 + continue + + directive = parts[0].lower() + value = " ".join(parts[1:]) + directives[directive] = value + + return directives + + def _check_config_directives(self, config: dict[str, str]) -> list[AuditFindings]: + findings: list[AuditFindings] = [] + + if config.get("permitrootlogin", "prohibit-password").lower() not in { + "no", + "forced-commands-only", + }: + findings.append( + AuditFindings( + name="permit_root_login_enabled", + description="Directive PermitRootLogin should be set to 'no' or 'forced-commands-only'.", + severity=AuditSeverity.CRITICAL, + ) + ) + + if config.get("passwordauthentication", "yes").lower() != "no": + findings.append( + AuditFindings( + name="password_authentication_enabled", + description="Directive PasswordAuthentication should be disabled to prefer key-based access.", + severity=AuditSeverity.HIGH, + ) + ) + + if config.get("pubkeyauthentication", "yes").lower() != "yes": + findings.append( + AuditFindings( + name="pubkey_authentication_disabled", + description="Directive PubkeyAuthentication should be enabled.", + severity=AuditSeverity.HIGH, + ) + ) + + if config.get("x11forwarding", "no").lower() != "no": + findings.append( + AuditFindings( + name="x11_forwarding_enabled", + description="Directive X11Forwarding should usually be disabled on hardened servers.", + severity=AuditSeverity.MEDIUM, + ) + ) + + max_auth_tries = config.get("maxauthtries", "6") + if max_auth_tries.isdigit() and int(max_auth_tries) > 4: # noqa: PLR2004 + findings.append( + AuditFindings( + name="max_auth_tries_too_high", + description="Directive MaxAuthTries should be 4 or less.", + severity=AuditSeverity.MEDIUM, + ) + ) + + if config.get("permitemptypasswords", "no").lower() != "no": + findings.append( + AuditFindings( + name="empty_passwords_permitted", + description="Directive PermitEmptyPasswords must be disabled.", + severity=AuditSeverity.CRITICAL, + ) + ) + + return findings + + def _check_config_permissions(self, config_path: Path) -> list[AuditFindings]: + findings: list[AuditFindings] = [] + file_stat = config_path.stat() + mode = stat.S_IMODE(file_stat.st_mode) + + if file_stat.st_uid != 0: + findings.append( + AuditFindings( + name="sshd_config_not_owned_by_root", + description="File /etc/ssh/sshd_config should be owned by root.", + severity=AuditSeverity.HIGH, + ) + ) + + if mode & stat.S_IWGRP or mode & stat.S_IWOTH: + findings.append( + AuditFindings( + name="sshd_config_writable_by_non_root", + description="File /etc/ssh/sshd_config must not be writable by group or others.", + severity=AuditSeverity.HIGH, + ) + ) + + return findings + + def _check_host_key_permissions(self, ssh_dir: Path) -> list[AuditFindings]: + findings: list[AuditFindings] = [] + + for key_path in sorted(ssh_dir.glob("ssh_host_*_key")): + file_stat = key_path.stat() + mode = stat.S_IMODE(file_stat.st_mode) + + if file_stat.st_uid != 0: + findings.append( + AuditFindings( + name=f"{key_path.name}_not_owned_by_root", + description=f"Private host key {key_path} should be owned by root.", + severity=AuditSeverity.HIGH, + ) + ) + + if mode & (stat.S_IRWXG | stat.S_IRWXO): + findings.append( + AuditFindings( + name=f"{key_path.name}_permissions_too_open", + description=f"Private host key {key_path} should not be accessible to group or others.", + severity=AuditSeverity.CRITICAL, + ) + ) + + return findings + + def _calculate_risk(self, findings: list[AuditFindings]) -> float: + if not findings: + return 0.0 + + severity_scores = { + AuditSeverity.CRITICAL: 0.45, + AuditSeverity.HIGH: 0.3, + AuditSeverity.MEDIUM: 0.2, + AuditSeverity.LOW: 0.1, + } + risk = sum(severity_scores[finding.severity] for finding in findings) + return min(1.0, risk) diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..6fc4a8e --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,59 @@ +[tool.black] +line-length = 100 +target-version = ['py311'] +include = '\.pyi?$' +extend-exclude = ''' +/( + \.git + | \.venv + | venv + | build + | dist + | alembic +)/ +''' + +[tool.ruff] +# Базовые настройки +line-length = 100 +target-version = "py311" + +exclude = [ + ".git", + ".venv", + "venv", + "build", + "dist", + "alembic", +] + +[tool.ruff.lint] +select = [ + "E", + "W", + "F", + "I", + "N", + "UP", + "B", + "SIM", + "PL", + "RUF", + "TID", + "PT", +] + +ignore = [ + "E501", + "D100", + "D104", + "G004", + "PLR0913", + "RUF001", + "RUF002", + "RUF003", + "B008" +] + +[tool.ruff.lint.isort] +combine-as-imports = true \ No newline at end of file diff --git a/reporting/__init__.py b/reporting/__init__.py new file mode 100644 index 0000000..b99b564 --- /dev/null +++ b/reporting/__init__.py @@ -0,0 +1,4 @@ +from reporting.bootstrap import build_renderer_registry, build_transport_registry +from reporting.service import ReportService + +__all__ = ["ReportService", "build_renderer_registry", "build_transport_registry"] diff --git a/reporting/bootstrap.py b/reporting/bootstrap.py new file mode 100644 index 0000000..f40280e --- /dev/null +++ b/reporting/bootstrap.py @@ -0,0 +1,81 @@ +from __future__ import annotations + +from dataclasses import dataclass + +from core.base import BaseRenderer, BaseTransport +from core.plugin_types import PluginGroupConfig, PluginLoadError, discover_plugins +from core.schemas.config import PluginPackagesConfig +from reporting.registry import RendererRegistry +from reporting.transport_registry import TransportRegistry + + +@dataclass(frozen=True) +class ReportingBootstrapResult: + renderer_registry: RendererRegistry + transport_registry: TransportRegistry + errors: list[PluginLoadError] + + +def bootstrap_reporting(plugin_packages: PluginPackagesConfig) -> ReportingBootstrapResult: + renderer_descriptors, renderer_errors = discover_plugins( + PluginGroupConfig( + package_names=plugin_packages.renderers, + entry_point_group="hydrogen.renderers", + plugin_kind="renderer", + ), + BaseRenderer, + ) + transport_descriptors, transport_errors = discover_plugins( + PluginGroupConfig( + package_names=plugin_packages.transports, + entry_point_group="hydrogen.transports", + plugin_kind="transport", + ), + BaseTransport, + ) + + errors = [*renderer_errors, *transport_errors] + + renderer_registry = RendererRegistry() + for descriptor in renderer_descriptors: + try: + renderer_registry.register(descriptor.plugin) + except ValueError as exc: + errors.append( + PluginLoadError( + plugin_kind="renderer", + source=descriptor.source, + message=str(exc), + ) + ) + + transport_registry = TransportRegistry() + for descriptor in transport_descriptors: + try: + transport_registry.register(descriptor.plugin) + except ValueError as exc: + errors.append( + PluginLoadError( + plugin_kind="transport", + source=descriptor.source, + message=str(exc), + ) + ) + + return ReportingBootstrapResult( + renderer_registry=renderer_registry, + transport_registry=transport_registry, + errors=errors, + ) + + +def build_renderer_registry( + plugin_packages: PluginPackagesConfig | None = None, +) -> RendererRegistry: + return bootstrap_reporting(plugin_packages or PluginPackagesConfig()).renderer_registry + + +def build_transport_registry( + plugin_packages: PluginPackagesConfig | None = None, +) -> TransportRegistry: + return bootstrap_reporting(plugin_packages or PluginPackagesConfig()).transport_registry diff --git a/reporting/exporters/__init__.py b/reporting/exporters/__init__.py new file mode 100644 index 0000000..d57848a --- /dev/null +++ b/reporting/exporters/__init__.py @@ -0,0 +1,3 @@ +from reporting.exporters.json import JsonRenderer + +__all__ = ["JsonRenderer"] diff --git a/reporting/exporters/json.py b/reporting/exporters/json.py new file mode 100644 index 0000000..8da0d76 --- /dev/null +++ b/reporting/exporters/json.py @@ -0,0 +1,21 @@ +import json + +from core.base import BaseRenderer +from core.plugin_types import EmptyPluginConfig +from reporting.models import AuditReport, RenderedReport + + +class JsonRenderer(BaseRenderer): + content_type = "json" + media_type = "application/json" + file_extension = ".json" + aliases = ("application/json",) + config_model = EmptyPluginConfig + + def render(self, report: AuditReport, config: EmptyPluginConfig) -> RenderedReport: + return RenderedReport( + format_name=self.content_type, + media_type=self.media_type, + file_extension=self.file_extension, + content=json.dumps(report.model_dump(mode="json"), ensure_ascii=True, indent=2), + ) diff --git a/reporting/models.py b/reporting/models.py new file mode 100644 index 0000000..a90790a --- /dev/null +++ b/reporting/models.py @@ -0,0 +1,40 @@ +from datetime import UTC, datetime + +from pydantic import BaseModel, Field + +from core.schemas.manifest import ModuleManifest +from core.schemas.results import AuditResults +from core.schemas.status import AuditSeverity + + +class PluginRuntimeError(BaseModel): + plugin_kind: str + plugin_name: str + stage: str + message: str + + +class ModuleExecutionStats(BaseModel): + duration_seconds: float = Field(ge=0) + finding_counts: dict[AuditSeverity, int] = Field(default_factory=dict) + + +class ModuleAuditResult(BaseModel): + module: ModuleManifest + result: AuditResults + stats: ModuleExecutionStats | None = None + error: PluginRuntimeError | None = None + + +class AuditReport(BaseModel): + generated_at: datetime = Field(default_factory=lambda: datetime.now(UTC)) + results: list[ModuleAuditResult] + exit_code: int + plugin_errors: list[PluginRuntimeError] = Field(default_factory=list) + + +class RenderedReport(BaseModel): + format_name: str + media_type: str + file_extension: str + content: str diff --git a/reporting/registry.py b/reporting/registry.py new file mode 100644 index 0000000..f82a904 --- /dev/null +++ b/reporting/registry.py @@ -0,0 +1,32 @@ +from collections.abc import Iterable + +from core.base import BaseRenderer + + +class RendererRegistry: + def __init__(self, renderers: Iterable[BaseRenderer] | None = None) -> None: + self._renderers: dict[str, BaseRenderer] = {} + + for renderer in renderers or (): + self.register(renderer) + + def register(self, renderer: BaseRenderer) -> None: + self._register_key(renderer.content_type, renderer) + for alias in renderer.aliases: + self._register_key(alias, renderer) + + def _register_key(self, key: str, renderer: BaseRenderer) -> None: + if key in self._renderers: + registered = self._renderers[key] + raise ValueError( + f"Renderer key '{key}' is already registered by {type(registered).__name__}" + ) + self._renderers[key] = renderer + + def get(self, content_type: str) -> BaseRenderer: + try: + return self._renderers[content_type] + except KeyError as exc: + raise ValueError( + f"Renderer for content type '{content_type}' is not registered" + ) from exc diff --git a/reporting/service.py b/reporting/service.py new file mode 100644 index 0000000..ae765cb --- /dev/null +++ b/reporting/service.py @@ -0,0 +1,36 @@ +from core.schemas.config import PluginPackagesConfig, ResolvedReportOutput +from reporting.bootstrap import build_renderer_registry, build_transport_registry +from reporting.models import AuditReport, ModuleAuditResult, PluginRuntimeError, RenderedReport +from reporting.registry import RendererRegistry +from reporting.transport_registry import TransportRegistry + + +class ReportService: + def __init__( + self, + renderer_registry: RendererRegistry | None = None, + transport_registry: TransportRegistry | None = None, + plugin_packages: PluginPackagesConfig | None = None, + ) -> None: + self._renderer_registry = renderer_registry or build_renderer_registry(plugin_packages) + self._transport_registry = transport_registry or build_transport_registry(plugin_packages) + + def build_report( + self, + results: list[ModuleAuditResult], + exit_code: int, + plugin_errors: list[PluginRuntimeError] | None = None, + ) -> AuditReport: + return AuditReport(results=results, exit_code=exit_code, plugin_errors=plugin_errors or []) + + def render(self, report: AuditReport, report_config: ResolvedReportOutput) -> RenderedReport: + renderer = self._renderer_registry.get(report_config.renderer.type) + return renderer.render(report, report_config.renderer.config) + + def publish(self, report: AuditReport, report_config: ResolvedReportOutput) -> str: + rendered_report = self.render(report, report_config) + transport = self._transport_registry.get(report_config.transport.type) + return transport.publish(rendered_report, report_config.transport) + + def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]: + return [self.publish(report, output) for output in outputs] diff --git a/reporting/transport_registry.py b/reporting/transport_registry.py new file mode 100644 index 0000000..6befc5f --- /dev/null +++ b/reporting/transport_registry.py @@ -0,0 +1,26 @@ +from collections.abc import Iterable + +from core.base import BaseTransport + + +class TransportRegistry: + def __init__(self, transports: Iterable[BaseTransport] | None = None) -> None: + self._transports: dict[str, BaseTransport] = {} + + for transport in transports or (): + self.register(transport) + + def register(self, transport: BaseTransport) -> None: + if transport.transport_type in self._transports: + registered = self._transports[transport.transport_type] + raise ValueError( + "Transport " + f"'{transport.transport_type}' is already registered by {type(registered).__name__}" + ) + self._transports[transport.transport_type] = transport + + def get(self, transport_type: str) -> BaseTransport: + try: + return self._transports[transport_type] + except KeyError as exc: + raise ValueError(f"Transport '{transport_type}' is not registered") from exc diff --git a/reporting/transports/__init__.py b/reporting/transports/__init__.py new file mode 100644 index 0000000..c77e3ae --- /dev/null +++ b/reporting/transports/__init__.py @@ -0,0 +1,4 @@ +from reporting.transports.file import FileTransport +from reporting.transports.webhook import WebhookTransport + +__all__ = ["FileTransport", "WebhookTransport"] diff --git a/reporting/transports/file.py b/reporting/transports/file.py new file mode 100644 index 0000000..d75aa92 --- /dev/null +++ b/reporting/transports/file.py @@ -0,0 +1,25 @@ +from pathlib import Path + +from pydantic import BaseModel, Field + +from core.base import TypedTransport +from reporting.models import RenderedReport + + +class FileTransportConfig(BaseModel): + path: str = Field() + append_extension: bool = Field(True) + + +class FileTransport(TypedTransport[FileTransportConfig]): + transport_type = "file" + config_model = FileTransportConfig + + def publish_typed(self, rendered_report: RenderedReport, transport: FileTransportConfig) -> str: + destination = Path(transport.path) + if transport.append_extension and not destination.suffix: + destination = destination.with_suffix(rendered_report.file_extension) + + destination.parent.mkdir(parents=True, exist_ok=True) + destination.write_text(rendered_report.content, encoding="utf-8") + return str(destination) diff --git a/reporting/transports/webhook.py b/reporting/transports/webhook.py new file mode 100644 index 0000000..7525b6f --- /dev/null +++ b/reporting/transports/webhook.py @@ -0,0 +1,69 @@ +import json +from enum import StrEnum +from urllib.request import Request, urlopen + +from pydantic import AnyHttpUrl, BaseModel, Field + +from core.base import TypedTransport +from reporting.models import RenderedReport + + +class WebhookMethod(StrEnum): + POST = "POST" + PUT = "PUT" + PATCH = "PATCH" + + +class WebhookPayloadMode(StrEnum): + RENDERED = "rendered" + ENVELOPE = "envelope" + + +class WebhookTransportConfig(BaseModel): + url: AnyHttpUrl = Field() + method: WebhookMethod = Field(WebhookMethod.POST) + headers: dict[str, str] = Field(default_factory=dict) + timeout_seconds: float = Field(10.0, gt=0) + payload_mode: WebhookPayloadMode = Field(WebhookPayloadMode.RENDERED) + + +class WebhookTransport(TypedTransport[WebhookTransportConfig]): + transport_type = "webhook" + config_model = WebhookTransportConfig + + def publish_typed( + self, rendered_report: RenderedReport, transport: WebhookTransportConfig + ) -> str: + body, headers = self._build_request(rendered_report, transport) + request = Request( + url=str(transport.url), + data=body, + headers=headers, + method=transport.method.value, + ) + + with urlopen(request, timeout=transport.timeout_seconds) as response: + return f"{response.status} {transport.url}" + + def _build_request( + self, + rendered_report: RenderedReport, + transport: WebhookTransportConfig, + ) -> tuple[bytes, dict[str, str]]: + headers = dict(transport.headers) + + if transport.payload_mode is WebhookPayloadMode.ENVELOPE: + payload = json.dumps( + { + "format": rendered_report.format_name, + "content_type": rendered_report.media_type, + "file_extension": rendered_report.file_extension, + "content": rendered_report.content, + }, + ensure_ascii=True, + ).encode("utf-8") + headers.setdefault("Content-Type", "application/json") + return payload, headers + + headers.setdefault("Content-Type", rendered_report.media_type) + return rendered_report.content.encode("utf-8"), headers diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..5ae16bd --- /dev/null +++ b/requirements.txt @@ -0,0 +1,2 @@ +pydantic>=2.13.0 +pyyaml>=6.0.0