first commit

This commit is contained in:
2026-07-11 23:54:36 +07:00
commit b07562921b
40 changed files with 3813 additions and 0 deletions

177
.gitignore vendored Normal file
View File

@@ -0,0 +1,177 @@
# ---> Python
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
*$py.class
# C extensions
*.so
# Distribution / packaging
.Python
build/
develop-eggs/
dist/
downloads/
eggs/
.eggs/
lib/
lib64/
parts/
sdist/
var/
wheels/
share/python-wheels/
*.egg-info/
.installed.cfg
*.egg
MANIFEST
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.nox/
.coverage
.coverage.*
.cache
nosetests.xml
coverage.xml
*.cover
*.py,cover
.hypothesis/
.pytest_cache/
cover/
# Translations
*.mo
*.pot
# Django stuff:
*.log
local_settings.py
db.sqlite3
db.sqlite3-journal
# Flask stuff:
instance/
.webassets-cache
# Scrapy stuff:
.scrapy
# Sphinx documentation
docs/_build/
# PyBuilder
.pybuilder/
target/
# Jupyter Notebook
.ipynb_checkpoints
# IPython
profile_default/
ipython_config.py
# pyenv
# For a library or package, you might want to ignore these files since the code is
# intended to run in multiple environments; otherwise, check them in:
# .python-version
# pipenv
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
# However, in case of collaboration, if having platform-specific dependencies or dependencies
# having no cross-platform support, pipenv may install dependencies that don't work, or not
# install all needed dependencies.
#Pipfile.lock
# UV
# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
#uv.lock
# poetry
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
# This is especially recommended for binary packages to ensure reproducibility, and is more
# commonly ignored for libraries.
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
#poetry.lock
# pdm
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
#pdm.lock
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
# in version control.
# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
.pdm.toml
.pdm-python
.pdm-build/
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
__pypackages__/
# Celery stuff
celerybeat-schedule
celerybeat.pid
# SageMath parsed files
*.sage.py
# Environments
.env
*.env
dev.env
.venv
env/
venv/
ENV/
env.bak/
venv.bak/
# Spyder project settings
.spyderproject
.spyproject
# Rope project settings
.ropeproject
# mkdocs documentation
/site
# mypy
.mypy_cache/
.dmypy.json
dmypy.json
# Pyre type checker
.pyre/
# pytype static type analyzer
.pytype/
# Cython debug symbols
cython_debug/
# PyCharm
# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
# and can be added to the global gitignore or merged into this file. For a more nuclear
# option (not recommended) you can uncomment the following to ignore the entire idea folder.
#.idea/
# Ruff stuff:
.ruff_cache/
# PyPI configuration file
.pypirc

606
HYDROGEN_CONFIGURATION.md Normal file
View File

@@ -0,0 +1,606 @@
# Hydrogen Configuration Guide
This document explains how configuration works in Hydrogen, which settings are actually used at runtime, and how configuration connects to module loading, plugin discovery, and report delivery.
## Configuration Entry Points
Hydrogen starts in `main.py`.
- `--config` (`-c`) defaults to `config.yaml` in the current working directory.
- `--path` (`-p`) defaults to the `modules/` directory in the current working directory.
- `--package-prefix` defaults to `modules`.
The configuration file is loaded by `config.py`:
```python
def load_config(path: Path) -> config.Config:
with path.open(encoding="utf-8") as f:
content = yaml.safe_load(f)
return config.Config.model_validate(content)
```
Hydrogen validates the YAML against the Pydantic model `core.schemas.config.Config` before the audit starts.
## Full Configuration Shape
Hydrogen currently expects this top-level structure:
```yaml
exclude_categories: []
fail_fast: false
dry_run: false
max_concurrency: 4
logging:
level: INFO
output: stdout
plugin_packages:
renderers:
- reporting.exporters
transports:
- reporting.transports
modules: []
strict_mode: true
allow_failures_below: medium
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/latest
append_extension: true
modules:
ssh:
enabled: true
```
The schema is defined in `core/schemas/config.py`.
## Top-Level Fields
### `exclude_categories`
Type: `list[str]`
Default: `[]`
This field is actively used in `core/module_loader.py`.
Hydrogen compares each module manifest's `category` value against this list:
```python
if module.manifest.category in config.exclude_categories:
logger.info("module %s was skipped due to config exclusion", module.manifest.identifier)
continue
```
Important details:
- Exclusion is based on `manifest.category`, not on the directory name.
- Exclusion happens before `build_worker()` is called.
- If multiple modules share the same category, one category entry disables all of them.
Example:
```yaml
exclude_categories:
- ssh
- tls
```
### `strict_mode`
Type: `bool`
This field is used in `core/evaluation.py` to determine the final process exit code.
Behavior:
- If a finding reaches the configured severity threshold and `strict_mode` is `true`, Hydrogen returns exit code `1`.
- Otherwise, Hydrogen returns exit code `0`.
Relevant code:
```python
if threshold_crossed and config.strict_mode:
return 1
return 0
```
This is the main switch controlling whether Hydrogen behaves as a CI gate.
### `allow_failures_below`
Type: `AuditSeverity`
Allowed values:
- `low`
- `medium`
- `high`
- `critical`
This field is also used in `core/evaluation.py`.
Hydrogen maps severities to numeric weights:
- `low` -> `1`
- `medium` -> `2`
- `high` -> `3`
- `critical` -> `4`
Then it fails the run when any finding has a weight greater than or equal to the configured threshold.
How to read the field name correctly:
- `allow_failures_below: medium` means Hydrogen tolerates only findings below `medium`.
- In practice, `medium`, `high`, and `critical` findings can trigger exit code `1` when `strict_mode: true`.
- Only `low` findings are below that threshold.
Examples:
```yaml
allow_failures_below: high
```
Meaning:
- `low` and `medium` findings are tolerated.
- `high` and `critical` findings can fail the run when `strict_mode` is enabled.
```yaml
allow_failures_below: critical
```
Meaning:
- Only `critical` findings can fail the run.
### `fail_fast`
Type: `bool`
Default: `false`
When `true`, Hydrogen stops submitting new worker tasks as soon as any module returns a `fail` status. Already-running workers are allowed to finish.
This is used in `core/runner.py`:
```python
if config.fail_fast and result.result.status is AuditStatus.FAIL:
stop_submitting = True
```
### `dry_run`
Type: `bool`
Default: `false`
When `true`, Hydrogen runs all security modules and evaluates exit codes but skips report publishing. Useful for testing module behavior without side effects.
```python
if resolved_config.dry_run:
logger.info("dry_run enabled, skipping report publish")
```
### `max_concurrency`
Type: `int`
Default: `4`
Controls the maximum number of worker threads in the `ThreadPoolExecutor`. Hydrogen runs module workers concurrently up to this limit.
### `logging`
Type: object
Controls logging behavior.
```yaml
logging:
level: INFO
output: stdout
```
Fields:
- `level`: any standard Python logging level name (`DEBUG`, `INFO`, `WARNING`, `ERROR`, `CRITICAL`).
- `output`: `stdout` or `stderr`.
### `plugin_packages`
Type: `PluginPackagesConfig`
This section controls where Hydrogen discovers renderers, transports, and modules.
Schema:
```yaml
plugin_packages:
renderers:
- reporting.exporters
transports:
- reporting.transports
modules:
- mycompany.hydrogen_modules
```
Fields:
- `renderers`: list of Python package names to scan for `BaseRenderer` subclasses.
- `transports`: list of Python package names to scan for `BaseTransport` subclasses.
- `modules`: list of Python package names to scan for modules (in addition to filesystem discovery).
Hydrogen uses `reporting/bootstrap.py` and `core/plugin_types.py` to discover plugins:
```python
def discover_plugins(group_config: PluginGroupConfig, base_class: type[TPlugin]):
for package_name in group_config.package_names:
for plugin_class in _discover_plugin_classes(package_name, base_class):
descriptors.append(PluginDescriptor(plugin=plugin_class(), source=...))
```
Additionally, Hydrogen discovers plugins via Python `entry_points` under the following groups:
- `hydrogen.renderers` for renderers.
- `hydrogen.transports` for transports.
- `hydrogen.modules` for modules.
This allows third-party packages installed via pip to register plugins without manual configuration.
### `reports`
Type: `ReportsConfig`
This section controls report rendering and delivery. Unlike the old single-output schema, Hydrogen now supports **multiple report outputs**.
Schema:
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/audit
append_extension: true
- renderer:
type: json
transport:
type: webhook
url: https://collector.example.com/hydrogen
method: POST
payload_mode: envelope
```
Each output contains:
- `renderer`: a `PluginConfigRef` with a `type` field and optional extra fields.
- `transport`: a `PluginConfigRef` with a `type` field and optional extra fields.
Hydrogen uses it in `reporting/service.py`:
```python
def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]:
return [self.publish(report, output) for output in outputs]
```
The `PluginConfigRef` model:
```python
class PluginConfigRef(BaseModel):
model_config = ConfigDict(extra="allow")
type: str = Field(min_length=1)
def payload(self) -> dict[str, Any]:
return dict(self.model_extra or {})
```
This means:
- The `type` field selects which renderer/transport is used.
- All other fields (the "payload") are passed to the plugin's `config_model` for validation.
- Hydrogen validates the payload using each plugin's own `config_model` Pydantic model.
### `modules`
Type: `dict[str, dict[str, Any]]`
Default: `{}`
This section stores per-module configuration.
Hydrogen looks up module settings by manifest identifier and validates them against the module's `CONFIG_MODEL`:
```python
module_config = config.modules.get(module.manifest.identifier, {})
validated_config = module.config_model.model_validate(module_config)
worker = build_worker(module, validated_config)
```
Important implications:
- The key must match `MANIFEST.identifier`, not the human-readable module name.
- If the key is missing, your module receives an empty dict, which is then validated by the module's `config_model`.
- Module configuration is validated at runtime by the module's own Pydantic config model.
- If validation fails, Hydrogen reports a `PluginRuntimeError` and continues with other modules.
Example:
```yaml
modules:
ssh:
enabled: true
test-failure: false
```
## Report Output Configuration
Each output in `reports.outputs` defines one renderer + transport pair. Hydrogen sends the rendered report through each transport.
### Renderer Config Ref
```yaml
outputs:
- renderer:
type: json
```
Fields:
- `type`: must match a registered renderer's `content_type` or one of its `aliases`.
Extra fields are passed to the renderer's `config_model` for validation.
### Transport Config Ref
```yaml
outputs:
- transport:
type: file
path: reports/security-report
append_extension: true
```
Fields:
- `type`: must match a registered transport's `transport_type`.
- Additional fields are plugin-specific (e.g., `path`, `url`, `method`, `headers`).
Extra fields are validated against the transport's `config_model`.
### Example: File Transport
```yaml
outputs:
- renderer:
type: json
transport:
type: file
path: reports/security-report
append_extension: true
```
Fields:
- `type`: must be `file`
- `path`: output file path
- `append_extension`: when `true`, Hydrogen appends the renderer's extension if the path has no suffix
Example result with the JSON renderer:
- `reports/security-report` becomes `reports/security-report.json`
### Example: Webhook Transport
```yaml
outputs:
- renderer:
type: json
transport:
type: webhook
url: https://example.internal/security-ingest
method: POST
headers:
Authorization: Bearer secret-token
timeout_seconds: 10
payload_mode: envelope
```
Fields:
- `type`: must be `webhook`
- `url`: validated as an HTTP or HTTPS URL by Pydantic
- `method`: `POST`, `PUT`, or `PATCH`
- `headers`: arbitrary string-to-string HTTP headers
- `timeout_seconds`: positive float
- `payload_mode`: `rendered` or `envelope`
Payload behavior:
- `rendered`: Hydrogen sends the renderer output directly with the renderer's `media_type` as `Content-Type`.
- `envelope`: Hydrogen sends a JSON object containing format metadata and the rendered content with `Content-Type: application/json`.
## Plugin Discovery via Entry Points
Hydrogen supports discovering plugins via Python package entry points. This is the recommended way to distribute third-party plugins.
Entry point groups:
- `hydrogen.renderers` — for custom renderers
- `hydrogen.transports` — for custom transports
- `hydrogen.modules` — for custom security modules
Example `pyproject.toml` for a third-party renderer:
```toml
[project.entry-points."hydrogen.renderers"]
markdown = "my_package.renderers:MarkdownRenderer"
```
Entry points can point to either a class (which Hydrogen instantiates) or a pre-built instance. Hydrogen validates that the result is a non-abstract subclass or instance of the expected base class.
## Example Production-Oriented Configurations
### Save a JSON Report to Disk
```yaml
fail_fast: false
dry_run: false
exclude_categories: []
logging:
level: INFO
output: stdout
strict_mode: true
allow_failures_below: high
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/hydrogen-audit
append_extension: true
modules:
ssh:
test-failure: false
```
### Send the Report to Multiple Destinations
```yaml
fail_fast: false
dry_run: false
exclude_categories:
- experimental
logging:
level: DEBUG
output: stdout
strict_mode: true
allow_failures_below: medium
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/hydrogen-audit
append_extension: true
- renderer:
type: json
transport:
type: webhook
url: https://collector.example.com/hydrogen/report
method: POST
headers:
X-Source: hydrogen
timeout_seconds: 5
payload_mode: envelope
modules:
ssh:
test-failure: false
```
## Validation and Failure Modes
Because Hydrogen validates configuration with Pydantic before execution, the following failures happen early:
- invalid enum values such as `allow_failures_below: severe`
- invalid logging output values
- missing required fields such as `reports.outputs`
Other failures happen later at runtime:
- unknown renderer names in `renderer.type` raise an error when `RendererRegistry` cannot find that name.
- unknown transport type names raise an error when `TransportRegistry` cannot find that name.
- plugin config validation errors (when a plugin's `config_model` rejects the provided payload) raise errors during config resolution in `core/runtime.py`.
- module import failures are not swallowed and can stop execution.
## Multi-Output Architecture
Hydrogen supports multiple report outputs per run. Each output defines an independent renderer + transport pair.
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/audit.json
- renderer:
type: markdown
transport:
type: webhook
url: https://wiki.internal/ingest
method: POST
```
Hydrogen iterates through all outputs and publishes each one:
```python
def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]:
return [self.publish(report, output) for output in outputs]
```
If an output fails, Hydrogen collects the error in `plugin_errors` and continues with the next output.
## Relationship Between `--path` and `--package-prefix`
This is not part of YAML, but it matters when you organize Hydrogen modules.
Hydrogen discovers modules from the filesystem path passed to `--path`, but imports them by Python package name using `--package-prefix`.
Example:
```bash
python main.py --path custom_modules --package-prefix custom_modules
```
For this to work:
- `custom_modules/` must exist.
- Each module must be a package directory with its own `__init__.py`.
- Python must be able to import `custom_modules.<module_name>`.
Additionally, modules can come from `plugin_packages.modules` in YAML and from `hydrogen.modules` entry points. All sources are combined in `discover_modules()` with deduplication.
## Internal Runtime Config
After loading and validation, Hydrogen builds a `ResolvedConfig` dataclass:
```python
@dataclass(frozen=True)
class ResolvedConfig:
exclude_categories: list[str]
allow_failures_below: AuditSeverity
strict_mode: bool
fail_fast: bool
dry_run: bool
max_concurrency: int
logging: LoggingConfig
plugin_packages: PluginPackagesConfig
reports: list[ResolvedReportOutput]
modules: dict[str, BaseModel]
```
This is the resolved internal config that modules and reporting use at runtime. Each module's config is already validated into its typed `config_model`.

427
HYDROGEN_RENDERERS.md Normal file
View File

@@ -0,0 +1,427 @@
# Hydrogen Report Rendering and Custom Renderers
This document explains how report rendering works in Hydrogen, how the built-in JSON renderer behaves, and how to add your own renderer.
Renderers transform an `AuditReport` into a transport-ready `RenderedReport`. They are the formatting layer between raw audit data and delivery.
## Where Rendering Happens in Hydrogen
Hydrogen builds reporting output through `ReportService` in `reporting/service.py`.
The render path is:
```python
def render(self, report: AuditReport, report_config: ResolvedReportOutput) -> RenderedReport:
renderer = self._renderer_registry.get(report_config.renderer.type)
return renderer.render(report, report_config.renderer.config)
```
This means:
- `reports.outputs[].renderer.type` selects the renderer by its `content_type` or alias.
- The renderer receives the full `AuditReport` and a validated config `BaseModel`.
- The renderer returns a `RenderedReport`.
- The chosen transport publishes that rendered output.
## Core Renderer Contract
The base class lives in `core/base.py`.
```python
class BaseRenderer(ABC):
content_type: str
media_type: str
file_extension: str
aliases: tuple[str, ...] = ()
config_model: type[BaseModel] = EmptyPluginConfig
@abstractmethod
def render(self, report: AuditReport, config: BaseModel) -> RenderedReport: ...
```
Every Hydrogen renderer must define:
- `content_type` — primary configuration key used in `config.yaml`.
- `media_type` — MIME type used by transports (e.g., `application/json`).
- `file_extension` — default extension for file-based delivery (e.g., `.json`).
- `aliases` — optional alternative lookup names (e.g., `("application/json",)`).
- `config_model` — optional Pydantic model for renderer-specific configuration. Defaults to `EmptyPluginConfig`.
- `render(report, config)` — the rendering implementation.
## The `RenderedReport` Shape
Renderers must return `reporting.models.RenderedReport`:
```python
class RenderedReport(BaseModel):
format_name: str
media_type: str
file_extension: str
content: str
```
Field meanings:
- `format_name` — logical renderer name such as `json` or `markdown`.
- `media_type` — content type such as `application/json`.
- `file_extension` — extension such as `.json`.
- `content` — the final text payload.
This object is the handoff contract between Hydrogen renderers and Hydrogen transports.
## Renderer Configuration
Renderers can accept configuration via `config_model`. Options are passed as extra fields alongside `type` in the YAML configuration.
```yaml
reports:
outputs:
- renderer:
type: my_renderer
pretty: true
include_passed: false
```
Hydrogen resolves this in `core/runtime.py`:
```python
def _resolve_plugin_config(plugin_type, payload, plugin):
config_model = ensure_plugin_config_model(plugin)
return ResolvedPluginConfig(
type=plugin_type,
config=config_model.model_validate(payload)
)
```
If your renderer uses `EmptyPluginConfig` (the default), any extra fields in YAML will cause a validation error because `EmptyPluginConfig` forbids extras. To accept options, define a custom config model.
## Built-In Renderer: JSON
Implementation: `reporting/exporters/json.py`
```python
class JsonRenderer(BaseRenderer):
content_type = "json"
media_type = "application/json"
file_extension = ".json"
aliases = ("application/json",)
config_model = EmptyPluginConfig
def render(self, report: AuditReport, config: EmptyPluginConfig) -> RenderedReport:
return RenderedReport(
format_name=self.content_type,
media_type=self.media_type,
file_extension=self.file_extension,
content=json.dumps(report.model_dump(mode="json"), ensure_ascii=True, indent=2),
)
```
Behavior:
- Serializes `AuditReport` with `report.model_dump(mode="json")`.
- Output is formatted with `json.dumps(..., ensure_ascii=True, indent=2)`.
- The result is UTF-8 safe ASCII JSON text.
Two renderer lookup keys work out of the box:
- `json`
- `application/json`
This is because `RendererRegistry.register()` stores both the primary `content_type` and every alias.
## How Hydrogen Discovers Renderers
Renderer discovery is implemented in `reporting/bootstrap.py` and `core/plugin_types.py`.
Hydrogen discovers renderers from two sources:
1. **Package scan** — Python packages listed in `plugin_packages.renderers` (default: `["reporting.exporters"]`).
2. **Entry points** — `hydrogen.renderers` entry points registered by installed packages.
```python
def bootstrap_reporting(plugin_packages: PluginPackagesConfig) -> ReportingBootstrapResult:
renderer_descriptors, renderer_errors = discover_plugins(
PluginGroupConfig(
package_names=plugin_packages.renderers,
entry_point_group="hydrogen.renderers",
plugin_kind="renderer",
),
BaseRenderer,
)
...
```
The discovery process:
- Imports the package and all its submodules via `pkgutil.walk_packages`.
- Inspects each module for non-abstract subclasses of `BaseRenderer`.
- Filters out classes that are only imported (not defined) in that module.
- Instantiates each discovered class with no constructor arguments.
- Registers them in `RendererRegistry`.
Important implications for custom renderers:
- Your renderer class must be defined in a module under one of the configured packages.
- It must be a concrete subclass of `BaseRenderer`.
- It must have a zero-argument constructor.
- Discovery ignores classes that are only re-exported from another module.
- Names and aliases are registered in a single dictionary — collisions raise `ValueError`.
## How Renderer Selection Works
Hydrogen selects the renderer by `reports.outputs[].renderer.type` from `config.yaml`.
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/latest
```
The lookup:
```python
renderer = self._renderer_registry.get(report_config.renderer.type)
```
If there is no renderer for that key, `RendererRegistry.get()` raises:
```python
ValueError("Renderer for content type '<value>' is not registered")
```
This is a runtime failure, not a configuration-schema failure. `renderer.type` is a plain string in `PluginConfigRef`.
## Adding a Custom Renderer
To add a renderer in Hydrogen, create a module under one of the configured plugin packages and implement a `BaseRenderer` subclass.
### Via Package (In-Repository)
Create a file under `reporting/exporters/`:
File: `reporting/exporters/markdown.py`
```python
from pydantic import BaseModel, Field
from core.base import BaseRenderer
from reporting.models import AuditReport, RenderedReport
class MarkdownRendererConfig(BaseModel):
include_passed: bool = Field(True, description="Include modules with no findings")
class MarkdownRenderer(BaseRenderer):
content_type = "markdown"
media_type = "text/markdown"
file_extension = ".md"
aliases = ("md", "text/markdown")
config_model = MarkdownRendererConfig
def render(self, report: AuditReport, config: MarkdownRendererConfig) -> RenderedReport:
lines = [
"# Hydrogen Security Report",
"",
f"Generated at: {report.generated_at.isoformat()}",
f"Exit code: {report.exit_code}",
"",
]
for module_result in report.results:
if not config.include_passed and not module_result.result.findings:
continue
lines.append(f"## {module_result.module.name}")
lines.append(f"Status: {module_result.result.status}")
lines.append(f"Risk level: {module_result.result.risk_level}")
lines.append("")
if not module_result.result.findings:
lines.append("No findings.")
lines.append("")
continue
for finding in module_result.result.findings:
lines.append(f"- [{finding.severity}] {finding.name}: {finding.description}")
lines.append("")
return RenderedReport(
format_name=self.content_type,
media_type=self.media_type,
file_extension=self.file_extension,
content="\n".join(lines),
)
```
Then use it in configuration:
```yaml
reports:
outputs:
- renderer:
type: markdown
include_passed: false
transport:
type: file
path: reports/security
append_extension: true
```
With the built-in file transport, that produces `reports/security.md`.
### Via Entry Points (Third-Party Package)
If you distribute your renderer as a pip-installable package, register it via entry points:
```toml
[project.entry-points."hydrogen.renderers"]
markdown = "my_package.renderers:MarkdownRenderer"
```
The entry point value can be:
- A class reference (Hydrogen instantiates it).
- A pre-built instance.
Hydrogen validates that the result is a non-abstract subclass or instance of `BaseRenderer`.
### Via `plugin_packages` Configuration
You can add additional packages for renderer discovery:
```yaml
plugin_packages:
renderers:
- reporting.exporters
- mycompany.reporting_extras
```
Hydrogen scans all listed packages and their submodules for `BaseRenderer` subclasses.
## Renderer Aliases
Aliases are optional, but useful.
```python
aliases = ("md", "text/markdown")
```
This allows the following Hydrogen config values to point to the same renderer:
- `markdown`
- `md`
- `text/markdown`
Registry behavior:
```python
self._renderers[renderer.content_type] = renderer
for alias in renderer.aliases:
self._renderers[alias] = renderer
```
## Renderer Name Collisions
Hydrogen stores renderers in a dictionary. If two renderers register the same `content_type` or alias, `register()` raises a `ValueError` with a descriptive message.
In `bootstrap.py`, this error is caught and converted to a `PluginLoadError`:
```python
try:
renderer_registry.register(descriptor.plugin)
except ValueError as exc:
errors.append(PluginLoadError(plugin_kind="renderer", source=descriptor.source, message=str(exc)))
```
Choose unique names and aliases for custom renderers.
## Configurable Renderer Behavior
Unlike the old architecture, Hydrogen now supports renderer-specific configuration via `config_model`.
Define a Pydantic model for your options:
```python
class MyRendererConfig(BaseModel):
pretty: bool = Field(True)
max_findings: int = Field(50, ge=1)
```
Set it on the renderer class:
```python
class MyRenderer(BaseRenderer):
content_type = "my_format"
config_model = MyRendererConfig
def render(self, report: AuditReport, config: MyRendererConfig) -> RenderedReport:
if config.pretty:
...
```
Configure it in YAML:
```yaml
outputs:
- renderer:
type: my_format
pretty: false
max_findings: 100
```
The extra fields (`pretty`, `max_findings`) are extracted from `PluginConfigRef.payload()` and validated against `MyRendererConfig`.
## The `__init__.py` Export
Renderers do not need to be re-exported from the package's `__init__.py` for discovery to work. Hydrogen discovers submodules directly with `pkgutil.walk_packages`.
An explicit export like this is optional:
```python
from reporting.exporters.markdown import MarkdownRenderer
```
It can still be useful for developer ergonomics and explicit imports.
## Troubleshooting Custom Renderers
If Hydrogen does not pick up your renderer, check the following:
1. The file is inside one of the configured plugin packages.
2. The class subclasses `BaseRenderer`.
3. The class is not abstract.
4. The class is defined in that module, not only imported into it.
5. The class can be instantiated with no arguments.
6. `config_model` is a valid Pydantic `BaseModel` subclass (if defined).
7. `reports.outputs[].renderer.type` matches either `content_type` or one of the aliases.
8. The renderer returns a proper `RenderedReport`.
9. The `render()` method accepts two arguments: `report` and `config`.
## Design Guidance for Hydrogen Renderers
Good custom renderers usually follow these rules:
1. Keep rendering pure and deterministic.
2. Treat the renderer as a formatting layer, not a transport.
3. Use `media_type` and `file_extension` consistently so transports behave correctly.
4. Keep `content_type` short and stable because it becomes part of configuration.
5. Return text content only, because `RenderedReport.content` is currently a string.
6. If you need configurable behavior, define a `config_model` and use the `config` parameter.
7. Do not rely on global state — all required inputs come through `render(report, config)`.
## Summary
Hydrogen renderers are straightforward extension points:
- They live under configurable plugin packages (default: `reporting/exporters/`).
- They convert `AuditReport` into `RenderedReport`.
- They are selected by `reports.outputs[].renderer.type`.
- They receive a typed config object validated by their `config_model`.
- They are auto-discovered from packages and entry points.
The main difference from the old architecture is that renderers now receive a config parameter and support per-renderer options through `config_model`.

View File

@@ -0,0 +1,520 @@
# Hydrogen Custom Security Modules
This document explains how to add your own security checking modules to Hydrogen. It is based on the actual module loading flow implemented in `core/module_loader.py`, `core/base.py`, `core/schemas/modules.py`, and the example SSH module under `modules/ssh`.
## How Hydrogen Loads Modules
Hydrogen starts module execution in `core/runner.py`:
```python
loaded_modules, module_discovery_errors = load_module_descriptors(modules_path, config, package_prefix)
```
The module loader then discovers modules from three sources (combined and deduplicated):
1. **Filesystem directory** — scanned via `--path` (default: `modules/`).
2. **`plugin_packages.modules`** — Python packages listed in `config.yaml`.
3. **`hydrogen.modules` entry points** — registered by installed Python packages.
```python
def discover_modules(path: Path, config: Config, package_prefix: str = "modules") -> list[str]:
# 1. Scan filesystem
for entry in path.iterdir():
if entry.is_dir() and not entry.name.startswith("__") and (entry / "__init__.py").exists():
package_names.append(f"{package_prefix}.{entry.name}")
# 2. From config
package_names.extend(config.plugin_packages.modules)
# 3. From entry points
for ep in entry_points(group="hydrogen.modules"):
package_names.append(ep.value.partition(":")[0])
return list(dict.fromkeys(package_names)) # deduplicate
```
Filesystem rules:
- Hydrogen only inspects direct children of the modules directory.
- Entries must be directories.
- Directory names starting with `__` are skipped.
- The directory must contain `__init__.py`.
- The package is imported as `<package_prefix>.<directory_name>`.
Hydrogen does **not** recursively discover module packages inside nested directories. A module must be a first-level package under the selected modules root.
## Required Module Contract
Each Hydrogen module package must provide three things:
- `MANIFEST` — a `ModuleManifest` instance.
- `build_worker(config)` — a callable that returns a `BaseWorker`.
- `CONFIG_MODEL` (optional, recommended) — a Pydantic `BaseModel` subclass for config validation.
If either `MANIFEST` or `build_worker` is missing, the loader logs a warning and skips the module.
### `MANIFEST`
`MANIFEST` is validated against `core.schemas.manifest.ModuleManifest`.
```python
class ModuleManifest(BaseModel):
model_config = ConfigDict(extra="forbid", frozen=True)
identifier: str = Field(pattern=r"^[a-z][a-z0-9_-]*$")
name: str = Field(min_length=1)
category: str = Field(min_length=1)
version: str = Field(pattern=r"^\d+\.\d+\.\d+$")
api_version: str = Field("1", pattern=r"^\d+$")
description: str = ""
```
Required fields:
- `identifier` — must match `^[a-z][a-z0-9_-]*$`
- `name` — human-readable name, non-empty
- `category` — group name used for exclusion, non-empty
- `version` — semantic version `X.Y.Z`
- `api_version` — must match Hydrogen's `HYDROGEN_API_VERSION` (currently `"1"`)
`description` is optional and defaults to an empty string.
Example from the built-in SSH module:
```python
MANIFEST = ModuleManifest(
identifier="ssh",
name="SSH Security Audit",
category="ssh",
version="0.1.0",
api_version="1",
description="Audits OpenSSH server configuration.",
)
```
**Important:** `api_version` must match Hydrogen's API version. If it does not match, the module is rejected with a `ValueError`.
### `CONFIG_MODEL`
Modules can export a `CONFIG_MODEL` — a Pydantic `BaseModel` subclass that defines the expected shape of the module's configuration section from `config.yaml`.
```python
# modules/my_check/config.py
from pydantic import BaseModel, Field
class MyCheckConfig(BaseModel):
enabled: bool = Field(True)
threshold: int = Field(5, ge=1)
```
```python
# modules/my_check/__init__.py
from .config import MyCheckConfig
CONFIG_MODEL = MyCheckConfig
```
When `CONFIG_MODEL` is defined, Hydrogen validates the module's config section from YAML against this model before passing it to `build_worker()`:
```python
module_config = config.modules.get(module.manifest.identifier, {})
validated_config = module.config_model.model_validate(module_config)
worker = build_worker(module, validated_config)
```
If `CONFIG_MODEL` is not defined, Hydrogen passes an `EmptyPluginConfig` instance (an empty Pydantic model).
### `build_worker(config)`
The loader expects a callable named `build_worker`.
Signature:
```python
def build_worker(config: MyCheckConfig) -> BaseWorker:
...
```
Rules:
- It must be callable.
- It receives the **validated** module-specific config (a `BaseModel` instance from `CONFIG_MODEL`).
- It must return an instance of `BaseWorker`.
If it returns anything else, Hydrogen logs a warning and skips the module.
## Worker Contract
All custom workers must inherit `core.base.BaseWorker`.
```python
class BaseWorker(ABC):
config_model: type[BaseModel] = EmptyPluginConfig
def __init__(self, config: BaseModel | dict[str, Any] | None = None) -> None:
self.raw_config = config
self.config: BaseModel | dict[str, Any]
if isinstance(config, BaseModel):
self.config = config.model_dump(mode="python")
else:
self.config = config or {}
@abstractmethod
def run(self) -> AuditResults:
...
```
Key points:
- `__init__` accepts either a `BaseModel` (preferred) or a raw `dict`.
- The config is stored as both `self.raw_config` (original form) and `self.config` (dict form for easy access).
- `config_model` is a class attribute that can be set on the worker for reference.
- `run()` must return `AuditResults`.
The built-in SSH module shows the recommended pattern:
```python
class SSHSecurityWorker(BaseWorker):
config_model = SSHModuleConfig
def __init__(self, config: SSHModuleConfig | None = None) -> None:
super().__init__(config)
self.module_config = config or SSHModuleConfig()
def run(self) -> AuditResults:
if not self.module_config.enabled:
return AuditResults(status=AuditStatus.SKIPPED, findings=[], risk_level=0.0)
...
```
## Required Result Shape
Your worker's `run()` method must return `core.schemas.results.AuditResults`.
```python
class AuditResults(BaseModel):
status: AuditStatus
findings: list[AuditFindings]
risk_level: float
```
### `status`
Allowed values from `AuditStatus`:
- `pass` — the audit ran and found no issues.
- `fail` — the audit ran and produced one or more findings.
- `skipped` — the audit was intentionally not applicable on the current system.
### `findings`
Each finding is an `AuditFindings` object:
```python
class AuditFindings(BaseModel):
name: str
description: str = ""
severity: AuditSeverity
```
Severity values:
- `low`
- `medium`
- `high`
- `critical`
These severities feed directly into Hydrogen's exit code calculation.
### `risk_level`
`risk_level` is a float from `0.0` to `1.0` indicating how dangerous the results are.
Practical guidance:
- Return `0.0` for clean results.
- Return a bounded value up to `1.0`.
- Keep the calculation deterministic so reports remain comparable over time.
## Full Module Layout
Here is the full structure Hydrogen expects:
```text
modules/
my_check/
__init__.py
config.py # optional, recommended
worker.py
```
### `modules/my_check/config.py`
```python
from pydantic import BaseModel, Field
class MyCheckConfig(BaseModel):
enabled: bool = Field(True)
must_be_enabled: bool = Field(False)
```
### `modules/my_check/__init__.py`
```python
from core.base import BaseWorker
from core.schemas import ModuleManifest
from .config import MyCheckConfig
from .worker import MyCheckWorker
MANIFEST = ModuleManifest(
identifier="my_check",
name="My Custom Check",
category="custom",
version="1.0.0",
api_version="1",
description="Checks a custom hardening rule.",
)
CONFIG_MODEL = MyCheckConfig
def build_worker(config: MyCheckConfig) -> BaseWorker:
return MyCheckWorker(config)
```
### `modules/my_check/worker.py`
```python
from core.base import BaseWorker
from core.schemas.results import AuditFindings, AuditResults
from core.schemas.status import AuditSeverity, AuditStatus
from .config import MyCheckConfig
class MyCheckWorker(BaseWorker):
config_model = MyCheckConfig
def __init__(self, config: MyCheckConfig | None = None) -> None:
super().__init__(config)
self.module_config = config or MyCheckConfig()
def run(self) -> AuditResults:
findings: list[AuditFindings] = []
if self.module_config.must_be_enabled is not True:
findings.append(
AuditFindings(
name="custom_rule_disabled",
description="The custom rule is not enabled.",
severity=AuditSeverity.HIGH,
)
)
return AuditResults(
status=AuditStatus.PASS if not findings else AuditStatus.FAIL,
findings=findings,
risk_level=0.0 if not findings else 0.3,
)
```
## Wiring Module Configuration
Hydrogen validates and passes module config by manifest identifier.
Given this manifest:
```python
MANIFEST = ModuleManifest(identifier="my_check", ...)
```
The YAML section must be:
```yaml
modules:
my_check:
must_be_enabled: true
```
Hydrogen does this:
```python
module_config = config.modules.get(module.manifest.identifier, {})
validated_config = module.config_model.model_validate(module_config)
worker = build_worker(module, validated_config)
```
Important details:
- The key is `my_check`, not the directory label shown to users in a report title.
- If the key is missing, your module receives an empty dict, which is validated against your `CONFIG_MODEL` (using defaults).
- Config validation is strict: if you define `CONFIG_MODEL`, extra fields in YAML that are not in the model will raise a validation error (unless `extra="allow"` is set).
- If config validation fails, Hydrogen reports a `PluginRuntimeError` and continues with other modules — your worker is not created.
## Category-Based Exclusion
Hydrogen can disable modules by manifest category.
Example:
```yaml
exclude_categories:
- custom
```
If your module's manifest has `category="custom"`, Hydrogen skips it before worker creation.
This is useful for:
- Experimental checks.
- Platform-specific checks.
- Expensive checks that should be disabled in some environments.
## Module Discovery via Entry Points
Third-party modules can register themselves via Python entry points in the `hydrogen.modules` group.
Example `pyproject.toml`:
```toml
[project.entry-points."hydrogen.modules"]
my_check = "mycompany.hydrogen_modules.my_check"
```
The entry point value must be a Python package path (module or package). Hydrogen imports it and looks for `MANIFEST` and `build_worker` inside.
## Import Mechanics and Packaging Rules
Hydrogen imports modules using Python package names, not only filesystem paths.
For a module directory named `my_check` and default package prefix `modules`, Hydrogen imports:
```python
import modules.my_check
```
Therefore:
- The module directory must be importable as a Python package.
- Its `__init__.py` is the integration entry point.
- Import-time exceptions are not swallowed — they propagate and fail the module.
For third-party packages installed via pip, set `plugin_packages.modules` in config:
```yaml
plugin_packages:
modules:
- mycompany.hydrogen_modules
```
Or use entry points as described above.
## Behavior on Loader Errors
Hydrogen is tolerant of contract violations, but only up to a point.
**Handled by warning and skip:**
- Missing `MANIFEST`.
- Missing `build_worker`.
- Non-callable `build_worker`.
- `build_worker()` returning something that is not a `BaseWorker`.
**Handled by error reporting (module skipped, run continues):**
- Import failures while importing the package (caught in `load_modules()`).
- Config validation errors in `resolve_config()`.
- Runtime exceptions inside `worker.run()` (caught in `_run_single_worker()`).
**Not handled silently (can fail the entire run):**
- Manifest validation failures (e.g., `api_version` mismatch).
- Missing `CONFIG_MODEL` export with invalid type.
## Duplicate Module Detection
Hydrogen detects duplicate module identifiers during config resolution:
```python
if module.manifest.identifier in seen_module_ids:
runtime_errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=module.manifest.identifier,
stage="discovery",
message="duplicate module identifier",
)
)
continue
```
If two sources provide the same module identifier, the second one is rejected with an error.
## Recommended Development Pattern
For custom Hydrogen modules, the safest pattern is:
1. Define a `CONFIG_MODEL` Pydantic model for type-safe config validation.
2. Keep `__init__.py` small — just `MANIFEST`, `CONFIG_MODEL`, and `build_worker()`.
3. Put most logic in a separate worker module.
4. Set `config_model` on your worker class for consistency.
5. Use stable, machine-friendly finding names.
6. Return `skipped` when the check is not applicable on the current platform.
7. Use `api_version="1"` (match Hydrogen's `HYDROGEN_API_VERSION`).
## Example: Running a Custom Module Tree
If your module packages live outside the default `modules/` directory, you must align the filesystem path and package prefix.
Example:
```bash
python main.py --path custom_checks --package-prefix custom_checks
```
Then Hydrogen expects packages like:
```text
custom_checks/
__init__.py
my_check/
__init__.py
config.py
worker.py
```
Alternatively, add your package to `plugin_packages.modules`:
```yaml
plugin_packages:
modules:
- custom_checks.my_check
```
## Testing Checklist for a New Module
Before relying on a new Hydrogen module, verify all of the following:
1. The package imports cleanly.
2. `MANIFEST` validates successfully (especially `api_version`).
3. `CONFIG_MODEL` (if defined) validates correctly.
4. `build_worker()` returns a real `BaseWorker` instance.
5. `run()` always returns `AuditResults`.
6. All findings use valid `AuditSeverity` values.
7. The module behaves correctly when its config section is missing.
8. The module behaves correctly when it is excluded by category.
9. The module handles runtime errors gracefully (Hydrogen wraps exceptions).
## Summary
Hydrogen custom security modules are intentionally simple:
- One package per module.
- One manifest with `api_version` matching Hydrogen's.
- Optional `CONFIG_MODEL` for validated config.
- One worker builder receiving a validated config model.
- One worker object that returns structured results.
The most important implementation details are that module discovery supports three sources (filesystem, config packages, entry points), module config is validated by `CONFIG_MODEL` and keyed by `MANIFEST.identifier`, and import-time failures are reported but do not stop the entire run.

403
HYDROGEN_TRANSPORTS.md Normal file
View File

@@ -0,0 +1,403 @@
# Hydrogen Custom Report Transports
This document explains how report transports work in Hydrogen, how the built-in transports behave, and what you need to do to add your own transport implementation.
## Where Transports Fit in the Hydrogen Pipeline
Hydrogen builds a report after all security modules finish.
The reporting flow in `reporting/service.py` is:
1. Build an `AuditReport`.
2. Render it into a `RenderedReport`.
3. Publish the rendered report through a transport.
The relevant method is:
```python
def publish(self, report: AuditReport, report_config: ResolvedReportOutput) -> str:
rendered_report = self.render(report, report_config)
transport = self._transport_registry.get(report_config.transport.type)
return transport.publish(rendered_report, report_config.transport)
```
This means transports operate on already-rendered content. They do not receive raw findings directly.
## Core Transport Contract
The base transport contract lives in `core/base.py`.
```python
class BaseTransport(ABC):
transport_type: str
config_model: type[BaseModel] = EmptyPluginConfig
@abstractmethod
def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: ...
```
Every transport must define:
- `transport_type` — the name used in `config.yaml`.
- `config_model` — optional Pydantic model for transport-specific configuration.
- `publish(...)` — the delivery implementation.
The `transport` parameter is a `ResolvedPluginConfig`:
```python
@dataclass(frozen=True)
class ResolvedPluginConfig:
type: str
config: BaseModel
```
- `transport.type` — the transport type string.
- `transport.config` — the validated config model instance.
The return value of `publish()` is a string. The built-in transports return values such as the output path or an HTTP status string. Hydrogen currently does not use that return value elsewhere, but your transport still needs to return one.
## Typed Transport Pattern
Hydrogen provides `TypedTransport` as a convenient base class for transports with their own config model:
```python
class TypedTransport(BaseTransport, Generic[TReportTransport], ABC):
config_model: type[TReportTransport]
def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str:
if not isinstance(transport.config, self.config_model):
raise TypeError(
f"Transport '{self.transport_type}' expected {self.config_model.__name__}, "
f"got {type(transport.config).__name__}"
)
return self.publish_typed(rendered_report, cast(TReportTransport, transport.config))
@abstractmethod
def publish_typed(self, rendered_report: RenderedReport, transport: TReportTransport) -> str: ...
```
This is the recommended pattern when your transport has its own configuration model.
Benefits:
- Runtime type checking for the config object.
- Cleaner transport code — `publish_typed()` receives the typed config directly.
- No manual casting inside `publish()`.
## How Hydrogen Discovers Transports
Transport discovery is implemented in `reporting/bootstrap.py` and `core/plugin_types.py`.
Hydrogen discovers transports from two sources:
1. **Package scan** — Python packages listed in `plugin_packages.transports` (default: `["reporting.transports"]`).
2. **Entry points** — `hydrogen.transports` entry points registered by installed packages.
```python
transport_descriptors, transport_errors = discover_plugins(
PluginGroupConfig(
package_names=plugin_packages.transports,
entry_point_group="hydrogen.transports",
plugin_kind="transport",
),
BaseTransport,
)
```
The discovery process:
- Imports the package and all its submodules via `pkgutil.walk_packages`.
- Inspects each module for non-abstract subclasses of `BaseTransport`.
- Filters out classes that are only imported (not defined) in that module.
- Instantiates each discovered class with no constructor arguments.
- Registers them in `TransportRegistry` by `transport_type`.
Important consequences for custom transports:
- Your class must be a non-abstract subclass of `BaseTransport`.
- The class must be defined in the module itself, not only imported from another module.
- The class constructor must work with no arguments.
- Registration is based on `transport_type` — names must be unique.
## Built-In Transport: `file`
Implementation: `reporting/transports/file.py`
Config model: `FileTransportConfig`
```python
class FileTransportConfig(BaseModel):
path: str = Field()
append_extension: bool = Field(True)
```
Configuration:
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: file
path: reports/latest
append_extension: true
```
Behavior:
- `path` is converted to `Path`.
- If `append_extension` is `true` and the destination has no suffix, Hydrogen appends the renderer's file extension.
- Parent directories are created automatically.
- The report is written as UTF-8 text.
```python
destination = Path(transport.path)
if transport.append_extension and not destination.suffix:
destination = destination.with_suffix(rendered_report.file_extension)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(rendered_report.content, encoding="utf-8")
```
Practical example:
- `path: reports/audit` with the JSON renderer becomes `reports/audit.json`.
- `path: reports/audit.txt` stays `reports/audit.txt`.
## Built-In Transport: `webhook`
Implementation: `reporting/transports/webhook.py`
Config model: `WebhookTransportConfig`
```python
class WebhookTransportConfig(BaseModel):
url: AnyHttpUrl = Field()
method: WebhookMethod = Field(WebhookMethod.POST)
headers: dict[str, str] = Field(default_factory=dict)
timeout_seconds: float = Field(10.0, gt=0)
payload_mode: WebhookPayloadMode = Field(WebhookPayloadMode.RENDERED)
```
Configuration:
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: webhook
url: https://example.internal/ingest
method: POST
headers:
Authorization: Bearer token
timeout_seconds: 10
payload_mode: rendered
```
Supported fields:
- `url` — validated as HTTP or HTTPS URL by Pydantic.
- `method` — `POST`, `PUT`, or `PATCH`.
- `headers` — arbitrary string-to-string HTTP headers.
- `timeout_seconds` — positive float.
- `payload_mode` — `rendered` or `envelope`.
### `payload_mode: rendered`
Hydrogen sends the renderer output as the raw HTTP body.
Header behavior:
- Existing custom headers are preserved.
- `Content-Type` defaults to the renderer media type if you did not set it explicitly.
For the built-in JSON renderer, that means `Content-Type: application/json` unless overridden.
### `payload_mode: envelope`
Hydrogen sends a JSON object with report metadata and content:
```json
{
"format": "json",
"content_type": "application/json",
"file_extension": ".json",
"content": "...rendered report..."
}
```
Header behavior:
- `Content-Type` defaults to `application/json`.
## Adding a Custom Transport
To add a custom transport, you need two things:
1. A transport config model.
2. A transport class implementing `BaseTransport` or `TypedTransport`.
### Via Package (In-Repository)
Create a file under `reporting/transports/`:
`reporting/transports/stdout.py`:
```python
from pydantic import BaseModel, Field
from core.base import TypedTransport
from reporting.models import RenderedReport
class StdoutTransportConfig(BaseModel):
prefix: str = Field("[Hydrogen]")
class StdoutTransport(TypedTransport[StdoutTransportConfig]):
transport_type = "stdout"
config_model = StdoutTransportConfig
def publish_typed(self, rendered_report: RenderedReport, transport: StdoutTransportConfig) -> str:
print(f"{transport.prefix}\n{rendered_report.content}")
return "stdout"
```
Use it in `config.yaml`:
```yaml
reports:
outputs:
- renderer:
type: json
transport:
type: stdout
prefix: "[Hydrogen Report]"
```
### Via Entry Points (Third-Party Package)
If you distribute your transport as a pip-installable package, register it via entry points:
```toml
[project.entry-points."hydrogen.transports"]
stdout = "my_package.transports:StdoutTransport"
```
The entry point value can be:
- A class reference (Hydrogen instantiates it).
- A pre-built instance.
Hydrogen validates that the result is a non-abstract subclass or instance of `BaseTransport`.
### Via `plugin_packages` Configuration
You can add additional packages for transport discovery:
```yaml
plugin_packages:
transports:
- reporting.transports
- mycompany.transport_extras
```
Hydrogen scans all listed packages and their submodules for `BaseTransport` subclasses.
## Configuration Schema for Custom Transports
Unlike the old architecture (which required a Pydantic discriminated union), transports no longer need to be added to a central type union. The new `PluginConfigRef` model uses `extra="allow"`:
```python
class PluginConfigRef(BaseModel):
model_config = ConfigDict(extra="allow")
type: str = Field(min_length=1)
def payload(self) -> dict[str, Any]:
return dict(self.model_extra or {})
```
This means:
- The `type` field selects the transport by name.
- Any additional YAML fields are captured as the "payload".
- Hydrogen validates the payload against your transport's `config_model` at runtime.
There is **no central transport union** to modify. As long as your transport is discovered and registered, and its `config_model` matches the provided payload, it will work.
## The `__init__.py` Export
You may export your custom transport from `reporting/transports/__init__.py`, but Hydrogen discovery does not depend on that export.
Why:
- Hydrogen walks the package tree and imports submodules directly.
- Discovery inspects those imported submodules, not just the `__init__.py` exports.
So this file is optional for plugin loading:
```python
from reporting.transports.stdout import StdoutTransport
```
It is still useful if you want a cleaner package API for developers.
## Transport Name Collisions
`TransportRegistry.register()` stores transports in a dictionary by `transport_type`:
```python
self._transports[transport.transport_type] = transport
```
This means:
- Names must be unique.
- If two transports declare the same `transport_type`, `register()` raises a `ValueError`.
In `bootstrap.py`, collisions are caught and converted to `PluginLoadError`:
```python
try:
transport_registry.register(descriptor.plugin)
except ValueError as exc:
errors.append(PluginLoadError(plugin_kind="transport", source=descriptor.source, message=str(exc)))
```
Choose stable, explicit names for custom Hydrogen transports.
## Limits of the Current Hydrogen Transport Design
Before designing a transport extension, be aware of the current limits:
- Transports are instantiated with no constructor arguments.
- There is no dependency injection container.
- There is no built-in retry, queueing, or backoff abstraction.
- The transport type string must match `transport_type` exactly.
These limits do not make custom transports impossible. They just mean the extension point is code-first rather than fully pluggable from YAML alone.
## Troubleshooting Custom Transports
If a transport does not work in Hydrogen, check these points first:
1. The class subclasses `BaseTransport` or `TypedTransport`.
2. The class is defined inside a module under one of the configured transport packages.
3. The class is not abstract.
4. The class can be instantiated with no arguments.
5. `transport_type` matches the YAML `transport.type` value.
6. The transport's `config_model` validates the YAML payload correctly.
7. The renderer selected by `renderer.type` exists and runs successfully.
8. If using extra YAML fields, they are validated by the transport's `config_model` (not by `PluginConfigRef`).
## Summary
In Hydrogen, a transport is the final delivery mechanism for a rendered report. The implementation is straightforward:
- A transport class discoverable from configured packages or entry points.
- An optional config model validated at runtime via `PluginConfigRef`.
If you remember that discovery and configuration are separate concerns, extending Hydrogen transports becomes predictable and low-risk. Unlike the old architecture, there is no central transport union to modify — any discovered transport with a valid `config_model` is immediately usable.

0
README.md Normal file
View File

12
config.py Normal file
View File

@@ -0,0 +1,12 @@
from pathlib import Path
import yaml
from core.schemas import config
def load_config(path: Path) -> config.Config:
with path.open(encoding="utf-8") as f:
content = yaml.safe_load(f)
return config.Config.model_validate(content)

37
config.yaml Normal file
View File

@@ -0,0 +1,37 @@
## Script Behavior
exclude_categories: []
fail_fast: false
dry_run: false
max_concurrency: 4
logging:
level: INFO
output: stdout
plugin_packages:
renderers:
- reporting.exporters
transports:
- reporting.transports
modules: []
## Compliance
strict_mode: true # True - exits with code 1, triggering the CI response, False - exits with code: 0.
allow_failures_below: medium
## Reports
reports:
outputs:
- renderer:
type: json
transport:
type: webhook
url: http://localhost:5000/webhook
method: POST
payload_mode: rendered
## Module Specific Configuration
modules:
ssh:
enabled: true
test-failure: true

0
core/__init__.py Normal file
View File

76
core/base.py Normal file
View File

@@ -0,0 +1,76 @@
from __future__ import annotations
from abc import ABC, abstractmethod
from typing import TYPE_CHECKING, Any, Generic, TypeVar, cast
from pydantic import BaseModel
from core.plugin_types import EmptyPluginConfig
from core.schemas.results import AuditResults
if TYPE_CHECKING:
from core.schemas.config import ResolvedPluginConfig
from reporting.models import AuditReport, RenderedReport
TReportTransport = TypeVar("TReportTransport", bound=BaseModel)
TWorkerConfig = TypeVar("TWorkerConfig", bound=BaseModel)
class BaseWorker(ABC):
config_model: type[BaseModel] = EmptyPluginConfig
def __init__(self, config: BaseModel | dict[str, Any] | None = None) -> None:
"""
Config dict will be provided on initialization, providing settings, parsed from config.yaml
"""
self.raw_config = config
self.config: BaseModel | dict[str, Any]
if isinstance(config, BaseModel):
self.config = config.model_dump(mode="python")
else:
self.config = config or {}
@abstractmethod
def run(self) -> AuditResults:
"""
Executes the scanning.
Method should return AuditResults object, otherwise the output won't be processed.
"""
...
class BaseRenderer(ABC):
content_type: str
media_type: str
file_extension: str
aliases: tuple[str, ...] = ()
config_model: type[BaseModel] = EmptyPluginConfig
@abstractmethod
def render(self, report: AuditReport, config: BaseModel) -> RenderedReport: ...
class BaseTransport(ABC):
transport_type: str
config_model: type[BaseModel] = EmptyPluginConfig
@abstractmethod
def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str: ...
class TypedTransport(BaseTransport, Generic[TReportTransport], ABC):
config_model: type[TReportTransport]
def publish(self, rendered_report: RenderedReport, transport: ResolvedPluginConfig) -> str:
if not isinstance(transport.config, self.config_model):
raise TypeError(
f"Transport '{self.transport_type}' expected {self.config_model.__name__}, "
f"got {type(transport.config).__name__}"
)
return self.publish_typed(rendered_report, cast(TReportTransport, transport.config))
@abstractmethod
def publish_typed(
self, rendered_report: RenderedReport, transport: TReportTransport
) -> str: ...

26
core/evaluation.py Normal file
View File

@@ -0,0 +1,26 @@
from core.schemas.config import Config, ResolvedConfig
from core.schemas.results import AuditResults
from core.schemas.status import AuditSeverity
SEVERITY_WEIGHTS = {
AuditSeverity.LOW: 1,
AuditSeverity.MEDIUM: 2,
AuditSeverity.HIGH: 3,
AuditSeverity.CRITICAL: 4,
}
def evaluate_exit_code(results: list[AuditResults], config: Config | ResolvedConfig) -> int:
severity_threshold = SEVERITY_WEIGHTS.get(
config.allow_failures_below, SEVERITY_WEIGHTS[AuditSeverity.HIGH]
)
threshold_crossed = any(
SEVERITY_WEIGHTS.get(finding.severity, 1) >= severity_threshold
for result in results
for finding in result.findings
)
if threshold_crossed and config.strict_mode:
return 1
return 0

199
core/module_loader.py Normal file
View File

@@ -0,0 +1,199 @@
from __future__ import annotations
import importlib
import logging
from importlib.metadata import entry_points
from pathlib import Path
from typing import Any
from pydantic import BaseModel
from core.base import BaseWorker
from core.plugin_types import HYDROGEN_API_VERSION, EmptyPluginConfig
from core.schemas import ModuleManifest
from core.schemas.config import Config, ResolvedConfig
from core.schemas.modules import LoadedModule, LoadedWorker, ModuleLoadResult
from reporting.models import PluginRuntimeError
logger = logging.getLogger(__name__)
def load_module(package_name: str) -> LoadedModule | None:
module = importlib.import_module(package_name)
raw_manifest = getattr(module, "MANIFEST", None)
if raw_manifest is None:
logger.warning("module package %s does not export MANIFEST", package_name)
return None
manifest = ModuleManifest.model_validate(raw_manifest)
if manifest.api_version != HYDROGEN_API_VERSION:
raise ValueError(
f"module package {package_name} targets api_version={manifest.api_version}, "
f"expected {HYDROGEN_API_VERSION}"
)
builder = getattr(module, "build_worker", None)
if builder is None:
logger.warning("module package %s does not export build_worker", package_name)
return None
if not callable(builder):
logger.warning("build_worker in package %s is not callable", package_name)
return None
config_model = getattr(module, "CONFIG_MODEL", None)
if config_model is None:
config_model = getattr(getattr(builder, "__self__", None), "config_model", None)
if config_model is None:
config_model = EmptyPluginConfig
if not isinstance(config_model, type) or not issubclass(config_model, BaseModel):
raise TypeError(f"module package {package_name} exports invalid CONFIG_MODEL")
return LoadedModule(
package_name=package_name,
manifest=manifest,
build_worker=builder,
config_model=config_model,
)
def build_worker(
loaded_module: LoadedModule, config: BaseModel | dict[str, Any]
) -> LoadedWorker | None:
worker = loaded_module.build_worker(config)
if not isinstance(worker, BaseWorker):
logger.warning(
"build_worker in package %s did not return BaseWorker", loaded_module.package_name
)
return None
return LoadedWorker(worker=worker, manifest=loaded_module.manifest, config=config)
def discover_modules(path: Path, config: Config, package_prefix: str = "modules") -> list[str]:
package_names: list[str] = []
if path.exists():
for entry in path.iterdir():
if not entry.is_dir() or entry.name.startswith("__"):
continue
if not (entry / "__init__.py").exists():
continue
package_names.append(f"{package_prefix}.{entry.name}")
package_names.extend(config.plugin_packages.modules)
try:
module_entry_points = entry_points(group="hydrogen.modules")
except TypeError:
module_entry_points = entry_points().select(group="hydrogen.modules")
for module_entry_point in module_entry_points:
package_names.append(module_entry_point.value.partition(":")[0])
return list(dict.fromkeys(package_names))
def load_modules(path: Path, config: Config, package_prefix: str = "modules") -> ModuleLoadResult:
workers: list[LoadedWorker] = []
errors: list[PluginRuntimeError] = []
for package_name in discover_modules(path, config, package_prefix):
try:
module = load_module(package_name)
if module is None:
continue
if module.manifest.category in config.exclude_categories:
logger.info(
"module %s was skipped due to config exclusion", module.manifest.identifier
)
continue
module_config = config.modules.get(module.manifest.identifier, {})
validated_config = module.config_model.model_validate(module_config)
worker = build_worker(module, validated_config)
if worker is not None:
workers.append(worker)
logger.info("module %s has been successfully loaded!", module.manifest.identifier)
except Exception as exc:
logger.exception("failed to load module package %s", package_name)
errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=package_name,
stage="load",
message=str(exc),
)
)
return ModuleLoadResult(loaded_workers=workers, errors=errors)
def load_prevalidated_workers(
loaded_modules: list[LoadedModule],
resolved_config: ResolvedConfig,
) -> ModuleLoadResult:
workers: list[LoadedWorker] = []
errors: list[PluginRuntimeError] = []
for loaded_module in loaded_modules:
if loaded_module.manifest.category in resolved_config.exclude_categories:
continue
module_config = resolved_config.modules.get(loaded_module.manifest.identifier)
if module_config is None:
continue
try:
worker = build_worker(loaded_module, module_config)
if worker is None:
errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=loaded_module.manifest.identifier,
stage="build_worker",
message="build_worker returned an invalid worker instance",
)
)
continue
workers.append(worker)
except Exception as exc:
logger.exception("failed to build worker for module %s", loaded_module.package_name)
errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=loaded_module.manifest.identifier,
stage="build_worker",
message=str(exc),
)
)
return ModuleLoadResult(loaded_workers=workers, errors=errors)
def load_module_descriptors(
path: Path, config: Config, package_prefix: str = "modules"
) -> tuple[list[LoadedModule], list[PluginRuntimeError]]:
modules: list[LoadedModule] = []
errors: list[PluginRuntimeError] = []
for package_name in discover_modules(path, config, package_prefix):
try:
module = load_module(package_name)
if module is not None:
modules.append(module)
except Exception as exc:
logger.exception("failed to discover module package %s", package_name)
errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=package_name,
stage="discovery",
message=str(exc),
)
)
return modules, errors

155
core/plugin_types.py Normal file
View File

@@ -0,0 +1,155 @@
from __future__ import annotations
import inspect
import logging
import pkgutil
from dataclasses import dataclass
from importlib import import_module
from importlib.metadata import entry_points
from types import ModuleType
from typing import Any, TypeVar
from pydantic import BaseModel
logger = logging.getLogger(__name__)
TPlugin = TypeVar("TPlugin", bound=object)
HYDROGEN_API_VERSION = "1"
class EmptyPluginConfig(BaseModel):
pass
@dataclass(frozen=True)
class PluginLoadError:
plugin_kind: str
source: str
message: str
@dataclass(frozen=True)
class PluginDescriptor:
plugin: object
source: str
@dataclass(frozen=True)
class PluginGroupConfig:
package_names: list[str]
entry_point_group: str
plugin_kind: str
def discover_plugins(
group_config: PluginGroupConfig,
base_class: type[TPlugin],
) -> tuple[list[PluginDescriptor], list[PluginLoadError]]:
descriptors: list[PluginDescriptor] = []
errors: list[PluginLoadError] = []
for package_name in group_config.package_names:
try:
for plugin_class in _discover_plugin_classes(package_name, base_class):
descriptors.append(
PluginDescriptor(plugin=plugin_class(), source=f"package:{package_name}")
)
except Exception as exc:
logger.exception(
"failed to discover %s package %s", group_config.plugin_kind, package_name
)
errors.append(
PluginLoadError(
plugin_kind=group_config.plugin_kind,
source=f"package:{package_name}",
message=str(exc),
)
)
try:
available_entry_points = entry_points(group=group_config.entry_point_group)
except TypeError:
available_entry_points = entry_points().select(group=group_config.entry_point_group)
for entry_point in available_entry_points:
try:
candidate = entry_point.load()
plugin = _instantiate_entry_point_plugin(candidate, base_class)
descriptors.append(
PluginDescriptor(
plugin=plugin,
source=f"entry-point:{group_config.entry_point_group}:{entry_point.name}",
)
)
except Exception as exc:
logger.exception(
"failed to load %s entry point %s",
group_config.plugin_kind,
entry_point.name,
)
errors.append(
PluginLoadError(
plugin_kind=group_config.plugin_kind,
source=f"entry-point:{group_config.entry_point_group}:{entry_point.name}",
message=str(exc),
)
)
return descriptors, errors
def ensure_plugin_config_model(plugin: object) -> type[BaseModel]:
config_model = getattr(plugin, "config_model", EmptyPluginConfig)
if not inspect.isclass(config_model) or not issubclass(config_model, BaseModel):
raise TypeError(f"plugin {type(plugin).__name__} has invalid config_model")
return config_model
def _instantiate_entry_point_plugin(candidate: Any, base_class: type[TPlugin]) -> TPlugin:
if inspect.isclass(candidate):
if not issubclass(candidate, base_class):
raise TypeError(f"{candidate.__name__} is not a {base_class.__name__}")
if inspect.isabstract(candidate):
raise TypeError(f"{candidate.__name__} is abstract")
return candidate()
if not isinstance(candidate, base_class):
raise TypeError(f"entry point did not return {base_class.__name__}")
return candidate
def _discover_plugin_classes(
package_name: str, base_class: type[TPlugin]
) -> tuple[type[TPlugin], ...]:
package = import_module(package_name)
modules = [package, *(_load_modules(package))]
discovered: list[type[TPlugin]] = []
seen: set[type[object]] = set()
for module in modules:
for _, candidate in inspect.getmembers(module, inspect.isclass):
if candidate in seen:
continue
if candidate is base_class or not issubclass(candidate, base_class):
continue
if inspect.isabstract(candidate):
continue
if candidate.__module__ != module.__name__:
continue
seen.add(candidate)
discovered.append(candidate)
return tuple(discovered)
def _load_modules(package: ModuleType) -> list[ModuleType]:
if not hasattr(package, "__path__"):
return []
modules: list[ModuleType] = []
for module_info in pkgutil.walk_packages(package.__path__, prefix=f"{package.__name__}."):
modules.append(import_module(module_info.name))
return modules

171
core/runner.py Normal file
View File

@@ -0,0 +1,171 @@
from __future__ import annotations
import logging
import sys
from collections import Counter
from concurrent.futures import FIRST_COMPLETED, Future, ThreadPoolExecutor, wait
from pathlib import Path
from time import perf_counter
from core.evaluation import evaluate_exit_code
from core.module_loader import load_module_descriptors, load_prevalidated_workers
from core.runtime import resolve_config, validate_reporting_plugins
from core.schemas.config import Config
from core.schemas.results import AuditResults
from core.schemas.status import AuditStatus
from reporting.bootstrap import bootstrap_reporting
from reporting.models import ModuleAuditResult, ModuleExecutionStats, PluginRuntimeError
from reporting.service import ReportService
logger = logging.getLogger(__name__)
def run_audit(modules_path: Path, config: Config, package_prefix: str = "modules") -> int:
_configure_logging(config)
loaded_modules, module_discovery_errors = load_module_descriptors(
modules_path, config, package_prefix
)
reporting_bootstrap = bootstrap_reporting(config.plugin_packages)
reporting_validation_errors = validate_reporting_plugins(
reporting_bootstrap.renderer_registry,
reporting_bootstrap.transport_registry,
)
resolved_config, config_errors = resolve_config(
config,
loaded_modules,
reporting_bootstrap.renderer_registry,
reporting_bootstrap.transport_registry,
reporting_bootstrap.errors,
)
worker_load_result = load_prevalidated_workers(loaded_modules, resolved_config)
plugin_errors = [
*module_discovery_errors,
*reporting_validation_errors,
*config_errors,
*worker_load_result.errors,
]
results = _run_workers(worker_load_result.loaded_workers, resolved_config)
exit_code = evaluate_exit_code([r.result for r in results], resolved_config)
report_service = ReportService(
renderer_registry=reporting_bootstrap.renderer_registry,
transport_registry=reporting_bootstrap.transport_registry,
)
report = report_service.build_report(results, exit_code, plugin_errors)
if resolved_config.dry_run:
logger.info("dry_run enabled, skipping report publish")
else:
publish_errors = _publish_reports(report_service, report, resolved_config.reports)
if publish_errors:
report.plugin_errors.extend(publish_errors)
return report.exit_code
def _run_workers(workers: list, config) -> list[ModuleAuditResult]:
if not workers:
return []
indexed_workers = list(enumerate(workers))
results_by_index: dict[int, ModuleAuditResult] = {}
with ThreadPoolExecutor(max_workers=config.max_concurrency) as executor:
pending: dict[Future[ModuleAuditResult], int] = {}
next_index = 0
while next_index < len(indexed_workers) and len(pending) < config.max_concurrency:
index, worker = indexed_workers[next_index]
pending[executor.submit(_run_single_worker, worker)] = index
next_index += 1
stop_submitting = False
while pending:
done, _ = wait(tuple(pending), return_when=FIRST_COMPLETED)
for future in done:
index = pending.pop(future)
result = future.result()
results_by_index[index] = result
if config.fail_fast and result.result.status is AuditStatus.FAIL:
stop_submitting = True
while (
not stop_submitting
and next_index < len(indexed_workers)
and len(pending) < config.max_concurrency
):
index, worker = indexed_workers[next_index]
pending[executor.submit(_run_single_worker, worker)] = index
next_index += 1
return [results_by_index[index] for index in sorted(results_by_index)]
def _run_single_worker(worker) -> ModuleAuditResult:
started_at = perf_counter()
try:
result = worker.worker.run()
duration_seconds = perf_counter() - started_at
counts = Counter(finding.severity for finding in result.findings)
return ModuleAuditResult(
module=worker.manifest,
result=result,
stats=ModuleExecutionStats(
duration_seconds=duration_seconds, finding_counts=dict(counts)
),
)
except Exception as exc:
logger.exception("worker %s failed during execution", worker.manifest.identifier)
duration_seconds = perf_counter() - started_at
failed_result = AuditResults(status=AuditStatus.FAIL, findings=[], risk_level=1.0)
return ModuleAuditResult(
module=worker.manifest,
result=failed_result,
stats=ModuleExecutionStats(duration_seconds=duration_seconds, finding_counts={}),
error=PluginRuntimeError(
plugin_kind="module",
plugin_name=worker.manifest.identifier,
stage="run",
message=str(exc),
),
)
def _configure_logging(config: Config) -> None:
level_name = config.logging.level.upper()
level = getattr(logging, level_name, logging.INFO)
root_logger = logging.getLogger()
stream = sys.stdout if config.logging.output.value == "stdout" else sys.stderr
handler = logging.StreamHandler(stream)
handler.setLevel(level)
handler.setFormatter(logging.Formatter("%(levelname)s:%(name)s:%(message)s"))
root_logger.handlers.clear()
root_logger.addHandler(handler)
root_logger.setLevel(level)
def _publish_reports(report_service, report, outputs) -> list[PluginRuntimeError]:
errors: list[PluginRuntimeError] = []
for output in outputs:
try:
report_service.publish(report, output)
except Exception as exc:
logger.exception(
"failed to publish output %s via %s",
output.renderer.type,
output.transport.type,
)
errors.append(
PluginRuntimeError(
plugin_kind="report_output",
plugin_name=f"{output.renderer.type}->{output.transport.type}",
stage="publish",
message=str(exc),
)
)
return errors

148
core/runtime.py Normal file
View File

@@ -0,0 +1,148 @@
from __future__ import annotations
import logging
from pydantic import BaseModel, ValidationError
from core.base import BaseRenderer, BaseTransport
from core.plugin_types import PluginLoadError, ensure_plugin_config_model
from core.schemas.config import Config, ResolvedConfig, ResolvedPluginConfig, ResolvedReportOutput
from core.schemas.modules import LoadedModule
from reporting.models import PluginRuntimeError
from reporting.registry import RendererRegistry
from reporting.transport_registry import TransportRegistry
logger = logging.getLogger(__name__)
def resolve_config(
config: Config,
loaded_modules: list[LoadedModule],
renderer_registry: RendererRegistry,
transport_registry: TransportRegistry,
plugin_errors: list[PluginLoadError] | None = None,
) -> tuple[ResolvedConfig, list[PluginRuntimeError]]:
runtime_errors = [_plugin_load_error_to_runtime_error(error) for error in plugin_errors or ()]
resolved_modules: dict[str, BaseModel] = {}
seen_module_ids: set[str] = set()
for module in loaded_modules:
if module.manifest.identifier in seen_module_ids:
runtime_errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=module.manifest.identifier,
stage="discovery",
message="duplicate module identifier",
)
)
continue
seen_module_ids.add(module.manifest.identifier)
raw_module_config = config.modules.get(module.manifest.identifier, {})
try:
resolved_modules[module.manifest.identifier] = module.config_model.model_validate(
raw_module_config
)
except ValidationError as exc:
runtime_errors.append(
PluginRuntimeError(
plugin_kind="module",
plugin_name=module.manifest.identifier,
stage="config_validation",
message=str(exc),
)
)
resolved_outputs: list[ResolvedReportOutput] = []
for output in config.reports.outputs:
try:
renderer = renderer_registry.get(output.renderer.type)
transport = transport_registry.get(output.transport.type)
resolved_outputs.append(
ResolvedReportOutput(
renderer=_resolve_plugin_config(
output.renderer.type, output.renderer.payload(), renderer
),
transport=_resolve_plugin_config(
output.transport.type, output.transport.payload(), transport
),
)
)
except (ValidationError, ValueError, TypeError) as exc:
runtime_errors.append(
PluginRuntimeError(
plugin_kind="report_output",
plugin_name=f"{output.renderer.type}->{output.transport.type}",
stage="config_validation",
message=str(exc),
)
)
resolved_config = ResolvedConfig(
exclude_categories=config.exclude_categories,
allow_failures_below=config.allow_failures_below,
strict_mode=config.strict_mode,
fail_fast=config.fail_fast,
dry_run=config.dry_run,
max_concurrency=config.max_concurrency,
logging=config.logging,
plugin_packages=config.plugin_packages,
reports=resolved_outputs,
modules=resolved_modules,
)
return resolved_config, runtime_errors
def validate_reporting_plugins(
renderer_registry: RendererRegistry,
transport_registry: TransportRegistry,
) -> list[PluginRuntimeError]:
errors: list[PluginRuntimeError] = []
errors.extend(_validate_registry_plugins("renderer", renderer_registry._renderers.values()))
errors.extend(_validate_registry_plugins("transport", transport_registry._transports.values()))
return errors
def _validate_registry_plugins(plugin_kind: str, plugins: object) -> list[PluginRuntimeError]:
seen: set[type[object]] = set()
errors: list[PluginRuntimeError] = []
for plugin in plugins:
plugin_type = type(plugin)
if plugin_type in seen:
continue
seen.add(plugin_type)
try:
ensure_plugin_config_model(plugin)
except TypeError as exc:
errors.append(
PluginRuntimeError(
plugin_kind=plugin_kind,
plugin_name=plugin_type.__name__,
stage="plugin_validation",
message=str(exc),
)
)
return errors
def _resolve_plugin_config(
plugin_type: str,
payload: dict[str, object],
plugin: BaseRenderer | BaseTransport,
) -> ResolvedPluginConfig:
config_model = ensure_plugin_config_model(plugin)
return ResolvedPluginConfig(type=plugin_type, config=config_model.model_validate(payload))
def _plugin_load_error_to_runtime_error(error: PluginLoadError) -> PluginRuntimeError:
return PluginRuntimeError(
plugin_kind=error.plugin_kind,
plugin_name=error.source,
stage="discovery",
message=error.message,
)

5
core/schemas/__init__.py Normal file
View File

@@ -0,0 +1,5 @@
from .manifest import ModuleManifest
from .results import AuditFindings, AuditResults
from .status import AuditSeverity, AuditStatus
__all__ = ["AuditFindings", "AuditResults", "AuditSeverity", "AuditStatus", "ModuleManifest"]

83
core/schemas/config.py Normal file
View File

@@ -0,0 +1,83 @@
from dataclasses import dataclass
from enum import StrEnum
from typing import Any
from pydantic import BaseModel, ConfigDict, Field
from core.schemas.status import AuditSeverity
class LoggingOutput(StrEnum):
STDOUT = "stdout"
STDERR = "stderr"
class LoggingConfig(BaseModel):
level: str = Field("INFO", min_length=1)
output: LoggingOutput = Field(LoggingOutput.STDOUT)
class PluginPackagesConfig(BaseModel):
renderers: list[str] = Field(default_factory=lambda: ["reporting.exporters"])
transports: list[str] = Field(default_factory=lambda: ["reporting.transports"])
modules: list[str] = Field(default_factory=list)
class PluginConfigRef(BaseModel):
model_config = ConfigDict(extra="allow")
type: str = Field(min_length=1)
def payload(self) -> dict[str, Any]:
return dict(self.model_extra or {})
class ReportOutputConfig(BaseModel):
renderer: PluginConfigRef = Field()
transport: PluginConfigRef = Field()
class ReportsConfig(BaseModel):
outputs: list[ReportOutputConfig] = Field(min_length=1)
class Config(BaseModel):
exclude_categories: list[str] = Field(default_factory=list)
allow_failures_below: AuditSeverity = Field()
strict_mode: bool = Field()
fail_fast: bool = Field(False)
dry_run: bool = Field(False)
max_concurrency: int = Field(4, ge=1)
logging: LoggingConfig = Field(default_factory=LoggingConfig)
plugin_packages: PluginPackagesConfig = Field(default_factory=PluginPackagesConfig)
reports: ReportsConfig = Field()
modules: dict[str, dict[str, Any]] = Field(default_factory=dict)
@dataclass(frozen=True)
class ResolvedPluginConfig:
type: str
config: BaseModel
@dataclass(frozen=True)
class ResolvedReportOutput:
renderer: ResolvedPluginConfig
transport: ResolvedPluginConfig
@dataclass(frozen=True)
class ResolvedConfig:
exclude_categories: list[str]
allow_failures_below: AuditSeverity
strict_mode: bool
fail_fast: bool
dry_run: bool
max_concurrency: int
logging: LoggingConfig
plugin_packages: PluginPackagesConfig
reports: list[ResolvedReportOutput]
modules: dict[str, BaseModel]

15
core/schemas/manifest.py Normal file
View File

@@ -0,0 +1,15 @@
from pydantic import BaseModel, ConfigDict, Field
class ModuleManifest(BaseModel):
model_config = ConfigDict(
extra="forbid",
frozen=True,
)
identifier: str = Field(pattern=r"^[a-z][a-z0-9_-]*$")
name: str = Field(min_length=1)
category: str = Field(min_length=1)
version: str = Field(pattern=r"^\d+\.\d+\.\d+$")
api_version: str = Field("1", pattern=r"^\d+$")
description: str = ""

32
core/schemas/modules.py Normal file
View File

@@ -0,0 +1,32 @@
from collections.abc import Callable
from dataclasses import dataclass
from typing import Any
from pydantic import BaseModel
from core.base import BaseWorker
from core.schemas import ModuleManifest
from reporting.models import PluginRuntimeError
WorkerBuilder = Callable[[BaseModel | dict[str, Any]], BaseWorker]
@dataclass(frozen=True)
class LoadedModule:
package_name: str
manifest: ModuleManifest
build_worker: WorkerBuilder
config_model: type[BaseModel]
@dataclass(frozen=True)
class LoadedWorker:
worker: BaseWorker
manifest: ModuleManifest
config: BaseModel | dict[str, Any]
@dataclass(frozen=True)
class ModuleLoadResult:
loaded_workers: list[LoadedWorker]
errors: list[PluginRuntimeError]

17
core/schemas/results.py Normal file
View File

@@ -0,0 +1,17 @@
from pydantic import BaseModel, Field
from core.schemas.status import AuditSeverity, AuditStatus
class AuditFindings(BaseModel):
name: str = Field()
description: str = Field("")
severity: AuditSeverity = Field()
class AuditResults(BaseModel):
status: AuditStatus = Field()
findings: list[AuditFindings] = Field()
risk_level: float = Field(
description="value from 0.0 to 1.0, indicating how dangerous the auditresults are."
)

14
core/schemas/status.py Normal file
View File

@@ -0,0 +1,14 @@
from enum import StrEnum
class AuditStatus(StrEnum):
PASS = "pass"
SKIPPED = "skipped"
FAIL = "fail"
class AuditSeverity(StrEnum):
CRITICAL = "critical"
HIGH = "high"
MEDIUM = "medium"
LOW = "low"

3
loader.py Normal file
View File

@@ -0,0 +1,3 @@
from core.module_loader import LoadedModule, build_worker, load_module, load_modules
__all__ = ["LoadedModule", "build_worker", "load_module", "load_modules"]

30
main.py Normal file
View File

@@ -0,0 +1,30 @@
import argparse
import sys
from pathlib import Path
from config import load_config
from core.runner import run_audit
def parse_args() -> argparse.Namespace:
cwd = Path.cwd()
parser = argparse.ArgumentParser(
"hydrogen",
description="Lightweight highly extensible security framework for your server.",
)
parser.add_argument("-c", "--config", type=Path, default=cwd / "config.yaml")
parser.add_argument("-p", "--path", type=Path, default=cwd / "modules")
parser.add_argument("--package-prefix", default="modules")
return parser.parse_args()
def main() -> int:
args = parse_args()
config = load_config(args.config)
return run_audit(args.path, config, args.package_prefix)
if __name__ == "__main__":
sys.exit(main())

0
modules/__init__.py Normal file
View File

20
modules/ssh/__init__.py Normal file
View File

@@ -0,0 +1,20 @@
from core.base import BaseWorker
from core.schemas import ModuleManifest
from .config import SSHModuleConfig
from .ssh_security import SSHSecurityWorker
MANIFEST = ModuleManifest(
identifier="ssh",
name="SSH Security Audit",
category="ssh",
version="0.1.0",
api_version="1",
description="Audits OpenSSH server configuration.",
)
CONFIG_MODEL = SSHModuleConfig
def build_worker(config: SSHModuleConfig) -> BaseWorker:
return SSHSecurityWorker(config)

6
modules/ssh/config.py Normal file
View File

@@ -0,0 +1,6 @@
from pydantic import BaseModel, Field
class SSHModuleConfig(BaseModel):
enabled: bool = Field(True)
test_failure: bool = Field(False, alias="test-failure")

229
modules/ssh/ssh_security.py Normal file
View File

@@ -0,0 +1,229 @@
import os
import shlex
import stat
from pathlib import Path
from core.base import BaseWorker
from core.schemas.results import AuditFindings, AuditResults
from core.schemas.status import AuditSeverity, AuditStatus
from .config import SSHModuleConfig
class SSHSecurityWorker(BaseWorker):
config_model = SSHModuleConfig
def __init__(self, config: SSHModuleConfig | None = None) -> None:
super().__init__(config)
self.module_config = config or SSHModuleConfig()
def run(self) -> AuditResults:
if not self.module_config.enabled:
return AuditResults(
status=AuditStatus.SKIPPED,
findings=[],
risk_level=0.0,
)
if self.module_config.test_failure:
findings = [
AuditFindings(
name="im fucking dumbass rragh",
description="im testing!.",
severity=AuditSeverity.CRITICAL,
)
]
return AuditResults(
status=AuditStatus.PASS if not findings else AuditStatus.FAIL,
findings=findings,
risk_level=self._calculate_risk(findings),
)
if os.name != "posix":
return AuditResults(
status=AuditStatus.SKIPPED,
findings=[],
risk_level=0.0,
)
sshd_config = Path("/etc/ssh/sshd_config")
if not sshd_config.exists():
return AuditResults(
status=AuditStatus.SKIPPED,
findings=[
AuditFindings(
name="sshd_config_not_found",
description="SSH server config /etc/ssh/sshd_config was not found.",
severity=AuditSeverity.LOW,
)
],
risk_level=0.1,
)
findings: list[AuditFindings] = []
resolved_config = self._read_merged_config(sshd_config)
findings.extend(self._check_config_directives(resolved_config))
findings.extend(self._check_config_permissions(sshd_config))
findings.extend(self._check_host_key_permissions(Path("/etc/ssh")))
return AuditResults(
status=AuditStatus.PASS if not findings else AuditStatus.FAIL,
findings=findings,
risk_level=self._calculate_risk(findings),
)
def _read_merged_config(self, root_config: Path) -> dict[str, str]:
config_files = [root_config]
include_dir = root_config.parent / "sshd_config.d"
if include_dir.exists() and include_dir.is_dir():
config_files.extend(sorted(include_dir.glob("*.conf")))
directives: dict[str, str] = {}
for config_file in config_files:
for raw_line in config_file.read_text(encoding="utf-8").splitlines():
line = raw_line.strip()
if not line or line.startswith("#"):
continue
line = line.split("#", 1)[0].strip()
if not line:
continue
parts = shlex.split(line, comments=False, posix=True)
if len(parts) < 2: # noqa: PLR2004
continue
directive = parts[0].lower()
value = " ".join(parts[1:])
directives[directive] = value
return directives
def _check_config_directives(self, config: dict[str, str]) -> list[AuditFindings]:
findings: list[AuditFindings] = []
if config.get("permitrootlogin", "prohibit-password").lower() not in {
"no",
"forced-commands-only",
}:
findings.append(
AuditFindings(
name="permit_root_login_enabled",
description="Directive PermitRootLogin should be set to 'no' or 'forced-commands-only'.",
severity=AuditSeverity.CRITICAL,
)
)
if config.get("passwordauthentication", "yes").lower() != "no":
findings.append(
AuditFindings(
name="password_authentication_enabled",
description="Directive PasswordAuthentication should be disabled to prefer key-based access.",
severity=AuditSeverity.HIGH,
)
)
if config.get("pubkeyauthentication", "yes").lower() != "yes":
findings.append(
AuditFindings(
name="pubkey_authentication_disabled",
description="Directive PubkeyAuthentication should be enabled.",
severity=AuditSeverity.HIGH,
)
)
if config.get("x11forwarding", "no").lower() != "no":
findings.append(
AuditFindings(
name="x11_forwarding_enabled",
description="Directive X11Forwarding should usually be disabled on hardened servers.",
severity=AuditSeverity.MEDIUM,
)
)
max_auth_tries = config.get("maxauthtries", "6")
if max_auth_tries.isdigit() and int(max_auth_tries) > 4: # noqa: PLR2004
findings.append(
AuditFindings(
name="max_auth_tries_too_high",
description="Directive MaxAuthTries should be 4 or less.",
severity=AuditSeverity.MEDIUM,
)
)
if config.get("permitemptypasswords", "no").lower() != "no":
findings.append(
AuditFindings(
name="empty_passwords_permitted",
description="Directive PermitEmptyPasswords must be disabled.",
severity=AuditSeverity.CRITICAL,
)
)
return findings
def _check_config_permissions(self, config_path: Path) -> list[AuditFindings]:
findings: list[AuditFindings] = []
file_stat = config_path.stat()
mode = stat.S_IMODE(file_stat.st_mode)
if file_stat.st_uid != 0:
findings.append(
AuditFindings(
name="sshd_config_not_owned_by_root",
description="File /etc/ssh/sshd_config should be owned by root.",
severity=AuditSeverity.HIGH,
)
)
if mode & stat.S_IWGRP or mode & stat.S_IWOTH:
findings.append(
AuditFindings(
name="sshd_config_writable_by_non_root",
description="File /etc/ssh/sshd_config must not be writable by group or others.",
severity=AuditSeverity.HIGH,
)
)
return findings
def _check_host_key_permissions(self, ssh_dir: Path) -> list[AuditFindings]:
findings: list[AuditFindings] = []
for key_path in sorted(ssh_dir.glob("ssh_host_*_key")):
file_stat = key_path.stat()
mode = stat.S_IMODE(file_stat.st_mode)
if file_stat.st_uid != 0:
findings.append(
AuditFindings(
name=f"{key_path.name}_not_owned_by_root",
description=f"Private host key {key_path} should be owned by root.",
severity=AuditSeverity.HIGH,
)
)
if mode & (stat.S_IRWXG | stat.S_IRWXO):
findings.append(
AuditFindings(
name=f"{key_path.name}_permissions_too_open",
description=f"Private host key {key_path} should not be accessible to group or others.",
severity=AuditSeverity.CRITICAL,
)
)
return findings
def _calculate_risk(self, findings: list[AuditFindings]) -> float:
if not findings:
return 0.0
severity_scores = {
AuditSeverity.CRITICAL: 0.45,
AuditSeverity.HIGH: 0.3,
AuditSeverity.MEDIUM: 0.2,
AuditSeverity.LOW: 0.1,
}
risk = sum(severity_scores[finding.severity] for finding in findings)
return min(1.0, risk)

59
pyproject.toml Normal file
View File

@@ -0,0 +1,59 @@
[tool.black]
line-length = 100
target-version = ['py311']
include = '\.pyi?$'
extend-exclude = '''
/(
\.git
| \.venv
| venv
| build
| dist
| alembic
)/
'''
[tool.ruff]
# Базовые настройки
line-length = 100
target-version = "py311"
exclude = [
".git",
".venv",
"venv",
"build",
"dist",
"alembic",
]
[tool.ruff.lint]
select = [
"E",
"W",
"F",
"I",
"N",
"UP",
"B",
"SIM",
"PL",
"RUF",
"TID",
"PT",
]
ignore = [
"E501",
"D100",
"D104",
"G004",
"PLR0913",
"RUF001",
"RUF002",
"RUF003",
"B008"
]
[tool.ruff.lint.isort]
combine-as-imports = true

4
reporting/__init__.py Normal file
View File

@@ -0,0 +1,4 @@
from reporting.bootstrap import build_renderer_registry, build_transport_registry
from reporting.service import ReportService
__all__ = ["ReportService", "build_renderer_registry", "build_transport_registry"]

81
reporting/bootstrap.py Normal file
View File

@@ -0,0 +1,81 @@
from __future__ import annotations
from dataclasses import dataclass
from core.base import BaseRenderer, BaseTransport
from core.plugin_types import PluginGroupConfig, PluginLoadError, discover_plugins
from core.schemas.config import PluginPackagesConfig
from reporting.registry import RendererRegistry
from reporting.transport_registry import TransportRegistry
@dataclass(frozen=True)
class ReportingBootstrapResult:
renderer_registry: RendererRegistry
transport_registry: TransportRegistry
errors: list[PluginLoadError]
def bootstrap_reporting(plugin_packages: PluginPackagesConfig) -> ReportingBootstrapResult:
renderer_descriptors, renderer_errors = discover_plugins(
PluginGroupConfig(
package_names=plugin_packages.renderers,
entry_point_group="hydrogen.renderers",
plugin_kind="renderer",
),
BaseRenderer,
)
transport_descriptors, transport_errors = discover_plugins(
PluginGroupConfig(
package_names=plugin_packages.transports,
entry_point_group="hydrogen.transports",
plugin_kind="transport",
),
BaseTransport,
)
errors = [*renderer_errors, *transport_errors]
renderer_registry = RendererRegistry()
for descriptor in renderer_descriptors:
try:
renderer_registry.register(descriptor.plugin)
except ValueError as exc:
errors.append(
PluginLoadError(
plugin_kind="renderer",
source=descriptor.source,
message=str(exc),
)
)
transport_registry = TransportRegistry()
for descriptor in transport_descriptors:
try:
transport_registry.register(descriptor.plugin)
except ValueError as exc:
errors.append(
PluginLoadError(
plugin_kind="transport",
source=descriptor.source,
message=str(exc),
)
)
return ReportingBootstrapResult(
renderer_registry=renderer_registry,
transport_registry=transport_registry,
errors=errors,
)
def build_renderer_registry(
plugin_packages: PluginPackagesConfig | None = None,
) -> RendererRegistry:
return bootstrap_reporting(plugin_packages or PluginPackagesConfig()).renderer_registry
def build_transport_registry(
plugin_packages: PluginPackagesConfig | None = None,
) -> TransportRegistry:
return bootstrap_reporting(plugin_packages or PluginPackagesConfig()).transport_registry

View File

@@ -0,0 +1,3 @@
from reporting.exporters.json import JsonRenderer
__all__ = ["JsonRenderer"]

View File

@@ -0,0 +1,21 @@
import json
from core.base import BaseRenderer
from core.plugin_types import EmptyPluginConfig
from reporting.models import AuditReport, RenderedReport
class JsonRenderer(BaseRenderer):
content_type = "json"
media_type = "application/json"
file_extension = ".json"
aliases = ("application/json",)
config_model = EmptyPluginConfig
def render(self, report: AuditReport, config: EmptyPluginConfig) -> RenderedReport:
return RenderedReport(
format_name=self.content_type,
media_type=self.media_type,
file_extension=self.file_extension,
content=json.dumps(report.model_dump(mode="json"), ensure_ascii=True, indent=2),
)

40
reporting/models.py Normal file
View File

@@ -0,0 +1,40 @@
from datetime import UTC, datetime
from pydantic import BaseModel, Field
from core.schemas.manifest import ModuleManifest
from core.schemas.results import AuditResults
from core.schemas.status import AuditSeverity
class PluginRuntimeError(BaseModel):
plugin_kind: str
plugin_name: str
stage: str
message: str
class ModuleExecutionStats(BaseModel):
duration_seconds: float = Field(ge=0)
finding_counts: dict[AuditSeverity, int] = Field(default_factory=dict)
class ModuleAuditResult(BaseModel):
module: ModuleManifest
result: AuditResults
stats: ModuleExecutionStats | None = None
error: PluginRuntimeError | None = None
class AuditReport(BaseModel):
generated_at: datetime = Field(default_factory=lambda: datetime.now(UTC))
results: list[ModuleAuditResult]
exit_code: int
plugin_errors: list[PluginRuntimeError] = Field(default_factory=list)
class RenderedReport(BaseModel):
format_name: str
media_type: str
file_extension: str
content: str

32
reporting/registry.py Normal file
View File

@@ -0,0 +1,32 @@
from collections.abc import Iterable
from core.base import BaseRenderer
class RendererRegistry:
def __init__(self, renderers: Iterable[BaseRenderer] | None = None) -> None:
self._renderers: dict[str, BaseRenderer] = {}
for renderer in renderers or ():
self.register(renderer)
def register(self, renderer: BaseRenderer) -> None:
self._register_key(renderer.content_type, renderer)
for alias in renderer.aliases:
self._register_key(alias, renderer)
def _register_key(self, key: str, renderer: BaseRenderer) -> None:
if key in self._renderers:
registered = self._renderers[key]
raise ValueError(
f"Renderer key '{key}' is already registered by {type(registered).__name__}"
)
self._renderers[key] = renderer
def get(self, content_type: str) -> BaseRenderer:
try:
return self._renderers[content_type]
except KeyError as exc:
raise ValueError(
f"Renderer for content type '{content_type}' is not registered"
) from exc

36
reporting/service.py Normal file
View File

@@ -0,0 +1,36 @@
from core.schemas.config import PluginPackagesConfig, ResolvedReportOutput
from reporting.bootstrap import build_renderer_registry, build_transport_registry
from reporting.models import AuditReport, ModuleAuditResult, PluginRuntimeError, RenderedReport
from reporting.registry import RendererRegistry
from reporting.transport_registry import TransportRegistry
class ReportService:
def __init__(
self,
renderer_registry: RendererRegistry | None = None,
transport_registry: TransportRegistry | None = None,
plugin_packages: PluginPackagesConfig | None = None,
) -> None:
self._renderer_registry = renderer_registry or build_renderer_registry(plugin_packages)
self._transport_registry = transport_registry or build_transport_registry(plugin_packages)
def build_report(
self,
results: list[ModuleAuditResult],
exit_code: int,
plugin_errors: list[PluginRuntimeError] | None = None,
) -> AuditReport:
return AuditReport(results=results, exit_code=exit_code, plugin_errors=plugin_errors or [])
def render(self, report: AuditReport, report_config: ResolvedReportOutput) -> RenderedReport:
renderer = self._renderer_registry.get(report_config.renderer.type)
return renderer.render(report, report_config.renderer.config)
def publish(self, report: AuditReport, report_config: ResolvedReportOutput) -> str:
rendered_report = self.render(report, report_config)
transport = self._transport_registry.get(report_config.transport.type)
return transport.publish(rendered_report, report_config.transport)
def publish_many(self, report: AuditReport, outputs: list[ResolvedReportOutput]) -> list[str]:
return [self.publish(report, output) for output in outputs]

View File

@@ -0,0 +1,26 @@
from collections.abc import Iterable
from core.base import BaseTransport
class TransportRegistry:
def __init__(self, transports: Iterable[BaseTransport] | None = None) -> None:
self._transports: dict[str, BaseTransport] = {}
for transport in transports or ():
self.register(transport)
def register(self, transport: BaseTransport) -> None:
if transport.transport_type in self._transports:
registered = self._transports[transport.transport_type]
raise ValueError(
"Transport "
f"'{transport.transport_type}' is already registered by {type(registered).__name__}"
)
self._transports[transport.transport_type] = transport
def get(self, transport_type: str) -> BaseTransport:
try:
return self._transports[transport_type]
except KeyError as exc:
raise ValueError(f"Transport '{transport_type}' is not registered") from exc

View File

@@ -0,0 +1,4 @@
from reporting.transports.file import FileTransport
from reporting.transports.webhook import WebhookTransport
__all__ = ["FileTransport", "WebhookTransport"]

View File

@@ -0,0 +1,25 @@
from pathlib import Path
from pydantic import BaseModel, Field
from core.base import TypedTransport
from reporting.models import RenderedReport
class FileTransportConfig(BaseModel):
path: str = Field()
append_extension: bool = Field(True)
class FileTransport(TypedTransport[FileTransportConfig]):
transport_type = "file"
config_model = FileTransportConfig
def publish_typed(self, rendered_report: RenderedReport, transport: FileTransportConfig) -> str:
destination = Path(transport.path)
if transport.append_extension and not destination.suffix:
destination = destination.with_suffix(rendered_report.file_extension)
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_text(rendered_report.content, encoding="utf-8")
return str(destination)

View File

@@ -0,0 +1,69 @@
import json
from enum import StrEnum
from urllib.request import Request, urlopen
from pydantic import AnyHttpUrl, BaseModel, Field
from core.base import TypedTransport
from reporting.models import RenderedReport
class WebhookMethod(StrEnum):
POST = "POST"
PUT = "PUT"
PATCH = "PATCH"
class WebhookPayloadMode(StrEnum):
RENDERED = "rendered"
ENVELOPE = "envelope"
class WebhookTransportConfig(BaseModel):
url: AnyHttpUrl = Field()
method: WebhookMethod = Field(WebhookMethod.POST)
headers: dict[str, str] = Field(default_factory=dict)
timeout_seconds: float = Field(10.0, gt=0)
payload_mode: WebhookPayloadMode = Field(WebhookPayloadMode.RENDERED)
class WebhookTransport(TypedTransport[WebhookTransportConfig]):
transport_type = "webhook"
config_model = WebhookTransportConfig
def publish_typed(
self, rendered_report: RenderedReport, transport: WebhookTransportConfig
) -> str:
body, headers = self._build_request(rendered_report, transport)
request = Request(
url=str(transport.url),
data=body,
headers=headers,
method=transport.method.value,
)
with urlopen(request, timeout=transport.timeout_seconds) as response:
return f"{response.status} {transport.url}"
def _build_request(
self,
rendered_report: RenderedReport,
transport: WebhookTransportConfig,
) -> tuple[bytes, dict[str, str]]:
headers = dict(transport.headers)
if transport.payload_mode is WebhookPayloadMode.ENVELOPE:
payload = json.dumps(
{
"format": rendered_report.format_name,
"content_type": rendered_report.media_type,
"file_extension": rendered_report.file_extension,
"content": rendered_report.content,
},
ensure_ascii=True,
).encode("utf-8")
headers.setdefault("Content-Type", "application/json")
return payload, headers
headers.setdefault("Content-Type", rendered_report.media_type)
return rendered_report.content.encode("utf-8"), headers

2
requirements.txt Normal file
View File

@@ -0,0 +1,2 @@
pydantic>=2.13.0
pyyaml>=6.0.0