mirror of
https://github.com/fosrl/gerbil.git
synced 2026-09-08 07:01:30 +02:00
Compare commits
104 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
81624509f8 | ||
|
|
0058f0c7ee | ||
|
|
a29075b8bd | ||
|
|
b72f0ab05f | ||
|
|
551eb18e61 | ||
|
|
2eb4090c20 | ||
|
|
2297266a39 | ||
|
|
5d70c76bfc | ||
|
|
c63be17369 | ||
|
|
939fbd5d2c | ||
|
|
94594f40a5 | ||
|
|
e1b47b69aa | ||
|
|
654a3c777a | ||
|
|
6f1851639a | ||
|
|
74629a97f7 | ||
|
|
5c697afcac | ||
|
|
6a26078f9e | ||
|
|
f17612c5d9 | ||
|
|
e25e5766df | ||
|
|
2264c1c7b4 | ||
|
|
904199121d | ||
|
|
5b92bdeb1c | ||
|
|
daecacbd8e | ||
|
|
8efce78d94 | ||
|
|
c29e6ec535 | ||
|
|
0a1496e8de | ||
|
|
4f325dd92f | ||
|
|
8328a4751a | ||
|
|
860ba69a88 | ||
|
|
e1c9836dc1 | ||
|
|
4f457db7b5 | ||
|
|
6d548ea166 | ||
|
|
d611eb2022 | ||
|
|
4d8f5923c3 | ||
|
|
6af41f2a6e | ||
|
|
c638090ef7 | ||
|
|
cde79b5c50 | ||
|
|
af42ae3448 | ||
|
|
f7d874cd78 | ||
|
|
f3f70fff14 | ||
|
|
95e83da5c1 | ||
|
|
19a71e7c0a | ||
|
|
cf863702a0 | ||
|
|
b8bb1249c9 | ||
|
|
18078e0e06 | ||
|
|
44622a883f | ||
|
|
3926dcc6be | ||
|
|
65bb3fc101 | ||
|
|
8683a62b4b | ||
|
|
7d1f963820 | ||
|
|
775def1a97 | ||
|
|
180b5fe768 | ||
|
|
e5241751fc | ||
|
|
34f8abfd37 | ||
|
|
77b386ecac | ||
|
|
1b869c20de | ||
|
|
375cb8b0ba | ||
|
|
3f95e2da25 | ||
|
|
cda6fa6772 | ||
|
|
191b4fa26a | ||
|
|
73d4d4d37c | ||
|
|
bcb5cc4746 | ||
|
|
f130a7cdb8 | ||
|
|
4a322a436b | ||
|
|
b642df3e1e | ||
|
|
65e599f833 | ||
|
|
cae46757dd | ||
|
|
4b575016f3 | ||
|
|
9e5539a5ba | ||
|
|
279fc427a8 | ||
|
|
127ded4203 | ||
|
|
6c91e99497 | ||
|
|
58415dee7e | ||
|
|
c3ed355127 | ||
|
|
c6e1881e6a | ||
|
|
eedd813e2f | ||
|
|
3cf2ccdc54 | ||
|
|
726b6b171c | ||
|
|
037618acbc | ||
|
|
1a6bc81ddd | ||
|
|
a3dbdef7cc | ||
|
|
f07c83fde4 | ||
|
|
652d9c5c68 | ||
|
|
e47a57cb4f | ||
|
|
4357ddf64b | ||
|
|
f322b4c921 | ||
|
|
56f72d6643 | ||
|
|
367e5bfa08 | ||
|
|
aeb8b7c56f | ||
|
|
f5c77d7df8 | ||
|
|
a37aadddb5 | ||
|
|
80747bf98b | ||
|
|
69418a439c | ||
|
|
d065897c4d | ||
|
|
b57574cc4b | ||
|
|
a3862260c9 | ||
|
|
c7d9c72f29 | ||
|
|
abc744c647 | ||
|
|
986a2c6bb6 | ||
|
|
58674ec025 | ||
|
|
5dbe3dbb84 | ||
|
|
32d7af44ca | ||
|
|
fdc398eb9c | ||
|
|
fcd290272f |
1
.github/CODEOWNERS
vendored
Normal file
1
.github/CODEOWNERS
vendored
Normal file
@@ -0,0 +1 @@
|
||||
* @oschwartz10612 @miloschwartz
|
||||
5
.github/ISSUE_TEMPLATE/1.bug_report.yml
vendored
5
.github/ISSUE_TEMPLATE/1.bug_report.yml
vendored
@@ -14,12 +14,13 @@ body:
|
||||
label: Environment
|
||||
description: Please fill out the relevant details below for your environment.
|
||||
value: |
|
||||
- OS Type & Version: (e.g., Ubuntu 22.04)
|
||||
- OS Type & Version:
|
||||
- Pangolin Version:
|
||||
- Edition (Community or Enterprise):
|
||||
- Gerbil Version:
|
||||
- Traefik Version:
|
||||
- Newt Version:
|
||||
- Olm Version: (if applicable)
|
||||
- Client Version:
|
||||
validations:
|
||||
required: true
|
||||
|
||||
|
||||
36
.github/dependabot.yml
vendored
36
.github/dependabot.yml
vendored
@@ -1,40 +1,32 @@
|
||||
version: 2
|
||||
|
||||
updates:
|
||||
- package-ecosystem: "gomod"
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: "daily"
|
||||
open-pull-requests-limit: 1
|
||||
groups:
|
||||
dev-patch-updates:
|
||||
dependency-type: "development"
|
||||
update-types:
|
||||
- "patch"
|
||||
dev-minor-updates:
|
||||
dependency-type: "development"
|
||||
update-types:
|
||||
- "minor"
|
||||
prod-patch-updates:
|
||||
dependency-type: "production"
|
||||
update-types:
|
||||
- "patch"
|
||||
prod-minor-updates:
|
||||
dependency-type: "production"
|
||||
update-types:
|
||||
- "minor"
|
||||
go-dependencies:
|
||||
patterns:
|
||||
- "*"
|
||||
|
||||
- package-ecosystem: "docker"
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: "daily"
|
||||
open-pull-requests-limit: 1
|
||||
groups:
|
||||
patch-updates:
|
||||
update-types:
|
||||
- "patch"
|
||||
minor-updates:
|
||||
update-types:
|
||||
- "minor"
|
||||
docker-dependencies:
|
||||
patterns:
|
||||
- "*"
|
||||
|
||||
- package-ecosystem: "github-actions"
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: "weekly"
|
||||
open-pull-requests-limit: 1
|
||||
groups:
|
||||
github-actions-dependencies:
|
||||
patterns:
|
||||
- "*"
|
||||
|
||||
21
.github/workflows/cicd.yml
vendored
21
.github/workflows/cicd.yml
vendored
@@ -36,16 +36,16 @@ jobs:
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@c7c53464625b32c7a7e944ae62b3e17d2b600130 # v3.7.0
|
||||
uses: docker/setup-qemu-action@96fe6ef7f33517b61c61be40b68a1882f3264fb8 # v4.2.0
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
uses: docker/setup-buildx-action@37fe631027851001ddb9b187196cc803df7f5f0e # v4.3.0
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@5e57cd118135c172c3672efd75eb46360885c0ef # v3.6.0
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: docker.io
|
||||
username: ${{ secrets.DOCKER_HUB_USERNAME }}
|
||||
@@ -57,9 +57,9 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: Install Go
|
||||
uses: actions/setup-go@7a3fe6cf4cb3a834922a1244abfce67bcef6a0c5 # v6.2.0
|
||||
uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0
|
||||
with:
|
||||
go-version: 1.25
|
||||
go-version: 1.26
|
||||
|
||||
- name: Update version in main.go
|
||||
run: |
|
||||
@@ -80,7 +80,7 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: Login in to GHCR
|
||||
uses: docker/login-action@5e57cd118135c172c3672efd75eb46360885c0ef # v3.6.0
|
||||
uses: docker/login-action@dbcb813823bdd20940b903addbd779551569679f # v4.6.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.actor }}
|
||||
@@ -107,8 +107,9 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: Install cosign
|
||||
# cosign is used to sign and verify container images (key and keyless)
|
||||
uses: sigstore/cosign-installer@faadad0cce49287aee09b3a48701e75088a2c6ad # v4.0.0
|
||||
uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 # v4.1.2
|
||||
with:
|
||||
cosign-release: v3.0.6
|
||||
|
||||
- name: Dual-sign and verify (GHCR & Docker Hub)
|
||||
# Sign each image by digest using keyless (OIDC) and key-based signing,
|
||||
@@ -155,7 +156,7 @@ jobs:
|
||||
shell: bash
|
||||
|
||||
- name: Upload artifacts from /bin
|
||||
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6.0.0
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: binaries
|
||||
path: bin/
|
||||
|
||||
2
.github/workflows/mirror.yaml
vendored
2
.github/workflows/mirror.yaml
vendored
@@ -23,7 +23,7 @@ jobs:
|
||||
skopeo --version
|
||||
|
||||
- name: Install cosign
|
||||
uses: sigstore/cosign-installer@faadad0cce49287aee09b3a48701e75088a2c6ad # v4.0.0
|
||||
uses: sigstore/cosign-installer@6f9f17788090df1f26f669e9d70d6ae9567deba6 # v4.1.2
|
||||
|
||||
- name: Input check
|
||||
run: |
|
||||
|
||||
10
.github/workflows/test.yml
vendored
10
.github/workflows/test.yml
vendored
@@ -14,12 +14,16 @@ jobs:
|
||||
runs-on: amd64-runner
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@8e8c483db84b4bee98b60c0593521ed34d9990e8 # v6.0.1
|
||||
- uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
|
||||
|
||||
- name: Read go version
|
||||
id: goversion
|
||||
run: echo "version=$(cat .go-version)" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@7a3fe6cf4cb3a834922a1244abfce67bcef6a0c5 # v6.2.0
|
||||
uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0
|
||||
with:
|
||||
go-version: 1.25
|
||||
go-version: ${{ steps.goversion.outputs.version }}
|
||||
|
||||
- name: Build go
|
||||
run: go build
|
||||
|
||||
@@ -1 +1 @@
|
||||
1.25
|
||||
1.26
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
FROM golang:1.25-alpine AS builder
|
||||
ARG GO_VERSION=1.26
|
||||
FROM golang:${GO_VERSION}-alpine AS builder
|
||||
|
||||
# Set the working directory inside the container
|
||||
WORKDIR /app
|
||||
@@ -16,7 +17,7 @@ COPY . .
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -o /gerbil
|
||||
|
||||
# Start a new stage from scratch
|
||||
FROM alpine:3.23 AS runner
|
||||
FROM alpine:3.24 AS runner
|
||||
|
||||
RUN apk add --no-cache iptables iproute2
|
||||
|
||||
@@ -25,4 +26,4 @@ COPY entrypoint.sh /
|
||||
|
||||
RUN chmod +x /entrypoint.sh
|
||||
ENTRYPOINT ["/entrypoint.sh"]
|
||||
CMD ["gerbil"]
|
||||
CMD ["gerbil"]
|
||||
|
||||
6
Makefile
6
Makefile
@@ -1,4 +1,6 @@
|
||||
|
||||
GO_VERSION := $(shell cat .go-version 2>/dev/null)
|
||||
|
||||
all: build push
|
||||
|
||||
docker-build-release:
|
||||
@@ -7,10 +9,10 @@ docker-build-release:
|
||||
exit 1; \
|
||||
fi
|
||||
docker buildx build --platform linux/arm64,linux/amd64 -t fosrl/gerbil:latest -f Dockerfile --push .
|
||||
docker buildx build --platform linux/arm64,linux/amd64 -t fosrl/gerbil:$(tag) -f Dockerfile --push .
|
||||
docker buildx build --platform linux/arm64,linux/amd64 -t fosrl/gerbil:$(tag) -f Dockerfile --build-arg GO_VERSION=$(GO_VERSION) --push .
|
||||
|
||||
build:
|
||||
docker build -t fosrl/gerbil:latest .
|
||||
docker build -t fosrl/gerbil:latest --build-arg GO_VERSION=$(GO_VERSION) .
|
||||
|
||||
push:
|
||||
docker push fosrl/gerbil:latest
|
||||
|
||||
20
README.md
20
README.md
@@ -40,6 +40,24 @@ The PROXY protocol allows downstream proxies to know the real client IP address
|
||||
|
||||
In single node (self hosted) Pangolin deployments this can be bypassed by using port 443:443 to route to Traefik instead of the SNI proxy at 8443.
|
||||
|
||||
### Observability with OpenTelemetry
|
||||
|
||||
Gerbil includes comprehensive OpenTelemetry metrics instrumentation for monitoring and observability. Metrics can be exported via:
|
||||
|
||||
- **Prometheus**: Pull-based metrics at the `/metrics` endpoint (enabled by default)
|
||||
- **OTLP**: Push-based metrics to any OpenTelemetry-compatible collector
|
||||
|
||||
Key metrics include:
|
||||
|
||||
- WireGuard interface and peer status
|
||||
- Bandwidth usage per peer
|
||||
- Active relay sessions and proxy connections
|
||||
- Handshake success/failure rates
|
||||
- Route lookup cache hit/miss ratios
|
||||
- Go runtime metrics (GC, goroutines, memory)
|
||||
|
||||
See [docs/observability.md](docs/observability.md) for complete documentation, metrics reference, and examples.
|
||||
|
||||
## CLI Args
|
||||
|
||||
Important:
|
||||
@@ -120,7 +138,7 @@ make
|
||||
|
||||
### Binary
|
||||
|
||||
Make sure to have Go 1.23.1 installed.
|
||||
Make sure to have Go 1.26 installed.
|
||||
|
||||
```bash
|
||||
make local
|
||||
|
||||
273
docs/observability.md
Normal file
273
docs/observability.md
Normal file
@@ -0,0 +1,273 @@
|
||||
<!-- markdownlint-disable MD036 MD060 -->
|
||||
# Gerbil Observability Architecture
|
||||
|
||||
This document describes the metrics subsystem for Gerbil, explains the design
|
||||
decisions, and shows how to configure each backend.
|
||||
|
||||
---
|
||||
|
||||
## Architecture Overview
|
||||
|
||||
Gerbil's metrics subsystem uses a **pluggable backend** design:
|
||||
|
||||
```text
|
||||
main.go ─── internal/metrics ─── internal/observability ─── backend
|
||||
(facade) (interface) Prometheus
|
||||
OR OTel/OTLP
|
||||
OR Noop (disabled)
|
||||
```
|
||||
|
||||
Application code (main, relay, proxy) calls only the `metrics.Record*`
|
||||
functions in `internal/metrics`. That package delegates to whichever backend
|
||||
was selected at startup via `internal/observability.Backend`.
|
||||
|
||||
### Why Prometheus-native and OTel are mutually exclusive
|
||||
|
||||
**Exactly one** metrics backend may be active at runtime:
|
||||
|
||||
| Mode | What happens |
|
||||
|------|-------------|
|
||||
| `prometheus` | Native Prometheus client registers metrics on a dedicated registry and exposes `/metrics`. No OTel SDK is initialised. |
|
||||
| `otel` | OTel SDK pushes metrics via OTLP/gRPC or OTLP/HTTP to an external collector. No `/metrics` endpoint is exposed. |
|
||||
| `none` | A safe noop backend is used. All `Record*` calls are discarded. |
|
||||
|
||||
Running both simultaneously would mean every metric is recorded twice through
|
||||
two different code paths, with differing semantics (pull vs. push, different
|
||||
naming rules, different cardinality handling). The design enforces a single
|
||||
source of truth.
|
||||
|
||||
### Future OTel tracing and logging
|
||||
|
||||
The `internal/observability/otel/` package is designed so that tracing and
|
||||
logging support can be added **beside** the existing metrics code without
|
||||
touching the Prometheus-native path:
|
||||
|
||||
```bash
|
||||
internal/observability/otel/
|
||||
backend.go ← metrics
|
||||
exporter.go ← OTLP exporter creation
|
||||
resource.go ← OTel resource
|
||||
trace.go ← future: TracerProvider setup
|
||||
log.go ← future: LoggerProvider setup
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Configuration
|
||||
|
||||
### Config precedence
|
||||
|
||||
1. CLI flags (highest priority)
|
||||
2. Environment variables
|
||||
3. Defaults
|
||||
|
||||
### Config struct
|
||||
|
||||
```go
|
||||
type MetricsConfig struct {
|
||||
Enabled bool
|
||||
Backend string // "prometheus" | "otel" | "none"
|
||||
Prometheus PrometheusConfig
|
||||
OTel OTelConfig
|
||||
ServiceName string
|
||||
ServiceVersion string
|
||||
DeploymentEnvironment string
|
||||
}
|
||||
|
||||
type PrometheusConfig struct {
|
||||
Path string // default: "/metrics"
|
||||
}
|
||||
|
||||
type OTelConfig struct {
|
||||
Protocol string // "grpc" (default) or "http"
|
||||
Endpoint string // default: "localhost:4317"
|
||||
Insecure bool // default: true
|
||||
ExportInterval time.Duration // default: 60s
|
||||
Timeout time.Duration // default: 10s
|
||||
}
|
||||
```
|
||||
|
||||
### Environment variables
|
||||
|
||||
| Variable | Default | Description |
|
||||
|----------|---------|-------------|
|
||||
| `METRICS_ENABLED` | `true` | Enable/disable metrics |
|
||||
| `METRICS_BACKEND` | `prometheus` | Backend: `prometheus`, `otel`, or `none` |
|
||||
| `METRICS_PATH` | `/metrics` | HTTP path for Prometheus endpoint |
|
||||
| `OTEL_METRICS_PROTOCOL` | `grpc` | OTLP transport: `grpc` or `http` |
|
||||
| `OTEL_METRICS_ENDPOINT` | `localhost:4317` | OTLP collector address |
|
||||
| `OTEL_METRICS_INSECURE` | `true` | Disable TLS for OTLP |
|
||||
| `OTEL_METRICS_EXPORT_INTERVAL` | `60s` | Push interval (e.g. `10s`, `1m`) |
|
||||
| `OTEL_METRICS_TIMEOUT` | `10s` | Timeout for OTLP exporter connection setup |
|
||||
| `DEPLOYMENT_ENVIRONMENT` | _(unset)_ | OTel deployment.environment attribute |
|
||||
|
||||
### CLI flags
|
||||
|
||||
```bash
|
||||
--metrics-enabled bool (default: true)
|
||||
--metrics-backend string (default: prometheus)
|
||||
--metrics-path string (default: /metrics)
|
||||
--otel-metrics-protocol string (default: grpc)
|
||||
--otel-metrics-endpoint string (default: localhost:4317)
|
||||
--otel-metrics-insecure bool (default: true)
|
||||
--otel-metrics-export-interval duration (default: 60s)
|
||||
--otel-metrics-timeout duration (default: 10s)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## When to choose each backend
|
||||
|
||||
| Criterion | Prometheus | OTel/OTLP |
|
||||
|-----------|-----------|-----------|
|
||||
| Existing Prometheus/Grafana stack | ✅ | |
|
||||
| Pull-based scraping | ✅ | |
|
||||
| No external collector required | ✅ | |
|
||||
| Vendor-neutral telemetry | | ✅ |
|
||||
| Push-based export | | ✅ |
|
||||
| Grafana Cloud / managed OTLP | | ✅ |
|
||||
| Future traces + logs via same pipeline | | ✅ |
|
||||
|
||||
---
|
||||
|
||||
## Enabling Prometheus-native mode
|
||||
|
||||
### Environment variables
|
||||
|
||||
```bash
|
||||
METRICS_ENABLED=true
|
||||
METRICS_BACKEND=prometheus
|
||||
METRICS_PATH=/metrics
|
||||
```
|
||||
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
./gerbil --metrics-enabled --metrics-backend=prometheus --metrics-path=/metrics \
|
||||
--config=/etc/gerbil/config.json
|
||||
```
|
||||
|
||||
The metrics config is supplied separately via env/flags; it is not embedded
|
||||
in the WireGuard config file.
|
||||
|
||||
The Prometheus `/metrics` endpoint is registered only when
|
||||
`--metrics-backend=prometheus`. All gerbil_* metrics plus Go runtime metrics
|
||||
are available.
|
||||
|
||||
---
|
||||
|
||||
## Enabling OTel mode
|
||||
|
||||
### Environment variables
|
||||
|
||||
```bash
|
||||
export METRICS_ENABLED=true
|
||||
export METRICS_BACKEND=otel
|
||||
export OTEL_METRICS_PROTOCOL=grpc
|
||||
export OTEL_METRICS_ENDPOINT=otel-collector:4317
|
||||
export OTEL_METRICS_INSECURE=true
|
||||
export OTEL_METRICS_EXPORT_INTERVAL=10s
|
||||
export OTEL_METRICS_TIMEOUT=10s
|
||||
export DEPLOYMENT_ENVIRONMENT=production
|
||||
```
|
||||
|
||||
### CLI
|
||||
|
||||
```bash
|
||||
./gerbil --metrics-enabled \
|
||||
--metrics-backend=otel \
|
||||
--otel-metrics-protocol=grpc \
|
||||
--otel-metrics-endpoint=otel-collector:4317 \
|
||||
--otel-metrics-insecure \
|
||||
--otel-metrics-export-interval=10s \
|
||||
--otel-metrics-timeout=10s \
|
||||
--config=/etc/gerbil/config.json
|
||||
```
|
||||
|
||||
### HTTP mode (OTLP/HTTP)
|
||||
|
||||
```bash
|
||||
export OTEL_METRICS_PROTOCOL=http
|
||||
export OTEL_METRICS_ENDPOINT=otel-collector:4318
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Disabling metrics
|
||||
|
||||
```bash
|
||||
export METRICS_ENABLED=false
|
||||
# or
|
||||
./gerbil --metrics-enabled=false
|
||||
# or
|
||||
./gerbil --metrics-backend=none
|
||||
```
|
||||
|
||||
When disabled, all `Record*` calls are directed to a safe noop backend that
|
||||
discards observations without allocating or locking.
|
||||
|
||||
---
|
||||
|
||||
## Metric catalog
|
||||
|
||||
All metrics use the prefix `gerbil_<component>_<name>`.
|
||||
|
||||
### WireGuard metrics
|
||||
|
||||
| Metric | Type | Labels | Description |
|
||||
|--------|------|--------|-------------|
|
||||
| `gerbil_wg_interface_up` | Gauge | `ifname`, `instance` | 1=up, 0=down |
|
||||
| `gerbil_wg_peers_total` | UpDownCounter | `ifname` | Configured peers |
|
||||
| `gerbil_wg_peer_connected` | Gauge | `ifname`, `peer` | 1=connected, 0=disconnected |
|
||||
| `gerbil_wg_bytes_received_total` | Counter | `ifname`, `peer` | Bytes received |
|
||||
| `gerbil_wg_bytes_transmitted_total` | Counter | `ifname`, `peer` | Bytes transmitted |
|
||||
| `gerbil_wg_handshakes_total` | Counter | `ifname`, `peer`, `result` | Handshake attempts |
|
||||
| `gerbil_wg_handshake_latency_seconds` | Histogram | `ifname`, `peer` | Handshake duration |
|
||||
| `gerbil_wg_peer_rtt_seconds` | Histogram | `ifname`, `peer` | Peer round-trip time |
|
||||
|
||||
### Relay metrics
|
||||
|
||||
| Metric | Type | Labels |
|
||||
|--------|------|--------|
|
||||
| `gerbil_proxy_mapping_active` | UpDownCounter | `ifname` |
|
||||
| `gerbil_active_sessions` | UpDownCounter | `ifname` |
|
||||
| `gerbil_udp_packets_total` | Counter | `ifname`, `type`, `direction` |
|
||||
| `gerbil_hole_punch_events_total` | Counter | `ifname`, `result` |
|
||||
|
||||
### SNI proxy metrics
|
||||
|
||||
| Metric | Type | Labels |
|
||||
|--------|------|--------|
|
||||
| `gerbil_sni_connections_total` | Counter | `result` |
|
||||
| `gerbil_sni_active_connections` | UpDownCounter | _(none)_ |
|
||||
| `gerbil_sni_route_cache_hits_total` | Counter | `result` |
|
||||
| `gerbil_sni_route_api_requests_total` | Counter | `result` |
|
||||
| `gerbil_proxy_route_lookups_total` | Counter | `result`, `hostname` |
|
||||
|
||||
### HTTP metrics
|
||||
|
||||
| Metric | Type | Labels |
|
||||
|--------|------|--------|
|
||||
| `gerbil_http_requests_total` | Counter | `endpoint`, `method`, `status_code` |
|
||||
| `gerbil_http_request_duration_seconds` | Histogram | `endpoint`, `method` |
|
||||
|
||||
---
|
||||
|
||||
## Using Docker Compose
|
||||
|
||||
The `docker-compose.metrics.yml` provides a complete observability stack.
|
||||
|
||||
**Prometheus mode:**
|
||||
|
||||
```bash
|
||||
METRICS_BACKEND=prometheus docker-compose -f docker compose.metrics.yml up -d
|
||||
# Scrape at http://localhost:3003/metrics
|
||||
# Grafana at http://localhost:3000 (admin/admin)
|
||||
```
|
||||
|
||||
**OTel mode:**
|
||||
|
||||
```bash
|
||||
METRICS_BACKEND=otel OTEL_METRICS_ENDPOINT=otel-collector:4317 \
|
||||
docker compose -f docker-compose.metrics.yml up -d
|
||||
```
|
||||
47
examples/otel-collector-config.yaml
Normal file
47
examples/otel-collector-config.yaml
Normal file
@@ -0,0 +1,47 @@
|
||||
file_format: '1.0'
|
||||
receivers:
|
||||
otlp:
|
||||
protocols:
|
||||
grpc:
|
||||
endpoint: 0.0.0.0:4317
|
||||
http:
|
||||
endpoint: 0.0.0.0:4318
|
||||
|
||||
processors:
|
||||
batch:
|
||||
timeout: 10s
|
||||
send_batch_size: 1024
|
||||
|
||||
# Add resource attributes
|
||||
resource:
|
||||
attributes:
|
||||
- key: service.environment
|
||||
value: "development"
|
||||
action: insert
|
||||
|
||||
exporters:
|
||||
# Prometheus exporter for scraping
|
||||
prometheus:
|
||||
endpoint: "0.0.0.0:8889"
|
||||
namespace: "gerbil"
|
||||
send_timestamps: true
|
||||
metric_expiration: 5m
|
||||
resource_to_telemetry_conversion:
|
||||
enabled: true
|
||||
|
||||
# Prometheus remote write (optional)
|
||||
prometheusremotewrite:
|
||||
endpoint: "http://prometheus:9090/api/v1/write"
|
||||
tls:
|
||||
insecure: true
|
||||
|
||||
# Debug exporter for debugging
|
||||
debug:
|
||||
verbosity: normal
|
||||
|
||||
service:
|
||||
pipelines:
|
||||
metrics:
|
||||
receivers: [otlp]
|
||||
processors: [batch, resource]
|
||||
exporters: [prometheus, prometheusremotewrite, debug]
|
||||
24
examples/prometheus.yml
Normal file
24
examples/prometheus.yml
Normal file
@@ -0,0 +1,24 @@
|
||||
global:
|
||||
scrape_interval: 15s
|
||||
evaluation_interval: 15s
|
||||
external_labels:
|
||||
cluster: 'gerbil-dev'
|
||||
|
||||
scrape_configs:
|
||||
# Scrape Gerbil's /metrics endpoint directly
|
||||
- job_name: 'gerbil'
|
||||
static_configs:
|
||||
- targets: ['gerbil:3003']
|
||||
labels:
|
||||
service: 'gerbil'
|
||||
environment: 'development'
|
||||
|
||||
# Scrape OpenTelemetry Collector metrics
|
||||
- job_name: 'otel-collector'
|
||||
static_configs:
|
||||
- targets: ['otel-collector:8888']
|
||||
labels:
|
||||
service: 'otel-collector'
|
||||
- targets: ['otel-collector:8889']
|
||||
labels:
|
||||
service: 'otel-collector-prometheus-exporter'
|
||||
38
go.mod
38
go.mod
@@ -1,23 +1,49 @@
|
||||
module github.com/fosrl/gerbil
|
||||
|
||||
go 1.25
|
||||
go 1.26.0
|
||||
|
||||
require (
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
github.com/vishvananda/netlink v1.3.1
|
||||
golang.org/x/crypto v0.46.0
|
||||
golang.org/x/sync v0.1.0
|
||||
go.opentelemetry.io/otel v1.46.0
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.46.0
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.46.0
|
||||
go.opentelemetry.io/otel/metric v1.46.0
|
||||
go.opentelemetry.io/otel/sdk v1.46.0
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0
|
||||
golang.org/x/crypto v0.55.0
|
||||
golang.org/x/sync v0.22.0
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20230429144221-925a1e7659e6
|
||||
)
|
||||
|
||||
require (
|
||||
github.com/google/go-cmp v0.5.9 // indirect
|
||||
github.com/beorn7/perks v1.0.1 // indirect
|
||||
github.com/cenkalti/backoff/v5 v5.0.3 // indirect
|
||||
github.com/cespare/xxhash/v2 v2.3.0 // indirect
|
||||
github.com/go-logr/logr v1.4.4 // indirect
|
||||
github.com/go-logr/stdr v1.2.2 // indirect
|
||||
github.com/google/go-cmp v0.7.0 // indirect
|
||||
github.com/google/uuid v1.6.0 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0 // indirect
|
||||
github.com/josharian/native v1.1.0 // indirect
|
||||
github.com/mdlayher/genetlink v1.3.2 // indirect
|
||||
github.com/mdlayher/netlink v1.7.2 // indirect
|
||||
github.com/mdlayher/socket v0.4.1 // indirect
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 // indirect
|
||||
github.com/prometheus/client_model v0.6.2 // indirect
|
||||
github.com/prometheus/common v0.70.1 // indirect
|
||||
github.com/prometheus/procfs v0.21.1 // indirect
|
||||
github.com/vishvananda/netns v0.0.5 // indirect
|
||||
golang.org/x/net v0.47.0 // indirect
|
||||
golang.org/x/sys v0.39.0 // indirect
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.46.0 // indirect
|
||||
go.opentelemetry.io/proto/otlp v1.11.0 // indirect
|
||||
golang.org/x/net v0.58.0 // indirect
|
||||
golang.org/x/sys v0.47.0 // indirect
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
golang.zx2c4.com/wireguard v0.0.0-20230325221338-052af4a8072b // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
google.golang.org/grpc v1.83.1 // indirect
|
||||
google.golang.org/protobuf v1.36.12 // indirect
|
||||
)
|
||||
|
||||
91
go.sum
91
go.sum
@@ -1,7 +1,28 @@
|
||||
github.com/google/go-cmp v0.5.9 h1:O2Tfq5qg4qc4AmwVlvv0oLiVAGB7enBSJ2x2DqQFi38=
|
||||
github.com/google/go-cmp v0.5.9/go.mod h1:17dUlkBOakJ0+DkrSSNjCkIjxS6bF9zb3elmeNGIjoY=
|
||||
github.com/beorn7/perks v1.0.1 h1:VlbKKnNfV8bJzeqoa4cOKqO6bYr3WgKZxO8Z16+hsOM=
|
||||
github.com/beorn7/perks v1.0.1/go.mod h1:G2ZrVWU2WbWT9wwq4/hrbKbnv/1ERSJQ0ibhJ6rlkpw=
|
||||
github.com/cenkalti/backoff/v5 v5.0.3 h1:ZN+IMa753KfX5hd8vVaMixjnqRZ3y8CuJKRKj1xcsSM=
|
||||
github.com/cenkalti/backoff/v5 v5.0.3/go.mod h1:rkhZdG3JZukswDf7f0cwqPNk4K0sa+F97BxZthm/crw=
|
||||
github.com/cespare/xxhash/v2 v2.3.0 h1:UL815xU9SqsFlibzuggzjXhog7bL6oX9BbNZnL2UFvs=
|
||||
github.com/cespare/xxhash/v2 v2.3.0/go.mod h1:VGX0DQ3Q6kWi7AoAeZDth3/j3BFtOZR5XLFGgcrjCOs=
|
||||
github.com/go-logr/logr v1.2.2/go.mod h1:jdQByPbusPIv2/zmleS9BjJVeZ6kBagPoEUsqbVz/1A=
|
||||
github.com/go-logr/logr v1.4.4 h1:tG4xh9yMsRCAiodLVTxyrkzSZ9+o0L1Kg/+cPVcbP/8=
|
||||
github.com/go-logr/logr v1.4.4/go.mod h1:9T104GzyrTigFIr8wt5mBrctHMim0Nb2HLGrmQ40KvY=
|
||||
github.com/go-logr/stdr v1.2.2 h1:hSWxHoqTgW2S2qGc0LTAI563KZ5YKYRhT3MFKZMbjag=
|
||||
github.com/go-logr/stdr v1.2.2/go.mod h1:mMo/vtBO5dYbehREoey6XUKy/eSumjCCveDpRre4VKE=
|
||||
github.com/golang/protobuf v1.5.4 h1:i7eJL8qZTpSEXOPTxNKhASYpMn+8e5Q6AdndVa1dWek=
|
||||
github.com/golang/protobuf v1.5.4/go.mod h1:lnTiLA8Wa4RWRcIUkrtSVa5nRhsEGBg48fD6rSs7xps=
|
||||
github.com/google/go-cmp v0.7.0 h1:wk8382ETsv4JYUZwIsn6YpYiWiBsYLSJiTsyBybVuN8=
|
||||
github.com/google/go-cmp v0.7.0/go.mod h1:pXiqmnSA92OHEEa9HXL2W4E7lf9JzCmGVUdgjX3N/iU=
|
||||
github.com/google/uuid v1.6.0 h1:NIvaJDMOsjHA8n1jAhLSgzrAzy1Hgr+hNrb57e+94F0=
|
||||
github.com/google/uuid v1.6.0/go.mod h1:TIyPZe4MgqvfeYDBFedMoGGpEw/LqOeaOT+nhxU+yHo=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0 h1:/Tnpcb2E0Pz/tN9s3bfEY2Q8ePCEX9iuS+cneUwncnw=
|
||||
github.com/grpc-ecosystem/grpc-gateway/v2 v2.30.0/go.mod h1:zOBXOsUaBSjKgmH4OGzV1esUpR3oUSCPYVd2cUBjKYY=
|
||||
github.com/josharian/native v1.1.0 h1:uuaP0hAbW7Y4l0ZRQ6C9zfb7Mg1mbFKry/xzDAfmtLA=
|
||||
github.com/josharian/native v1.1.0/go.mod h1:7X/raswPFr05uY3HiLlYeyQntB6OO7E/d2Cu7qoaN2w=
|
||||
github.com/klauspost/compress v1.19.1 h1:VsB4HPswih7mmZ8WleSFQ75c/Ui1M4trX5oAsJnhSlk=
|
||||
github.com/klauspost/compress v1.19.1/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||
github.com/kylelemons/godebug v1.1.0 h1:RPNrshWIDI6G2gRW9EHilWtl7Z6Sb1BR0xunSBf0SNc=
|
||||
github.com/kylelemons/godebug v1.1.0/go.mod h1:9/0rRGxNHcop5bhtWyNeEfOS8JIWk580+fNqagV/RAw=
|
||||
github.com/mdlayher/genetlink v1.3.2 h1:KdrNKe+CTu+IbZnm/GVUMXSqBBLqcGpRDa0xkQy56gw=
|
||||
github.com/mdlayher/genetlink v1.3.2/go.mod h1:tcC3pkCrPUGIKKsCsp0B3AdaaKuHtaxoJRz3cc+528o=
|
||||
github.com/mdlayher/netlink v1.7.2 h1:/UtM3ofJap7Vl4QWCPDGXY8d3GIY2UGSDbK+QWmY8/g=
|
||||
@@ -10,23 +31,73 @@ github.com/mdlayher/socket v0.4.1 h1:eM9y2/jlbs1M615oshPQOHZzj6R6wMT7bX5NPiQvn2U
|
||||
github.com/mdlayher/socket v0.4.1/go.mod h1:cAqeGjoufqdxWkD7DkpyS+wcefOtmu5OQ8KuoJGIReA=
|
||||
github.com/mikioh/ipaddr v0.0.0-20190404000644-d465c8ab6721 h1:RlZweED6sbSArvlE924+mUcZuXKLBHA35U7LN621Bws=
|
||||
github.com/mikioh/ipaddr v0.0.0-20190404000644-d465c8ab6721/go.mod h1:Ickgr2WtCLZ2MDGd4Gr0geeCH5HybhRJbonOgQpvSxc=
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc=
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible/go.mod h1:3Qf8kWWT7OJRJbdiICTKqZju1ZixQ/KpMGzzAfe6+WQ=
|
||||
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
|
||||
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
|
||||
github.com/prometheus/client_model v0.6.2 h1:oBsgwpGs7iVziMvrGhE53c/GrLUsZdHnqNwqPLxwZyk=
|
||||
github.com/prometheus/client_model v0.6.2/go.mod h1:y3m2F6Gdpfy6Ut/GBsUqTWZqCUvMVzSfMLjcu6wAwpE=
|
||||
github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
|
||||
github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
|
||||
github.com/prometheus/procfs v0.21.1 h1:GljZCt+zSTS+NZq88cyQ1LjZ+RCHp3uVuabBWA5+OJI=
|
||||
github.com/prometheus/procfs v0.21.1/go.mod h1:aB55Cww9pdSJVHk0hUf0inxWyyjPogFIjmHKYgMKmtY=
|
||||
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
|
||||
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
|
||||
github.com/vishvananda/netlink v1.3.1 h1:3AEMt62VKqz90r0tmNhog0r/PpWKmrEShJU0wJW6bV0=
|
||||
github.com/vishvananda/netlink v1.3.1/go.mod h1:ARtKouGSTGchR8aMwmkzC0qiNPrrWO5JS/XMVl45+b4=
|
||||
github.com/vishvananda/netns v0.0.5 h1:DfiHV+j8bA32MFM7bfEunvT8IAqQ/NzSJHtcmW5zdEY=
|
||||
github.com/vishvananda/netns v0.0.5/go.mod h1:SpkAiCQRtJ6TvvxPnOSyH3BMl6unz3xZlaprSwhNNJM=
|
||||
golang.org/x/crypto v0.46.0 h1:cKRW/pmt1pKAfetfu+RCEvjvZkA9RimPbh7bhFjGVBU=
|
||||
golang.org/x/crypto v0.46.0/go.mod h1:Evb/oLKmMraqjZ2iQTwDwvCtJkczlDuTmdJXoZVzqU0=
|
||||
golang.org/x/net v0.47.0 h1:Mx+4dIFzqraBXUugkia1OOvlD6LemFo1ALMHjrXDOhY=
|
||||
golang.org/x/net v0.47.0/go.mod h1:/jNxtkgq5yWUGYkaZGqo27cfGZ1c5Nen03aYrrKpVRU=
|
||||
golang.org/x/sync v0.1.0 h1:wsuoTGHzEhffawBOhz5CYhcrV4IdKZbEyZjBMuTp12o=
|
||||
golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
|
||||
go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc=
|
||||
go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.46.0 h1:qkDYCAFiZXLcs1L4aY+tP2wguQ4kURANqHOQMA2et2s=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.46.0/go.mod h1:tkipS4DRzmpAmvg+Gw4++O1IdDq6TVDnvnYU6cmbQVs=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.46.0 h1:AP23h/mFgb/lc7tdck1Kfn9qxsM8TAeNPCU5C3pzaps=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.46.0/go.mod h1:K4EqCe1b4kGk5WR690ntg9LaBfsPoV32FwthbyoptuA=
|
||||
go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8=
|
||||
go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o=
|
||||
go.opentelemetry.io/otel/metric/x v0.68.0 h1:TA/cBT23D3MnxYPwHL7YFOdYGdx0A0v+s7Mzotpd1dU=
|
||||
go.opentelemetry.io/otel/metric/x v0.68.0/go.mod h1:agudOmvWhwUTjgibWDzxD2PoWYnpw5Ht5jISYOD2Hd4=
|
||||
go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI=
|
||||
go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4=
|
||||
go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c=
|
||||
go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI=
|
||||
go.opentelemetry.io/proto/otlp v1.11.0 h1:5rrYs0Ykyj50sdU/JU0x8etU+LubXWb+gED6TbEdMIk=
|
||||
go.opentelemetry.io/proto/otlp v1.11.0/go.mod h1:SmVizdCOAm3XBtG1g1NnOdhW6jtddT72hLMhv8VwA8E=
|
||||
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
|
||||
go.uber.org/goleak v1.3.0/go.mod h1:CoHD4mav9JJNrW/WLlf7HGZPjdw8EucARQHekz1X6bE=
|
||||
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
|
||||
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
|
||||
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
|
||||
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
|
||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||
golang.org/x/net v0.58.0 h1:ynWG7rqYi4ccpTEuPZ2QGWHktVEM9DMCj9yzDE0Q7To=
|
||||
golang.org/x/net v0.58.0/go.mod h1:YwCddHnFlT7eLQqVprV19OnhLGtc5xOKgE0RyqgfWAU=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sys v0.2.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.10.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.39.0 h1:CvCKL8MeisomCi6qNZ+wbb0DN9E5AATixKsvNtMoMFk=
|
||||
golang.org/x/sys v0.39.0/go.mod h1:OgkHotnGiDImocRcuBABYBEXf8A9a87e/uXjp9XT3ks=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
|
||||
golang.zx2c4.com/wireguard v0.0.0-20230325221338-052af4a8072b h1:J1CaxgLerRR5lgx3wnr6L04cJFbWoceSK9JWBdglINo=
|
||||
golang.zx2c4.com/wireguard v0.0.0-20230325221338-052af4a8072b/go.mod h1:tqur9LnfstdR9ep2LaJT4lFUl0EjlHtge+gAjmsHUG4=
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20230429144221-925a1e7659e6 h1:CawjfCvYQH2OU3/TnxLx97WDSUDRABfT18pCOYwc2GE=
|
||||
golang.zx2c4.com/wireguard/wgctrl v0.0.0-20230429144221-925a1e7659e6/go.mod h1:3rxYc4HtVcSG9gVaTs2GEBdehh+sYPOwKtyUWEOTb80=
|
||||
gonum.org/v1/gonum v0.17.0 h1:VbpOemQlsSMrYmn7T2OUvQ4dqxQXU+ouZFQsZOx50z4=
|
||||
gonum.org/v1/gonum v0.17.0/go.mod h1:El3tOrEuMpv2UdMrbNlKEh9vd86bmQ6vqIcDwxEOc1E=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688 h1:ax2KzoSRIZU/M0cIxri3pKxy99vniH1PVxWC6si/eZI=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260819154853-08b0e4226688/go.mod h1:1RJ9BQGyNdZwkGc1eTqkErfRZ6RJyYPHZo73BZ1vQqI=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 h1:cYNAzI2sUwhmCcoj9TxvihSrqsxt6uIkj3rDRhSDmW4=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA=
|
||||
google.golang.org/grpc v1.83.1 h1:HIO0+BEtBP6soyqvqC8sNUjZ7bTs+0hFQuFF+RAy++Y=
|
||||
google.golang.org/grpc v1.83.1/go.mod h1:kDyl6SKsiHKt0uylY5gtn5cEjkrIOhQOGDgIc4JGwzQ=
|
||||
google.golang.org/protobuf v1.36.12 h1:pJOKDDOyeXErUroCihFAd5LQuwXBSpVnKGrj5o/fwxc=
|
||||
google.golang.org/protobuf v1.36.12/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
|
||||
|
||||
921
internal/metrics/metrics.go
Normal file
921
internal/metrics/metrics.go
Normal file
@@ -0,0 +1,921 @@
|
||||
// Package metrics provides the application-level metrics facade for Gerbil.
|
||||
//
|
||||
// Application code (main, relay, proxy) uses only the Record* functions in this
|
||||
// package. The actual recording is delegated to the backend selected in
|
||||
// internal/observability. Neither Prometheus nor OTel packages are imported here.
|
||||
package metrics
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"sync"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/observability"
|
||||
)
|
||||
|
||||
// Config is the metrics configuration type. It is an alias for
|
||||
// observability.MetricsConfig so callers do not need to import observability.
|
||||
type Config = observability.MetricsConfig
|
||||
|
||||
// PrometheusConfig is re-exported for convenience.
|
||||
type PrometheusConfig = observability.PrometheusConfig
|
||||
|
||||
// OTelConfig is re-exported for convenience.
|
||||
type OTelConfig = observability.OTelConfig
|
||||
|
||||
var (
|
||||
backend observability.Backend
|
||||
initMu sync.Mutex
|
||||
|
||||
// Interface and peer metrics
|
||||
wgInterfaceUp observability.Int64Gauge
|
||||
wgPeersTotal observability.UpDownCounter
|
||||
wgPeerConnected observability.Int64Gauge
|
||||
wgHandshakesTotal observability.Counter
|
||||
wgHandshakeLatency observability.Histogram
|
||||
wgPeerRTT observability.Histogram
|
||||
wgBytesReceived observability.Counter
|
||||
wgBytesTransmitted observability.Counter
|
||||
allowedIPsCount observability.UpDownCounter
|
||||
keyRotationTotal observability.Counter
|
||||
|
||||
// System and proxy metrics
|
||||
netlinkEventsTotal observability.Counter
|
||||
netlinkErrorsTotal observability.Counter
|
||||
syncDuration observability.Histogram
|
||||
workqueueDepth observability.UpDownCounter
|
||||
kernelModuleLoads observability.Counter
|
||||
firewallRulesApplied observability.Counter
|
||||
activeSessions observability.UpDownCounter
|
||||
activeProxyConnections observability.UpDownCounter
|
||||
proxyRouteLookups observability.Counter
|
||||
proxyTLSHandshake observability.Histogram
|
||||
proxyBytesTransmitted observability.Counter
|
||||
|
||||
// UDP Relay / Proxy Metrics
|
||||
udpPacketsTotal observability.Counter
|
||||
udpPacketSizeBytes observability.Histogram
|
||||
holePunchEventsTotal observability.Counter
|
||||
proxyMappingActive observability.UpDownCounter
|
||||
relayUDPConnectionsActive observability.UpDownCounter
|
||||
sessionRebuiltTotal observability.Counter
|
||||
commPatternActive observability.UpDownCounter
|
||||
proxyCleanupRemovedTotal observability.Counter
|
||||
proxyConnectionErrorsTotal observability.Counter
|
||||
proxyInitialMappingsTotal observability.Int64Gauge
|
||||
proxyMappingUpdatesTotal observability.Counter
|
||||
proxyIdleCleanupDuration observability.Histogram
|
||||
|
||||
// SNI Proxy Metrics
|
||||
sniConnectionsTotal observability.Counter
|
||||
sniConnectionDuration observability.Histogram
|
||||
sniActiveConnections observability.UpDownCounter
|
||||
sniRouteCacheHitsTotal observability.Counter
|
||||
sniRouteAPIRequestsTotal observability.Counter
|
||||
sniRouteAPILatency observability.Histogram
|
||||
sniLocalOverrideTotal observability.Counter
|
||||
sniTrustedProxyEventsTotal observability.Counter
|
||||
sniProxyProtocolParseErrorsTotal observability.Counter
|
||||
sniDataBytesTotal observability.Counter
|
||||
sniTunnelTerminationsTotal observability.Counter
|
||||
|
||||
// HTTP API & Peer Management Metrics
|
||||
httpRequestsTotal observability.Counter
|
||||
httpRequestDuration observability.Histogram
|
||||
peerOperationsTotal observability.Counter
|
||||
proxyMappingUpdateRequestsTotal observability.Counter
|
||||
destinationsUpdateRequestsTotal observability.Counter
|
||||
|
||||
// Remote Configuration, Reporting & Housekeeping
|
||||
remoteConfigFetchesTotal observability.Counter
|
||||
bandwidthReportsTotal observability.Counter
|
||||
peerBandwidthBytesTotal observability.Counter
|
||||
memorySpikeTotal observability.Counter
|
||||
heapProfilesWrittenTotal observability.Counter
|
||||
|
||||
// Operational metrics
|
||||
configReloadsTotal observability.Counter
|
||||
restartTotal observability.Counter
|
||||
authFailuresTotal observability.Counter
|
||||
aclDeniedTotal observability.Counter
|
||||
certificateExpiryDays observability.Float64Gauge
|
||||
)
|
||||
|
||||
// DefaultConfig returns a default metrics configuration.
|
||||
func DefaultConfig() Config {
|
||||
return observability.DefaultMetricsConfig()
|
||||
}
|
||||
|
||||
// Initialize sets up the metrics system using the selected backend.
|
||||
// It returns the /metrics HTTP handler (non-nil only for Prometheus backend).
|
||||
func Initialize(cfg Config) (http.Handler, error) {
|
||||
initMu.Lock()
|
||||
defer initMu.Unlock()
|
||||
|
||||
if backend != nil {
|
||||
return backend.HTTPHandler(), nil
|
||||
}
|
||||
|
||||
b, err := observability.New(cfg)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
backend = b
|
||||
|
||||
if err := createInstruments(); err != nil {
|
||||
backend = nil
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return backend.HTTPHandler(), nil
|
||||
}
|
||||
|
||||
// Shutdown gracefully shuts down the metrics backend.
|
||||
func Shutdown(ctx context.Context) error {
|
||||
initMu.Lock()
|
||||
b := backend
|
||||
backend = nil
|
||||
initMu.Unlock()
|
||||
|
||||
if b != nil {
|
||||
return b.Shutdown(ctx)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func createInstruments() error {
|
||||
durationBuckets := []float64{0.005, 0.01, 0.025, 0.05, 0.1, 0.25, 0.5, 1, 2.5, 5, 10, 30}
|
||||
sizeBuckets := []float64{512, 1024, 4096, 16384, 65536, 262144, 1048576}
|
||||
sniDurationBuckets := []float64{0.1, 0.5, 1, 2.5, 5, 10, 30, 60, 120}
|
||||
|
||||
b := backend
|
||||
|
||||
newCounter := func(name, desc string, labelNames ...string) (observability.Counter, error) {
|
||||
c, err := b.NewCounter(name, desc, labelNames...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create counter %q: %w", name, err)
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
|
||||
newUpDownCounter := func(name, desc string, labelNames ...string) (observability.UpDownCounter, error) {
|
||||
c, err := b.NewUpDownCounter(name, desc, labelNames...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create updown counter %q: %w", name, err)
|
||||
}
|
||||
return c, nil
|
||||
}
|
||||
|
||||
newInt64Gauge := func(name, desc string, labelNames ...string) (observability.Int64Gauge, error) {
|
||||
g, err := b.NewInt64Gauge(name, desc, labelNames...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create int64 gauge %q: %w", name, err)
|
||||
}
|
||||
return g, nil
|
||||
}
|
||||
|
||||
newFloat64Gauge := func(name, desc string, labelNames ...string) (observability.Float64Gauge, error) {
|
||||
g, err := b.NewFloat64Gauge(name, desc, labelNames...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create float64 gauge %q: %w", name, err)
|
||||
}
|
||||
return g, nil
|
||||
}
|
||||
|
||||
newHistogram := func(name, desc string, buckets []float64, labelNames ...string) (observability.Histogram, error) {
|
||||
h, err := b.NewHistogram(name, desc, buckets, labelNames...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("create histogram %q: %w", name, err)
|
||||
}
|
||||
return h, nil
|
||||
}
|
||||
|
||||
var err error
|
||||
|
||||
wgInterfaceUp, err = newInt64Gauge("gerbil_wg_interface_up",
|
||||
"Operational state of a WireGuard interface (1=up, 0=down)", "ifname", "instance")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgPeersTotal, err = newUpDownCounter("gerbil_wg_peers_total",
|
||||
"Total number of configured peers per interface", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgPeerConnected, err = newInt64Gauge("gerbil_wg_peer_connected",
|
||||
"Whether a specific peer is connected (1=connected, 0=disconnected)", "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
allowedIPsCount, err = newUpDownCounter("gerbil_allowed_ips_count",
|
||||
"Number of allowed IPs configured per peer", "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
keyRotationTotal, err = newCounter("gerbil_key_rotation_total",
|
||||
"Key rotation events", "ifname", "reason")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgHandshakesTotal, err = newCounter("gerbil_wg_handshakes_total",
|
||||
"Count of handshake attempts with their result status", "ifname", "peer", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgHandshakeLatency, err = newHistogram("gerbil_wg_handshake_latency_seconds",
|
||||
"Distribution of handshake latencies in seconds", durationBuckets, "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgPeerRTT, err = newHistogram("gerbil_wg_peer_rtt_seconds",
|
||||
"Observed round-trip time to a peer in seconds", durationBuckets, "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgBytesReceived, err = newCounter("gerbil_wg_bytes_received_total",
|
||||
"Number of bytes received from a peer", "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
wgBytesTransmitted, err = newCounter("gerbil_wg_bytes_transmitted_total",
|
||||
"Number of bytes transmitted to a peer", "ifname", "peer")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
netlinkEventsTotal, err = newCounter("gerbil_netlink_events_total",
|
||||
"Number of netlink events processed", "event_type")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
netlinkErrorsTotal, err = newCounter("gerbil_netlink_errors_total",
|
||||
"Count of netlink or kernel errors", "component", "error_type")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
syncDuration, err = newHistogram("gerbil_sync_duration_seconds",
|
||||
"Duration of reconciliation/sync loops in seconds", durationBuckets, "component")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
workqueueDepth, err = newUpDownCounter("gerbil_workqueue_depth",
|
||||
"Current length of internal work queues", "queue")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
kernelModuleLoads, err = newCounter("gerbil_kernel_module_loads_total",
|
||||
"Count of kernel module load attempts", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
firewallRulesApplied, err = newCounter("gerbil_firewall_rules_applied_total",
|
||||
"IPTables/NFT rules applied", "result", "chain")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
activeSessions, err = newUpDownCounter("gerbil_active_sessions",
|
||||
"Number of active UDP relay sessions", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
activeProxyConnections, err = newUpDownCounter("gerbil_active_proxy_connections",
|
||||
"Active SNI proxy connections")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyRouteLookups, err = newCounter("gerbil_proxy_route_lookups_total",
|
||||
"Number of route lookups", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyTLSHandshake, err = newHistogram("gerbil_proxy_tls_handshake_seconds",
|
||||
"TLS handshake duration for SNI proxy in seconds", durationBuckets)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyBytesTransmitted, err = newCounter("gerbil_proxy_bytes_transmitted_total",
|
||||
"Bytes sent/received by the SNI proxy", "direction")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
configReloadsTotal, err = newCounter("gerbil_config_reloads_total",
|
||||
"Number of configuration reloads", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
restartTotal, err = newCounter("gerbil_restart_total",
|
||||
"Process restart count")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
authFailuresTotal, err = newCounter("gerbil_auth_failures_total",
|
||||
"Count of authentication or peer validation failures", "peer", "reason")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
aclDeniedTotal, err = newCounter("gerbil_acl_denied_total",
|
||||
"Access control denied events", "ifname", "peer", "policy")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
certificateExpiryDays, err = newFloat64Gauge("gerbil_certificate_expiry_days",
|
||||
"Days until certificate expiry", "cert_name", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
udpPacketsTotal, err = newCounter("gerbil_udp_packets_total",
|
||||
"Count of UDP packets processed by relay workers", "ifname", "type", "direction")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
udpPacketSizeBytes, err = newHistogram("gerbil_udp_packet_size_bytes",
|
||||
"Size distribution of packets forwarded through relay", sizeBuckets, "ifname", "type")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
holePunchEventsTotal, err = newCounter("gerbil_hole_punch_events_total",
|
||||
"Count of hole punch messages processed", "ifname", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyMappingActive, err = newUpDownCounter("gerbil_proxy_mapping_active",
|
||||
"Number of active proxy mappings", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
relayUDPConnectionsActive, err = newUpDownCounter("gerbil_relay_udp_connections_active",
|
||||
"Number of open per-peer outbound UDP sockets held by the relay connection pool", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sessionRebuiltTotal, err = newCounter("gerbil_session_rebuilt_total",
|
||||
"Count of sessions rebuilt from communication patterns", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
commPatternActive, err = newUpDownCounter("gerbil_comm_pattern_active",
|
||||
"Number of active communication patterns", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyCleanupRemovedTotal, err = newCounter("gerbil_proxy_cleanup_removed_total",
|
||||
"Count of items removed during cleanup routines", "ifname", "component")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyConnectionErrorsTotal, err = newCounter("gerbil_proxy_connection_errors_total",
|
||||
"Count of connection errors in proxy operations", "ifname", "error_type")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyInitialMappingsTotal, err = newInt64Gauge("gerbil_proxy_initial_mappings",
|
||||
"Number of initial proxy mappings loaded", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyMappingUpdatesTotal, err = newCounter("gerbil_proxy_mapping_updates_total",
|
||||
"Count of proxy mapping updates", "ifname")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyIdleCleanupDuration, err = newHistogram("gerbil_proxy_idle_cleanup_duration_seconds",
|
||||
"Duration of cleanup cycles", durationBuckets, "ifname", "component")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniConnectionsTotal, err = newCounter("gerbil_sni_connections_total",
|
||||
"Count of connections processed by SNI proxy", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniConnectionDuration, err = newHistogram("gerbil_sni_connection_duration_seconds",
|
||||
"Lifetime distribution of proxied TLS connections", sniDurationBuckets)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniActiveConnections, err = newUpDownCounter("gerbil_sni_active_connections",
|
||||
"Number of active SNI tunnels")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniRouteCacheHitsTotal, err = newCounter("gerbil_sni_route_cache_hits_total",
|
||||
"Count of route cache hits and misses", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniRouteAPIRequestsTotal, err = newCounter("gerbil_sni_route_api_requests_total",
|
||||
"Count of route API requests", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniRouteAPILatency, err = newHistogram("gerbil_sni_route_api_latency_seconds",
|
||||
"Distribution of route API call latencies", durationBuckets)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniLocalOverrideTotal, err = newCounter("gerbil_sni_local_override_total",
|
||||
"Count of routes using local overrides", "hit")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniTrustedProxyEventsTotal, err = newCounter("gerbil_sni_trusted_proxy_events_total",
|
||||
"Count of PROXY protocol events", "event")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniProxyProtocolParseErrorsTotal, err = newCounter("gerbil_sni_proxy_protocol_parse_errors_total",
|
||||
"Count of PROXY protocol parse failures")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniDataBytesTotal, err = newCounter("gerbil_sni_data_bytes_total",
|
||||
"Count of bytes proxied through SNI tunnels", "direction")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
sniTunnelTerminationsTotal, err = newCounter("gerbil_sni_tunnel_terminations_total",
|
||||
"Count of tunnel terminations by reason", "reason")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
httpRequestsTotal, err = newCounter("gerbil_http_requests_total",
|
||||
"Count of HTTP requests to management API", "endpoint", "method", "status_code")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
httpRequestDuration, err = newHistogram("gerbil_http_request_duration_seconds",
|
||||
"Distribution of HTTP request handling time", durationBuckets, "endpoint", "method")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
peerOperationsTotal, err = newCounter("gerbil_peer_operations_total",
|
||||
"Count of peer lifecycle operations", "operation", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
proxyMappingUpdateRequestsTotal, err = newCounter("gerbil_proxy_mapping_update_requests_total",
|
||||
"Count of proxy mapping update API calls", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
destinationsUpdateRequestsTotal, err = newCounter("gerbil_destinations_update_requests_total",
|
||||
"Count of destinations update API calls", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
remoteConfigFetchesTotal, err = newCounter("gerbil_remote_config_fetches_total",
|
||||
"Count of remote configuration fetch attempts", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
bandwidthReportsTotal, err = newCounter("gerbil_bandwidth_reports_total",
|
||||
"Count of bandwidth report transmissions", "result")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
peerBandwidthBytesTotal, err = newCounter("gerbil_peer_bandwidth_bytes_total",
|
||||
"Bytes per peer tracked by bandwidth calculation", "peer", "direction")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
memorySpikeTotal, err = newCounter("gerbil_memory_spike_total",
|
||||
"Count of memory spikes detected", "severity")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
heapProfilesWrittenTotal, err = newCounter("gerbil_heap_profiles_written_total",
|
||||
"Count of heap profile files generated")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func RecordInterfaceUp(ifname, instance string, up bool) {
|
||||
if wgInterfaceUp == nil {
|
||||
return
|
||||
}
|
||||
value := int64(0)
|
||||
if up {
|
||||
value = 1
|
||||
}
|
||||
wgInterfaceUp.Record(context.Background(), value, observability.Labels{"ifname": ifname, "instance": instance})
|
||||
}
|
||||
|
||||
func RecordPeersTotal(ifname string, delta int64) {
|
||||
if wgPeersTotal == nil {
|
||||
return
|
||||
}
|
||||
wgPeersTotal.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordPeerConnected(ifname, peer string, connected bool) {
|
||||
if wgPeerConnected == nil {
|
||||
return
|
||||
}
|
||||
value := int64(0)
|
||||
if connected {
|
||||
value = 1
|
||||
}
|
||||
wgPeerConnected.Record(context.Background(), value, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordHandshake(ifname, peer, result string) {
|
||||
if wgHandshakesTotal == nil {
|
||||
return
|
||||
}
|
||||
wgHandshakesTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "peer": peer, "result": result})
|
||||
}
|
||||
|
||||
func RecordHandshakeLatency(ifname, peer string, seconds float64) {
|
||||
if wgHandshakeLatency == nil {
|
||||
return
|
||||
}
|
||||
wgHandshakeLatency.Record(context.Background(), seconds, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordPeerRTT(ifname, peer string, seconds float64) {
|
||||
if wgPeerRTT == nil {
|
||||
return
|
||||
}
|
||||
wgPeerRTT.Record(context.Background(), seconds, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordBytesReceived(ifname, peer string, bytes int64) {
|
||||
if wgBytesReceived == nil {
|
||||
return
|
||||
}
|
||||
wgBytesReceived.Add(context.Background(), bytes, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordBytesTransmitted(ifname, peer string, bytes int64) {
|
||||
if wgBytesTransmitted == nil {
|
||||
return
|
||||
}
|
||||
wgBytesTransmitted.Add(context.Background(), bytes, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordAllowedIPsCount(ifname, peer string, delta int64) {
|
||||
if allowedIPsCount == nil {
|
||||
return
|
||||
}
|
||||
allowedIPsCount.Add(context.Background(), delta, observability.Labels{"ifname": ifname, "peer": peer})
|
||||
}
|
||||
|
||||
func RecordKeyRotation(ifname, reason string) {
|
||||
if keyRotationTotal == nil {
|
||||
return
|
||||
}
|
||||
keyRotationTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "reason": reason})
|
||||
}
|
||||
|
||||
func RecordNetlinkEvent(eventType string) {
|
||||
if netlinkEventsTotal == nil {
|
||||
return
|
||||
}
|
||||
netlinkEventsTotal.Add(context.Background(), 1, observability.Labels{"event_type": eventType})
|
||||
}
|
||||
|
||||
func RecordNetlinkError(component, errorType string) {
|
||||
if netlinkErrorsTotal == nil {
|
||||
return
|
||||
}
|
||||
netlinkErrorsTotal.Add(context.Background(), 1, observability.Labels{"component": component, "error_type": errorType})
|
||||
}
|
||||
|
||||
func RecordSyncDuration(component string, seconds float64) {
|
||||
if syncDuration == nil {
|
||||
return
|
||||
}
|
||||
syncDuration.Record(context.Background(), seconds, observability.Labels{"component": component})
|
||||
}
|
||||
|
||||
func RecordWorkqueueDepth(queue string, delta int64) {
|
||||
if workqueueDepth == nil {
|
||||
return
|
||||
}
|
||||
workqueueDepth.Add(context.Background(), delta, observability.Labels{"queue": queue})
|
||||
}
|
||||
|
||||
func RecordKernelModuleLoad(result string) {
|
||||
if kernelModuleLoads == nil {
|
||||
return
|
||||
}
|
||||
kernelModuleLoads.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordFirewallRuleApplied(result, chain string) {
|
||||
if firewallRulesApplied == nil {
|
||||
return
|
||||
}
|
||||
firewallRulesApplied.Add(context.Background(), 1, observability.Labels{"result": result, "chain": chain})
|
||||
}
|
||||
|
||||
func RecordActiveSession(ifname string, delta int64) {
|
||||
if activeSessions == nil {
|
||||
return
|
||||
}
|
||||
activeSessions.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordActiveProxyConnection(delta int64) {
|
||||
if activeProxyConnections == nil {
|
||||
return
|
||||
}
|
||||
activeProxyConnections.Add(context.Background(), delta, nil)
|
||||
}
|
||||
|
||||
func RecordProxyRouteLookup(result string) {
|
||||
if proxyRouteLookups == nil {
|
||||
return
|
||||
}
|
||||
proxyRouteLookups.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordProxyTLSHandshake(seconds float64) {
|
||||
if proxyTLSHandshake == nil {
|
||||
return
|
||||
}
|
||||
proxyTLSHandshake.Record(context.Background(), seconds, nil)
|
||||
}
|
||||
|
||||
func RecordProxyBytesTransmitted(direction string, bytes int64) {
|
||||
if proxyBytesTransmitted == nil {
|
||||
return
|
||||
}
|
||||
proxyBytesTransmitted.Add(context.Background(), bytes, observability.Labels{"direction": direction})
|
||||
}
|
||||
|
||||
func RecordConfigReload(result string) {
|
||||
if configReloadsTotal == nil {
|
||||
return
|
||||
}
|
||||
configReloadsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordRestart() {
|
||||
if restartTotal == nil {
|
||||
return
|
||||
}
|
||||
restartTotal.Add(context.Background(), 1, nil)
|
||||
}
|
||||
|
||||
func RecordAuthFailure(peer, reason string) {
|
||||
if authFailuresTotal == nil {
|
||||
return
|
||||
}
|
||||
authFailuresTotal.Add(context.Background(), 1, observability.Labels{"peer": peer, "reason": reason})
|
||||
}
|
||||
|
||||
func RecordACLDenied(ifname, peer, policy string) {
|
||||
if aclDeniedTotal == nil {
|
||||
return
|
||||
}
|
||||
aclDeniedTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "peer": peer, "policy": policy})
|
||||
}
|
||||
|
||||
func RecordCertificateExpiry(certName, ifname string, days float64) {
|
||||
if certificateExpiryDays == nil {
|
||||
return
|
||||
}
|
||||
certificateExpiryDays.Record(context.Background(), days, observability.Labels{"cert_name": certName, "ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordUDPPacket(ifname, packetType, direction string) {
|
||||
if udpPacketsTotal == nil {
|
||||
return
|
||||
}
|
||||
udpPacketsTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "type": packetType, "direction": direction})
|
||||
}
|
||||
|
||||
func RecordUDPPacketSize(ifname, packetType string, bytes float64) {
|
||||
if udpPacketSizeBytes == nil {
|
||||
return
|
||||
}
|
||||
udpPacketSizeBytes.Record(context.Background(), bytes, observability.Labels{"ifname": ifname, "type": packetType})
|
||||
}
|
||||
|
||||
func RecordHolePunchEvent(ifname, result string) {
|
||||
if holePunchEventsTotal == nil {
|
||||
return
|
||||
}
|
||||
holePunchEventsTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "result": result})
|
||||
}
|
||||
|
||||
func RecordProxyMapping(ifname string, delta int64) {
|
||||
if proxyMappingActive == nil {
|
||||
return
|
||||
}
|
||||
proxyMappingActive.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordSession(ifname string, delta int64) {
|
||||
if activeSessions == nil {
|
||||
return
|
||||
}
|
||||
activeSessions.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordUDPConnection(ifname string, delta int64) {
|
||||
if relayUDPConnectionsActive == nil {
|
||||
return
|
||||
}
|
||||
relayUDPConnectionsActive.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordSessionRebuilt(ifname string) {
|
||||
if sessionRebuiltTotal == nil {
|
||||
return
|
||||
}
|
||||
sessionRebuiltTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordCommPattern(ifname string, delta int64) {
|
||||
if commPatternActive == nil {
|
||||
return
|
||||
}
|
||||
commPatternActive.Add(context.Background(), delta, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordProxyCleanupRemoved(ifname, component string, count int64) {
|
||||
if proxyCleanupRemovedTotal == nil {
|
||||
return
|
||||
}
|
||||
proxyCleanupRemovedTotal.Add(context.Background(), count, observability.Labels{"ifname": ifname, "component": component})
|
||||
}
|
||||
|
||||
func RecordProxyConnectionError(ifname, errorType string) {
|
||||
if proxyConnectionErrorsTotal == nil {
|
||||
return
|
||||
}
|
||||
proxyConnectionErrorsTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname, "error_type": errorType})
|
||||
}
|
||||
|
||||
func RecordProxyInitialMappings(ifname string, count int64) {
|
||||
if proxyInitialMappingsTotal == nil {
|
||||
return
|
||||
}
|
||||
proxyInitialMappingsTotal.Record(context.Background(), count, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordProxyMappingUpdate(ifname string) {
|
||||
if proxyMappingUpdatesTotal == nil {
|
||||
return
|
||||
}
|
||||
proxyMappingUpdatesTotal.Add(context.Background(), 1, observability.Labels{"ifname": ifname})
|
||||
}
|
||||
|
||||
func RecordProxyIdleCleanupDuration(ifname, component string, seconds float64) {
|
||||
if proxyIdleCleanupDuration == nil {
|
||||
return
|
||||
}
|
||||
proxyIdleCleanupDuration.Record(context.Background(), seconds, observability.Labels{"ifname": ifname, "component": component})
|
||||
}
|
||||
|
||||
func RecordSNIConnection(result string) {
|
||||
if sniConnectionsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniConnectionsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordSNIConnectionDuration(seconds float64) {
|
||||
if sniConnectionDuration == nil {
|
||||
return
|
||||
}
|
||||
sniConnectionDuration.Record(context.Background(), seconds, nil)
|
||||
}
|
||||
|
||||
func RecordSNIActiveConnection(delta int64) {
|
||||
if sniActiveConnections == nil {
|
||||
return
|
||||
}
|
||||
sniActiveConnections.Add(context.Background(), delta, nil)
|
||||
}
|
||||
|
||||
func RecordSNIRouteCacheHit(result string) {
|
||||
if sniRouteCacheHitsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniRouteCacheHitsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordSNIRouteAPIRequest(result string) {
|
||||
if sniRouteAPIRequestsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniRouteAPIRequestsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordSNIRouteAPILatency(seconds float64) {
|
||||
if sniRouteAPILatency == nil {
|
||||
return
|
||||
}
|
||||
sniRouteAPILatency.Record(context.Background(), seconds, nil)
|
||||
}
|
||||
|
||||
func RecordSNILocalOverride(hit string) {
|
||||
if sniLocalOverrideTotal == nil {
|
||||
return
|
||||
}
|
||||
sniLocalOverrideTotal.Add(context.Background(), 1, observability.Labels{"hit": hit})
|
||||
}
|
||||
|
||||
func RecordSNITrustedProxyEvent(event string) {
|
||||
if sniTrustedProxyEventsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniTrustedProxyEventsTotal.Add(context.Background(), 1, observability.Labels{"event": event})
|
||||
}
|
||||
|
||||
func RecordSNIProxyProtocolParseError() {
|
||||
if sniProxyProtocolParseErrorsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniProxyProtocolParseErrorsTotal.Add(context.Background(), 1, nil)
|
||||
}
|
||||
|
||||
func RecordSNIDataBytes(direction string, bytes int64) {
|
||||
if sniDataBytesTotal == nil {
|
||||
return
|
||||
}
|
||||
sniDataBytesTotal.Add(context.Background(), bytes, observability.Labels{"direction": direction})
|
||||
}
|
||||
|
||||
func RecordSNITunnelTermination(reason string) {
|
||||
if sniTunnelTerminationsTotal == nil {
|
||||
return
|
||||
}
|
||||
sniTunnelTerminationsTotal.Add(context.Background(), 1, observability.Labels{"reason": reason})
|
||||
}
|
||||
|
||||
func RecordHTTPRequest(endpoint, method, statusCode string) {
|
||||
if httpRequestsTotal == nil {
|
||||
return
|
||||
}
|
||||
httpRequestsTotal.Add(context.Background(), 1, observability.Labels{"endpoint": endpoint, "method": method, "status_code": statusCode})
|
||||
}
|
||||
|
||||
func RecordHTTPRequestDuration(endpoint, method string, seconds float64) {
|
||||
if httpRequestDuration == nil {
|
||||
return
|
||||
}
|
||||
httpRequestDuration.Record(context.Background(), seconds, observability.Labels{"endpoint": endpoint, "method": method})
|
||||
}
|
||||
|
||||
func RecordPeerOperation(operation, result string) {
|
||||
if peerOperationsTotal == nil {
|
||||
return
|
||||
}
|
||||
peerOperationsTotal.Add(context.Background(), 1, observability.Labels{"operation": operation, "result": result})
|
||||
}
|
||||
|
||||
func RecordProxyMappingUpdateRequest(result string) {
|
||||
if proxyMappingUpdateRequestsTotal == nil {
|
||||
return
|
||||
}
|
||||
proxyMappingUpdateRequestsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordDestinationsUpdateRequest(result string) {
|
||||
if destinationsUpdateRequestsTotal == nil {
|
||||
return
|
||||
}
|
||||
destinationsUpdateRequestsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordRemoteConfigFetch(result string) {
|
||||
if remoteConfigFetchesTotal == nil {
|
||||
return
|
||||
}
|
||||
remoteConfigFetchesTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordBandwidthReport(result string) {
|
||||
if bandwidthReportsTotal == nil {
|
||||
return
|
||||
}
|
||||
bandwidthReportsTotal.Add(context.Background(), 1, observability.Labels{"result": result})
|
||||
}
|
||||
|
||||
func RecordPeerBandwidthBytes(peer, direction string, bytes int64) {
|
||||
if peerBandwidthBytesTotal == nil {
|
||||
return
|
||||
}
|
||||
peerBandwidthBytesTotal.Add(context.Background(), bytes, observability.Labels{"peer": peer, "direction": direction})
|
||||
}
|
||||
|
||||
func RecordMemorySpike(severity string) {
|
||||
if memorySpikeTotal == nil {
|
||||
return
|
||||
}
|
||||
memorySpikeTotal.Add(context.Background(), 1, observability.Labels{"severity": severity})
|
||||
}
|
||||
|
||||
func RecordHeapProfileWritten() {
|
||||
if heapProfilesWrittenTotal == nil {
|
||||
return
|
||||
}
|
||||
heapProfilesWrittenTotal.Add(context.Background(), 1, nil)
|
||||
}
|
||||
262
internal/metrics/metrics_test.go
Normal file
262
internal/metrics/metrics_test.go
Normal file
@@ -0,0 +1,262 @@
|
||||
package metrics_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/metrics"
|
||||
"github.com/fosrl/gerbil/internal/observability"
|
||||
)
|
||||
|
||||
const exampleHostname = "example.com"
|
||||
|
||||
func initPrometheus(t *testing.T) http.Handler {
|
||||
t.Helper()
|
||||
cfg := metrics.DefaultConfig()
|
||||
cfg.Enabled = true
|
||||
cfg.Backend = "prometheus"
|
||||
cfg.Prometheus.Path = "/metrics"
|
||||
|
||||
h, err := metrics.Initialize(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("Initialize failed: %v", err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
metrics.Shutdown(context.Background()) //nolint:errcheck
|
||||
})
|
||||
return h
|
||||
}
|
||||
|
||||
func initNoop(t *testing.T) {
|
||||
t.Helper()
|
||||
cfg := metrics.DefaultConfig()
|
||||
cfg.Enabled = false
|
||||
_, err := metrics.Initialize(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("Initialize noop failed: %v", err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
metrics.Shutdown(context.Background()) //nolint:errcheck
|
||||
})
|
||||
}
|
||||
|
||||
func scrape(t *testing.T, h http.Handler) string {
|
||||
t.Helper()
|
||||
req := httptest.NewRequest(http.MethodGet, "/metrics", http.NoBody)
|
||||
rr := httptest.NewRecorder()
|
||||
h.ServeHTTP(rr, req)
|
||||
if rr.Code != http.StatusOK {
|
||||
t.Fatalf("scrape returned %d", rr.Code)
|
||||
}
|
||||
b, _ := io.ReadAll(rr.Body)
|
||||
return string(b)
|
||||
}
|
||||
|
||||
func assertContains(t *testing.T, body, substr string) {
|
||||
t.Helper()
|
||||
if !strings.Contains(body, substr) {
|
||||
t.Errorf("expected %q in output\nbody:\n%s", substr, body)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Tests ---
|
||||
|
||||
func TestInitializePrometheus(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
if h == nil {
|
||||
t.Error("expected non-nil HTTP handler for prometheus backend")
|
||||
}
|
||||
}
|
||||
|
||||
func TestInitializeNoop(t *testing.T) {
|
||||
initNoop(t)
|
||||
// All Record* functions must not panic when noop backend is active.
|
||||
metrics.RecordRestart()
|
||||
metrics.RecordHTTPRequest("/test", "GET", "200")
|
||||
metrics.RecordSNIConnection("accepted")
|
||||
metrics.RecordPeersTotal("wg0", 1)
|
||||
}
|
||||
|
||||
func TestDefaultConfig(t *testing.T) {
|
||||
cfg := metrics.DefaultConfig()
|
||||
if cfg.Backend != "prometheus" {
|
||||
t.Errorf("expected prometheus default backend, got %q", cfg.Backend)
|
||||
}
|
||||
}
|
||||
|
||||
func TestShutdownNoInit(t *testing.T) {
|
||||
// Ensure a known clean global state before testing no-init shutdown behavior.
|
||||
_ = metrics.Shutdown(context.Background())
|
||||
|
||||
// Shutdown without Initialize should not panic or error.
|
||||
if err := metrics.Shutdown(context.Background()); err != nil {
|
||||
t.Errorf("unexpected error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRecordHTTPRequest(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordHTTPRequest("/peers", "POST", "201")
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_http_requests_total")
|
||||
}
|
||||
|
||||
func TestRecordHTTPRequestDuration(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordHTTPRequestDuration("/peers", "POST", 0.05)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_http_request_duration_seconds")
|
||||
}
|
||||
|
||||
func TestRecordInterfaceUp(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordInterfaceUp("wg0", "host1", true)
|
||||
metrics.RecordInterfaceUp("wg0", "host1", false)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_wg_interface_up")
|
||||
}
|
||||
|
||||
func TestRecordPeersTotal(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordPeersTotal("wg0", 3)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_wg_peers_total")
|
||||
}
|
||||
|
||||
func TestRecordBytesReceivedTransmitted(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordBytesReceived("wg0", "peer1", 1024)
|
||||
metrics.RecordBytesTransmitted("wg0", "peer1", 512)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_wg_bytes_received_total")
|
||||
assertContains(t, body, "gerbil_wg_bytes_transmitted_total")
|
||||
}
|
||||
|
||||
func TestRecordSNI(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordSNIConnection("accepted")
|
||||
metrics.RecordSNIActiveConnection(1)
|
||||
metrics.RecordSNIConnectionDuration(1.5)
|
||||
metrics.RecordSNIRouteCacheHit("hit")
|
||||
metrics.RecordSNIRouteAPIRequest("success")
|
||||
metrics.RecordSNIRouteAPILatency(0.01)
|
||||
metrics.RecordSNILocalOverride("yes")
|
||||
metrics.RecordSNITrustedProxyEvent("proxy_protocol_parsed")
|
||||
metrics.RecordSNIProxyProtocolParseError()
|
||||
metrics.RecordSNIDataBytes("client_to_target", 2048)
|
||||
metrics.RecordSNITunnelTermination("eof")
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_sni_connections_total")
|
||||
assertContains(t, body, "gerbil_sni_active_connections")
|
||||
}
|
||||
|
||||
func TestRecordRelay(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordUDPPacket("relay", "data", "in")
|
||||
metrics.RecordUDPPacketSize("relay", "data", 256)
|
||||
metrics.RecordHolePunchEvent("relay", "success")
|
||||
metrics.RecordProxyMapping("relay", 1)
|
||||
metrics.RecordSession("relay", 1)
|
||||
metrics.RecordSessionRebuilt("relay")
|
||||
metrics.RecordCommPattern("relay", 1)
|
||||
metrics.RecordProxyCleanupRemoved("relay", "session", 2)
|
||||
metrics.RecordProxyConnectionError("relay", "dial_udp")
|
||||
metrics.RecordProxyInitialMappings("relay", 5)
|
||||
metrics.RecordProxyMappingUpdate("relay")
|
||||
metrics.RecordProxyIdleCleanupDuration("relay", "conn", 0.1)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_udp_packets_total")
|
||||
assertContains(t, body, "gerbil_proxy_mapping_active")
|
||||
assertContains(t, body, "gerbil_active_sessions")
|
||||
}
|
||||
|
||||
func TestRecordWireGuard(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordHandshake("wg0", "peer1", "success")
|
||||
metrics.RecordHandshakeLatency("wg0", "peer1", 0.02)
|
||||
metrics.RecordPeerRTT("wg0", "peer1", 0.005)
|
||||
metrics.RecordPeerConnected("wg0", "peer1", true)
|
||||
metrics.RecordAllowedIPsCount("wg0", "peer1", 2)
|
||||
metrics.RecordKeyRotation("wg0", "scheduled")
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_wg_handshakes_total")
|
||||
assertContains(t, body, "gerbil_wg_peer_connected")
|
||||
}
|
||||
|
||||
func TestRecordHousekeeping(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordRemoteConfigFetch("success")
|
||||
metrics.RecordBandwidthReport("success")
|
||||
metrics.RecordPeerBandwidthBytes("peer1", "rx", 512)
|
||||
metrics.RecordMemorySpike("warning")
|
||||
metrics.RecordHeapProfileWritten()
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_remote_config_fetches_total")
|
||||
assertContains(t, body, "gerbil_memory_spike_total")
|
||||
}
|
||||
|
||||
func TestRecordOperational(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordConfigReload("success")
|
||||
metrics.RecordRestart()
|
||||
metrics.RecordAuthFailure("peer1", "bad_key")
|
||||
metrics.RecordACLDenied("wg0", "peer1", "default-deny")
|
||||
metrics.RecordCertificateExpiry(exampleHostname, "wg0", 90.0)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_config_reloads_total")
|
||||
assertContains(t, body, "gerbil_restart_total")
|
||||
}
|
||||
|
||||
func TestRecordNetlink(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordNetlinkEvent("link_up")
|
||||
metrics.RecordNetlinkError("wg", "timeout")
|
||||
metrics.RecordSyncDuration("config", 0.1)
|
||||
metrics.RecordWorkqueueDepth("main", 3)
|
||||
metrics.RecordKernelModuleLoad("success")
|
||||
metrics.RecordFirewallRuleApplied("success", "INPUT")
|
||||
metrics.RecordActiveSession("wg0", 1)
|
||||
metrics.RecordActiveProxyConnection(1)
|
||||
metrics.RecordProxyRouteLookup("hit")
|
||||
metrics.RecordProxyTLSHandshake(0.05)
|
||||
metrics.RecordProxyBytesTransmitted("tx", 1024)
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_netlink_events_total")
|
||||
assertContains(t, body, "gerbil_active_sessions")
|
||||
}
|
||||
|
||||
func TestRecordPeerOperation(t *testing.T) {
|
||||
h := initPrometheus(t)
|
||||
metrics.RecordPeerOperation("add", "success")
|
||||
metrics.RecordProxyMappingUpdateRequest("success")
|
||||
metrics.RecordDestinationsUpdateRequest("success")
|
||||
body := scrape(t, h)
|
||||
assertContains(t, body, "gerbil_peer_operations_total")
|
||||
}
|
||||
|
||||
func TestInitializeInvalidBackend(t *testing.T) {
|
||||
cfg := observability.MetricsConfig{Enabled: true, Backend: "invalid"}
|
||||
_, err := metrics.Initialize(cfg)
|
||||
if err == nil {
|
||||
t.Error("expected error for invalid backend")
|
||||
}
|
||||
}
|
||||
|
||||
func TestInitializeBackendNone(t *testing.T) {
|
||||
cfg := metrics.DefaultConfig()
|
||||
cfg.Backend = "none"
|
||||
h, err := metrics.Initialize(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
if h != nil {
|
||||
t.Error("none backend should return nil handler")
|
||||
}
|
||||
// All Record* calls should be noop
|
||||
metrics.RecordRestart()
|
||||
metrics.Shutdown(context.Background()) //nolint:errcheck
|
||||
}
|
||||
129
internal/observability/config.go
Normal file
129
internal/observability/config.go
Normal file
@@ -0,0 +1,129 @@
|
||||
// Package observability provides a backend-neutral metrics abstraction for Gerbil.
|
||||
//
|
||||
// Exactly one metrics backend may be enabled at runtime:
|
||||
// - "prometheus" – native Prometheus client; exposes /metrics (no OTel SDK required)
|
||||
// - "otel" – OpenTelemetry metrics pushed via OTLP (gRPC or HTTP)
|
||||
// - "none" – metrics disabled; a safe noop implementation is used
|
||||
//
|
||||
// Future OTel tracing and logging can be added to this package alongside the
|
||||
// existing otel sub-package without touching the Prometheus-native path.
|
||||
package observability
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// MetricsConfig is the top-level metrics configuration.
|
||||
type MetricsConfig struct {
|
||||
// Enabled controls whether any metrics backend is started.
|
||||
// When false the noop backend is used regardless of Backend.
|
||||
Enabled bool
|
||||
|
||||
// Backend selects the active backend: "prometheus", "otel", or "none".
|
||||
Backend string
|
||||
|
||||
// Prometheus holds settings used only by the Prometheus-native backend.
|
||||
Prometheus PrometheusConfig
|
||||
|
||||
// OTel holds settings used only by the OTel backend.
|
||||
OTel OTelConfig
|
||||
|
||||
// ServiceName is propagated to OTel resource attributes.
|
||||
ServiceName string
|
||||
|
||||
// ServiceVersion is propagated to OTel resource attributes.
|
||||
ServiceVersion string
|
||||
|
||||
// DeploymentEnvironment is an optional OTel resource attribute.
|
||||
DeploymentEnvironment string
|
||||
}
|
||||
|
||||
// PrometheusConfig holds Prometheus-native backend settings.
|
||||
type PrometheusConfig struct {
|
||||
// Path is the HTTP path to expose the /metrics endpoint.
|
||||
// Defaults to "/metrics".
|
||||
Path string
|
||||
}
|
||||
|
||||
// OTelConfig holds OpenTelemetry backend settings.
|
||||
type OTelConfig struct {
|
||||
// Protocol is the OTLP transport: "grpc" (default) or "http".
|
||||
Protocol string
|
||||
|
||||
// Endpoint is the OTLP collector address (e.g. "localhost:4317").
|
||||
Endpoint string
|
||||
|
||||
// Insecure disables TLS for the OTLP connection.
|
||||
Insecure bool
|
||||
|
||||
// ExportInterval is how often metrics are pushed to the collector.
|
||||
// Defaults to 60 s.
|
||||
ExportInterval time.Duration
|
||||
|
||||
// Timeout bounds OTLP exporter construction calls.
|
||||
// Defaults to 10 s.
|
||||
Timeout time.Duration
|
||||
}
|
||||
|
||||
// DefaultMetricsConfig returns a MetricsConfig with sensible defaults.
|
||||
func DefaultMetricsConfig() MetricsConfig {
|
||||
return MetricsConfig{
|
||||
Enabled: true,
|
||||
Backend: "prometheus",
|
||||
Prometheus: PrometheusConfig{
|
||||
Path: "/metrics",
|
||||
},
|
||||
OTel: OTelConfig{
|
||||
Protocol: "grpc",
|
||||
Endpoint: "localhost:4317",
|
||||
Insecure: true,
|
||||
ExportInterval: 60 * time.Second,
|
||||
Timeout: 10 * time.Second,
|
||||
},
|
||||
ServiceName: "gerbil",
|
||||
ServiceVersion: "1.0.0",
|
||||
}
|
||||
}
|
||||
|
||||
// Validate checks the configuration for logical errors.
|
||||
func (c *MetricsConfig) Validate() error {
|
||||
if !c.Enabled {
|
||||
return nil
|
||||
}
|
||||
|
||||
switch c.Backend {
|
||||
case "prometheus", "none":
|
||||
// valid
|
||||
case "":
|
||||
return fmt.Errorf("metrics: enabled requires a non-empty backend")
|
||||
case "otel":
|
||||
if c.OTel.Endpoint == "" {
|
||||
return fmt.Errorf("metrics: backend=otel requires a non-empty OTel endpoint")
|
||||
}
|
||||
if c.OTel.Protocol != "grpc" && c.OTel.Protocol != "http" {
|
||||
return fmt.Errorf("metrics: otel protocol must be \"grpc\" or \"http\", got %q", c.OTel.Protocol)
|
||||
}
|
||||
if c.OTel.ExportInterval <= 0 {
|
||||
return fmt.Errorf("metrics: otel export interval must be positive")
|
||||
}
|
||||
if c.OTel.Timeout <= 0 {
|
||||
return fmt.Errorf("metrics: otel timeout must be positive")
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("metrics: unknown backend %q (must be \"prometheus\", \"otel\", or \"none\")", c.Backend)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// effectiveBackend resolves the backend string, treating "" and "none" as noop.
|
||||
func (c *MetricsConfig) effectiveBackend() string {
|
||||
if !c.Enabled {
|
||||
return "none"
|
||||
}
|
||||
if c.Backend == "" {
|
||||
return "none"
|
||||
}
|
||||
return c.Backend
|
||||
}
|
||||
153
internal/observability/metrics.go
Normal file
153
internal/observability/metrics.go
Normal file
@@ -0,0 +1,153 @@
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
|
||||
obsotel "github.com/fosrl/gerbil/internal/observability/otel"
|
||||
obsprom "github.com/fosrl/gerbil/internal/observability/prometheus"
|
||||
)
|
||||
|
||||
// Labels is a set of key-value pairs attached to a metric observation.
|
||||
// Use only stable, bounded-cardinality label values.
|
||||
type Labels = map[string]string
|
||||
|
||||
// Counter is a monotonically increasing instrument.
|
||||
type Counter interface {
|
||||
Add(ctx context.Context, value int64, labels Labels)
|
||||
}
|
||||
|
||||
// UpDownCounter is a bidirectional integer instrument (can go up or down).
|
||||
type UpDownCounter interface {
|
||||
Add(ctx context.Context, value int64, labels Labels)
|
||||
}
|
||||
|
||||
// Int64Gauge records a snapshot integer value.
|
||||
type Int64Gauge interface {
|
||||
Record(ctx context.Context, value int64, labels Labels)
|
||||
}
|
||||
|
||||
// Float64Gauge records a snapshot float value.
|
||||
type Float64Gauge interface {
|
||||
Record(ctx context.Context, value float64, labels Labels)
|
||||
}
|
||||
|
||||
// Histogram records a distribution of values.
|
||||
type Histogram interface {
|
||||
Record(ctx context.Context, value float64, labels Labels)
|
||||
}
|
||||
|
||||
// Backend is the single interface that each metrics implementation must satisfy.
|
||||
// Application code must not import backend-specific packages (prometheus, otel).
|
||||
type Backend interface {
|
||||
// NewCounter creates a counter metric.
|
||||
// labelNames declares the set of label keys that will be passed at observation time.
|
||||
NewCounter(name, desc string, labelNames ...string) (Counter, error)
|
||||
|
||||
// NewUpDownCounter creates an up-down counter metric.
|
||||
NewUpDownCounter(name, desc string, labelNames ...string) (UpDownCounter, error)
|
||||
|
||||
// NewInt64Gauge creates an integer gauge metric.
|
||||
NewInt64Gauge(name, desc string, labelNames ...string) (Int64Gauge, error)
|
||||
|
||||
// NewFloat64Gauge creates a float gauge metric.
|
||||
NewFloat64Gauge(name, desc string, labelNames ...string) (Float64Gauge, error)
|
||||
|
||||
// NewHistogram creates a histogram metric.
|
||||
// buckets are the explicit upper-bound bucket boundaries.
|
||||
NewHistogram(name, desc string, buckets []float64, labelNames ...string) (Histogram, error)
|
||||
|
||||
// HTTPHandler returns the /metrics HTTP handler.
|
||||
// Implementations that do not expose an HTTP endpoint return nil.
|
||||
HTTPHandler() http.Handler
|
||||
|
||||
// Shutdown performs a graceful flush / shutdown of the backend.
|
||||
Shutdown(ctx context.Context) error
|
||||
}
|
||||
|
||||
// New creates the backend selected by cfg and returns it.
|
||||
// Exactly one backend is created; the selection is mutually exclusive.
|
||||
func New(cfg MetricsConfig) (Backend, error) {
|
||||
if err := cfg.Validate(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
switch cfg.effectiveBackend() {
|
||||
case "prometheus":
|
||||
b, err := obsprom.New(obsprom.Config{
|
||||
Path: cfg.Prometheus.Path,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &promAdapter{b: b}, nil
|
||||
case "otel":
|
||||
b, err := obsotel.New(obsotel.Config{
|
||||
Protocol: cfg.OTel.Protocol,
|
||||
Endpoint: cfg.OTel.Endpoint,
|
||||
Insecure: cfg.OTel.Insecure,
|
||||
ExportInterval: cfg.OTel.ExportInterval,
|
||||
Timeout: cfg.OTel.Timeout,
|
||||
ServiceName: cfg.ServiceName,
|
||||
ServiceVersion: cfg.ServiceVersion,
|
||||
DeploymentEnvironment: cfg.DeploymentEnvironment,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return &otelAdapter{b: b}, nil
|
||||
case "none":
|
||||
return &NoopBackend{}, nil
|
||||
default:
|
||||
return nil, fmt.Errorf("observability: unknown backend %q", cfg.effectiveBackend())
|
||||
}
|
||||
}
|
||||
|
||||
// promAdapter wraps obsprom.Backend to implement the observability.Backend interface.
|
||||
// The concrete instrument types from the prometheus sub-package satisfy the instrument
|
||||
// interfaces via Go's structural (duck) typing without importing this package.
|
||||
type promAdapter struct {
|
||||
b *obsprom.Backend
|
||||
}
|
||||
|
||||
func (a *promAdapter) NewCounter(name, desc string, labelNames ...string) (Counter, error) {
|
||||
return a.b.NewCounter(name, desc, labelNames...)
|
||||
}
|
||||
func (a *promAdapter) NewUpDownCounter(name, desc string, labelNames ...string) (UpDownCounter, error) {
|
||||
return a.b.NewUpDownCounter(name, desc, labelNames...)
|
||||
}
|
||||
func (a *promAdapter) NewInt64Gauge(name, desc string, labelNames ...string) (Int64Gauge, error) {
|
||||
return a.b.NewInt64Gauge(name, desc, labelNames...)
|
||||
}
|
||||
func (a *promAdapter) NewFloat64Gauge(name, desc string, labelNames ...string) (Float64Gauge, error) {
|
||||
return a.b.NewFloat64Gauge(name, desc, labelNames...)
|
||||
}
|
||||
func (a *promAdapter) NewHistogram(name, desc string, buckets []float64, labelNames ...string) (Histogram, error) {
|
||||
return a.b.NewHistogram(name, desc, buckets, labelNames...)
|
||||
}
|
||||
func (a *promAdapter) HTTPHandler() http.Handler { return a.b.HTTPHandler() }
|
||||
func (a *promAdapter) Shutdown(ctx context.Context) error { return a.b.Shutdown(ctx) }
|
||||
|
||||
// otelAdapter wraps obsotel.Backend to implement the observability.Backend interface.
|
||||
type otelAdapter struct {
|
||||
b *obsotel.Backend
|
||||
}
|
||||
|
||||
func (a *otelAdapter) NewCounter(name, desc string, labelNames ...string) (Counter, error) {
|
||||
return a.b.NewCounter(name, desc, labelNames...)
|
||||
}
|
||||
func (a *otelAdapter) NewUpDownCounter(name, desc string, labelNames ...string) (UpDownCounter, error) {
|
||||
return a.b.NewUpDownCounter(name, desc, labelNames...)
|
||||
}
|
||||
func (a *otelAdapter) NewInt64Gauge(name, desc string, labelNames ...string) (Int64Gauge, error) {
|
||||
return a.b.NewInt64Gauge(name, desc, labelNames...)
|
||||
}
|
||||
func (a *otelAdapter) NewFloat64Gauge(name, desc string, labelNames ...string) (Float64Gauge, error) {
|
||||
return a.b.NewFloat64Gauge(name, desc, labelNames...)
|
||||
}
|
||||
func (a *otelAdapter) NewHistogram(name, desc string, buckets []float64, labelNames ...string) (Histogram, error) {
|
||||
return a.b.NewHistogram(name, desc, buckets, labelNames...)
|
||||
}
|
||||
func (a *otelAdapter) HTTPHandler() http.Handler { return a.b.HTTPHandler() }
|
||||
func (a *otelAdapter) Shutdown(ctx context.Context) error { return a.b.Shutdown(ctx) }
|
||||
263
internal/observability/metrics_test.go
Normal file
263
internal/observability/metrics_test.go
Normal file
@@ -0,0 +1,263 @@
|
||||
package observability_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/observability"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultMetricsPath = "/metrics"
|
||||
otelGRPCEndpoint = "localhost:4317"
|
||||
errUnexpectedFmt = "unexpected error: %v"
|
||||
)
|
||||
|
||||
func TestDefaultMetricsConfig(t *testing.T) {
|
||||
cfg := observability.DefaultMetricsConfig()
|
||||
if !cfg.Enabled {
|
||||
t.Error("default config should have Enabled=true")
|
||||
}
|
||||
if cfg.Backend != "prometheus" {
|
||||
t.Errorf("default backend should be prometheus, got %q", cfg.Backend)
|
||||
}
|
||||
if cfg.Prometheus.Path != defaultMetricsPath {
|
||||
t.Errorf("default prometheus path should be %s, got %q", defaultMetricsPath, cfg.Prometheus.Path)
|
||||
}
|
||||
if cfg.OTel.Protocol != "grpc" {
|
||||
t.Errorf("default otel protocol should be grpc, got %q", cfg.OTel.Protocol)
|
||||
}
|
||||
if cfg.OTel.ExportInterval != 60*time.Second {
|
||||
t.Errorf("default otel export interval should be 60s, got %v", cfg.OTel.ExportInterval)
|
||||
}
|
||||
}
|
||||
func TestValidateValidConfigs(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
cfg observability.MetricsConfig
|
||||
}{
|
||||
{name: "disabled", cfg: observability.MetricsConfig{Enabled: false}},
|
||||
{name: "backend none", cfg: observability.MetricsConfig{Enabled: true, Backend: "none"}},
|
||||
{name: "prometheus", cfg: observability.MetricsConfig{Enabled: true, Backend: "prometheus"}},
|
||||
{
|
||||
name: "otel grpc",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "grpc", Endpoint: otelGRPCEndpoint, ExportInterval: 10 * time.Second, Timeout: 2 * time.Second},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "otel http",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "http", Endpoint: "localhost:4318", ExportInterval: 30 * time.Second, Timeout: 2 * time.Second},
|
||||
},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if err := tt.cfg.Validate(); err != nil {
|
||||
t.Errorf("unexpected validation error: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateInvalidConfigs(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
cfg observability.MetricsConfig
|
||||
}{
|
||||
{name: "unknown backend", cfg: observability.MetricsConfig{Enabled: true, Backend: "datadog"}},
|
||||
{
|
||||
name: "backend empty while enabled",
|
||||
cfg: observability.MetricsConfig{Enabled: true, Backend: ""},
|
||||
},
|
||||
{
|
||||
name: "otel missing endpoint",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "grpc", Endpoint: "", ExportInterval: 10 * time.Second, Timeout: 2 * time.Second},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "otel invalid protocol",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "tcp", Endpoint: otelGRPCEndpoint, ExportInterval: 10 * time.Second, Timeout: 2 * time.Second},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "otel zero interval",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "grpc", Endpoint: otelGRPCEndpoint, ExportInterval: 0, Timeout: 2 * time.Second},
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "otel zero timeout",
|
||||
cfg: observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "grpc", Endpoint: otelGRPCEndpoint, ExportInterval: 10 * time.Second, Timeout: 0},
|
||||
},
|
||||
},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if err := tt.cfg.Validate(); err == nil {
|
||||
t.Error("expected validation error but got nil")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewNoopBackend(t *testing.T) {
|
||||
b, err := observability.New(observability.MetricsConfig{Enabled: false})
|
||||
if err != nil {
|
||||
t.Fatalf(errUnexpectedFmt, err)
|
||||
}
|
||||
if b.HTTPHandler() != nil {
|
||||
t.Error("noop backend HTTPHandler should return nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewNoneBackend(t *testing.T) {
|
||||
b, err := observability.New(observability.MetricsConfig{Enabled: true, Backend: "none"})
|
||||
if err != nil {
|
||||
t.Fatalf(errUnexpectedFmt, err)
|
||||
}
|
||||
if b.HTTPHandler() != nil {
|
||||
t.Error("none backend HTTPHandler should return nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewPrometheusBackend(t *testing.T) {
|
||||
cfg := observability.MetricsConfig{
|
||||
Enabled: true, Backend: "prometheus",
|
||||
Prometheus: observability.PrometheusConfig{Path: defaultMetricsPath},
|
||||
}
|
||||
b, err := observability.New(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf(errUnexpectedFmt, err)
|
||||
}
|
||||
if b.HTTPHandler() == nil {
|
||||
t.Error("prometheus backend HTTPHandler should not be nil")
|
||||
}
|
||||
if err := b.Shutdown(context.Background()); err != nil {
|
||||
t.Errorf("prometheus shutdown error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewInvalidBackend(t *testing.T) {
|
||||
_, err := observability.New(observability.MetricsConfig{Enabled: true, Backend: "invalid"})
|
||||
if err == nil {
|
||||
t.Error("expected error for invalid backend")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusAdapterAllInstruments(t *testing.T) {
|
||||
b, err := observability.New(observability.MetricsConfig{
|
||||
Enabled: true, Backend: "prometheus",
|
||||
Prometheus: observability.PrometheusConfig{Path: defaultMetricsPath},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create backend: %v", err)
|
||||
}
|
||||
ctx := context.Background()
|
||||
labels := observability.Labels{"k": "v"}
|
||||
|
||||
c, err := b.NewCounter("prom_adapter_counter_total", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter error: %v", err)
|
||||
}
|
||||
u, err := b.NewUpDownCounter("prom_adapter_updown", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewUpDownCounter error: %v", err)
|
||||
}
|
||||
ig, err := b.NewInt64Gauge("prom_adapter_int_gauge", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewInt64Gauge error: %v", err)
|
||||
}
|
||||
fg, err := b.NewFloat64Gauge("prom_adapter_float_gauge", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewFloat64Gauge error: %v", err)
|
||||
}
|
||||
h, err := b.NewHistogram("prom_adapter_histogram", "desc", []float64{0.1, 1.0}, "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewHistogram error: %v", err)
|
||||
}
|
||||
|
||||
c.Add(ctx, 1, labels)
|
||||
u.Add(ctx, 2, labels)
|
||||
ig.Record(ctx, 99, labels)
|
||||
fg.Record(ctx, 1.23, labels)
|
||||
h.Record(ctx, 0.5, labels)
|
||||
|
||||
if b.HTTPHandler() == nil {
|
||||
t.Error("prometheus adapter HTTPHandler should not be nil")
|
||||
}
|
||||
if err := b.Shutdown(ctx); err != nil {
|
||||
t.Errorf("Shutdown error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtelAdapterAllInstruments(t *testing.T) {
|
||||
if os.Getenv("SKIP_OTEL_INTEGRATION") != "" {
|
||||
t.Skip("skipping OTel integration test because SKIP_OTEL_INTEGRATION is set")
|
||||
}
|
||||
|
||||
dialTimeout := 300 * time.Millisecond
|
||||
conn, err := net.DialTimeout("tcp", otelGRPCEndpoint, dialTimeout)
|
||||
if err != nil {
|
||||
t.Skipf("skipping OTel integration test; collector %s not reachable: %v", otelGRPCEndpoint, err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
|
||||
b, err := observability.New(observability.MetricsConfig{
|
||||
Enabled: true, Backend: "otel",
|
||||
OTel: observability.OTelConfig{Protocol: "grpc", Endpoint: otelGRPCEndpoint, Insecure: true, ExportInterval: 100 * time.Millisecond, Timeout: 2 * time.Second},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create otel backend: %v", err)
|
||||
}
|
||||
ctx := context.Background()
|
||||
labels := observability.Labels{"k": "v"}
|
||||
|
||||
c, err := b.NewCounter("otel_adapter_counter_total", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter error: %v", err)
|
||||
}
|
||||
u, err := b.NewUpDownCounter("otel_adapter_updown", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewUpDownCounter error: %v", err)
|
||||
}
|
||||
ig, err := b.NewInt64Gauge("otel_adapter_int_gauge", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewInt64Gauge error: %v", err)
|
||||
}
|
||||
fg, err := b.NewFloat64Gauge("otel_adapter_float_gauge", "desc", "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewFloat64Gauge error: %v", err)
|
||||
}
|
||||
h, err := b.NewHistogram("otel_adapter_histogram", "desc", []float64{0.1, 1.0}, "k")
|
||||
if err != nil {
|
||||
t.Fatalf("NewHistogram error: %v", err)
|
||||
}
|
||||
|
||||
c.Add(ctx, 1, labels)
|
||||
u.Add(ctx, 2, labels)
|
||||
ig.Record(ctx, 99, labels)
|
||||
fg.Record(ctx, 1.23, labels)
|
||||
h.Record(ctx, 0.5, labels)
|
||||
|
||||
if b.HTTPHandler() != nil {
|
||||
t.Error("OTel adapter HTTPHandler should be nil")
|
||||
}
|
||||
|
||||
shutdownCtx, cancel := context.WithTimeout(ctx, 2*time.Second)
|
||||
defer cancel()
|
||||
b.Shutdown(shutdownCtx) //nolint:errcheck
|
||||
}
|
||||
64
internal/observability/noop.go
Normal file
64
internal/observability/noop.go
Normal file
@@ -0,0 +1,64 @@
|
||||
package observability
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
)
|
||||
|
||||
// NoopBackend is a Backend that discards all observations.
|
||||
// It is used when metrics are disabled (Enabled=false or Backend="none").
|
||||
// All methods are safe to call concurrently.
|
||||
type NoopBackend struct{}
|
||||
|
||||
// Compile-time interface check.
|
||||
var _ Backend = (*NoopBackend)(nil)
|
||||
|
||||
func (n *NoopBackend) NewCounter(_ string, _ string, _ ...string) (Counter, error) {
|
||||
return noopCounter{}, nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) NewUpDownCounter(_ string, _ string, _ ...string) (UpDownCounter, error) {
|
||||
return noopUpDownCounter{}, nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) NewInt64Gauge(_ string, _ string, _ ...string) (Int64Gauge, error) {
|
||||
return noopInt64Gauge{}, nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) NewFloat64Gauge(_ string, _ string, _ ...string) (Float64Gauge, error) {
|
||||
return noopFloat64Gauge{}, nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) NewHistogram(_ string, _ string, _ []float64, _ ...string) (Histogram, error) {
|
||||
return noopHistogram{}, nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) HTTPHandler() http.Handler {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (n *NoopBackend) Shutdown(_ context.Context) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// --- noop instrument types ---
|
||||
|
||||
type noopCounter struct{}
|
||||
|
||||
func (noopCounter) Add(_ context.Context, _ int64, _ Labels) { /* intentionally no-op */ }
|
||||
|
||||
type noopUpDownCounter struct{}
|
||||
|
||||
func (noopUpDownCounter) Add(_ context.Context, _ int64, _ Labels) { /* intentionally no-op */ }
|
||||
|
||||
type noopInt64Gauge struct{}
|
||||
|
||||
func (noopInt64Gauge) Record(_ context.Context, _ int64, _ Labels) { /* intentionally no-op */ }
|
||||
|
||||
type noopFloat64Gauge struct{}
|
||||
|
||||
func (noopFloat64Gauge) Record(_ context.Context, _ float64, _ Labels) { /* intentionally no-op */ }
|
||||
|
||||
type noopHistogram struct{}
|
||||
|
||||
func (noopHistogram) Record(_ context.Context, _ float64, _ Labels) { /* intentionally no-op */ }
|
||||
102
internal/observability/noop_test.go
Normal file
102
internal/observability/noop_test.go
Normal file
@@ -0,0 +1,102 @@
|
||||
package observability_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/observability"
|
||||
)
|
||||
|
||||
func TestNoopBackendAllInstruments(t *testing.T) {
|
||||
n := &observability.NoopBackend{}
|
||||
|
||||
ctx := context.Background()
|
||||
labels := observability.Labels{"k": "v"}
|
||||
|
||||
t.Run("Counter", func(t *testing.T) {
|
||||
c, _ := n.NewCounter("test_counter", "desc")
|
||||
c.Add(ctx, 1, labels)
|
||||
c.Add(ctx, 0, nil)
|
||||
})
|
||||
|
||||
t.Run("UpDownCounter", func(t *testing.T) {
|
||||
u, _ := n.NewUpDownCounter("test_updown", "desc")
|
||||
u.Add(ctx, 1, labels)
|
||||
u.Add(ctx, -1, nil)
|
||||
})
|
||||
|
||||
t.Run("Int64Gauge", func(t *testing.T) {
|
||||
g, _ := n.NewInt64Gauge("test_int64gauge", "desc")
|
||||
g.Record(ctx, 42, labels)
|
||||
g.Record(ctx, 0, nil)
|
||||
})
|
||||
|
||||
t.Run("Float64Gauge", func(t *testing.T) {
|
||||
g, _ := n.NewFloat64Gauge("test_float64gauge", "desc")
|
||||
g.Record(ctx, 3.14, labels)
|
||||
g.Record(ctx, 0, nil)
|
||||
})
|
||||
|
||||
t.Run("Histogram", func(t *testing.T) {
|
||||
h, _ := n.NewHistogram("test_histogram", "desc", []float64{1, 5, 10})
|
||||
h.Record(ctx, 2.5, labels)
|
||||
h.Record(ctx, 0, nil)
|
||||
})
|
||||
|
||||
t.Run("HTTPHandler", func(t *testing.T) {
|
||||
if n.HTTPHandler() != nil {
|
||||
t.Error("noop HTTPHandler should be nil")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("Shutdown", func(t *testing.T) {
|
||||
if err := n.Shutdown(ctx); err != nil {
|
||||
t.Errorf("noop Shutdown should not error: %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestNoopBackendLabelNames(t *testing.T) {
|
||||
// Verify that label names passed at creation time are accepted without panic.
|
||||
n := &observability.NoopBackend{}
|
||||
|
||||
assertNoPanic := func(t *testing.T, constructor string, fn func()) {
|
||||
t.Helper()
|
||||
defer func() {
|
||||
if r := recover(); r != nil {
|
||||
t.Fatalf("%s panicked: %v", constructor, r)
|
||||
}
|
||||
}()
|
||||
fn()
|
||||
}
|
||||
|
||||
t.Run("NewCounter", func(t *testing.T) {
|
||||
assertNoPanic(t, "NewCounter", func() {
|
||||
_, _ = n.NewCounter("c", "d", "label1", "label2")
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("NewUpDownCounter", func(t *testing.T) {
|
||||
assertNoPanic(t, "NewUpDownCounter", func() {
|
||||
_, _ = n.NewUpDownCounter("u", "d", "l1")
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("NewInt64Gauge", func(t *testing.T) {
|
||||
assertNoPanic(t, "NewInt64Gauge", func() {
|
||||
_, _ = n.NewInt64Gauge("g1", "d", "l1", "l2", "l3")
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("NewFloat64Gauge", func(t *testing.T) {
|
||||
assertNoPanic(t, "NewFloat64Gauge", func() {
|
||||
_, _ = n.NewFloat64Gauge("g2", "d")
|
||||
})
|
||||
})
|
||||
|
||||
t.Run("NewHistogram", func(t *testing.T) {
|
||||
assertNoPanic(t, "NewHistogram", func() {
|
||||
_, _ = n.NewHistogram("h", "d", []float64{0.1, 1.0}, "l1")
|
||||
})
|
||||
})
|
||||
}
|
||||
309
internal/observability/otel/backend.go
Normal file
309
internal/observability/otel/backend.go
Normal file
@@ -0,0 +1,309 @@
|
||||
// Package otel implements the OpenTelemetry metrics backend for Gerbil.
|
||||
//
|
||||
// Metrics are exported via OTLP (gRPC or HTTP) to an external collector.
|
||||
// No Prometheus /metrics endpoint is exposed in this mode.
|
||||
// Future OTel tracing and logging can be added alongside this package
|
||||
// without touching the Prometheus-native path.
|
||||
package otel
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"log"
|
||||
"net/http"
|
||||
"regexp"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
"go.opentelemetry.io/otel/metric"
|
||||
sdkmetric "go.opentelemetry.io/otel/sdk/metric"
|
||||
)
|
||||
|
||||
var metricLabelNameRE = regexp.MustCompile(`^[a-zA-Z_][a-zA-Z0-9_]*$`)
|
||||
|
||||
// Config holds OTel backend configuration.
|
||||
type Config struct {
|
||||
// Protocol is "grpc" (default) or "http".
|
||||
Protocol string
|
||||
|
||||
// Endpoint is the OTLP collector address.
|
||||
Endpoint string
|
||||
|
||||
// Insecure disables TLS.
|
||||
Insecure bool
|
||||
|
||||
// ExportInterval is the period between pushes to the collector.
|
||||
ExportInterval time.Duration
|
||||
|
||||
// Timeout bounds exporter construction calls.
|
||||
Timeout time.Duration
|
||||
|
||||
ServiceName string
|
||||
ServiceVersion string
|
||||
DeploymentEnvironment string
|
||||
}
|
||||
|
||||
// Backend is the OTel metrics backend.
|
||||
type Backend struct {
|
||||
cfg Config
|
||||
provider *sdkmetric.MeterProvider
|
||||
meter metric.Meter
|
||||
}
|
||||
|
||||
// New creates and initialises an OTel backend.
|
||||
//
|
||||
// cfg.Protocol must be "grpc" (default) or "http".
|
||||
// cfg.Endpoint is the OTLP collector address (e.g. "localhost:4317").
|
||||
// cfg.ExportInterval sets the push period (defaults to 60 s if ≤ 0).
|
||||
// cfg.Insecure disables TLS on the OTLP connection.
|
||||
//
|
||||
// Connection to the collector is established lazily; New only validates cfg
|
||||
// and creates the SDK components. It returns an error only if the OTel resource
|
||||
// or exporter cannot be constructed.
|
||||
func New(cfg Config) (*Backend, error) {
|
||||
if cfg.Protocol == "" {
|
||||
cfg.Protocol = "grpc"
|
||||
}
|
||||
if strings.TrimSpace(cfg.Endpoint) == "" {
|
||||
return nil, fmt.Errorf("otel backend: empty cfg.Endpoint")
|
||||
}
|
||||
if cfg.ExportInterval <= 0 {
|
||||
cfg.ExportInterval = 60 * time.Second
|
||||
}
|
||||
if cfg.Timeout <= 0 {
|
||||
cfg.Timeout = 10 * time.Second
|
||||
}
|
||||
if cfg.ServiceName == "" {
|
||||
cfg.ServiceName = "gerbil"
|
||||
}
|
||||
|
||||
res, err := newResource(cfg.ServiceName, cfg.ServiceVersion, cfg.DeploymentEnvironment)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel backend: build resource: %w", err)
|
||||
}
|
||||
|
||||
exp, err := newExporter(context.Background(), cfg)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel backend: create exporter: %w", err)
|
||||
}
|
||||
|
||||
reader := sdkmetric.NewPeriodicReader(exp,
|
||||
sdkmetric.WithInterval(cfg.ExportInterval),
|
||||
)
|
||||
|
||||
provider := sdkmetric.NewMeterProvider(
|
||||
sdkmetric.WithResource(res),
|
||||
sdkmetric.WithReader(reader),
|
||||
)
|
||||
|
||||
meter := provider.Meter("github.com/fosrl/gerbil")
|
||||
|
||||
return &Backend{cfg: cfg, provider: provider, meter: meter}, nil
|
||||
}
|
||||
|
||||
// HTTPHandler returns nil – the OTel backend does not expose an HTTP endpoint.
|
||||
func (b *Backend) HTTPHandler() http.Handler {
|
||||
_ = b
|
||||
return nil
|
||||
}
|
||||
|
||||
// Shutdown flushes pending metrics and shuts down the MeterProvider.
|
||||
func (b *Backend) Shutdown(ctx context.Context) error {
|
||||
return b.provider.Shutdown(ctx)
|
||||
}
|
||||
|
||||
// NewCounter creates an OTel Int64Counter.
|
||||
func (b *Backend) NewCounter(name, desc string, labelNames ...string) (*Counter, error) {
|
||||
normalizedLabelNames, err := validateLabelNames(labelNames)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create counter %q: %w", name, err)
|
||||
}
|
||||
c, err := b.meter.Int64Counter(name, metric.WithDescription(desc))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create counter %q: %w", name, err)
|
||||
}
|
||||
return &Counter{c: c, labelNames: normalizedLabelNames}, nil
|
||||
}
|
||||
|
||||
// NewUpDownCounter creates an OTel Int64UpDownCounter.
|
||||
func (b *Backend) NewUpDownCounter(name, desc string, labelNames ...string) (*UpDownCounter, error) {
|
||||
normalizedLabelNames, err := validateLabelNames(labelNames)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create up-down counter %q: %w", name, err)
|
||||
}
|
||||
c, err := b.meter.Int64UpDownCounter(name, metric.WithDescription(desc))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create up-down counter %q: %w", name, err)
|
||||
}
|
||||
return &UpDownCounter{c: c, labelNames: normalizedLabelNames}, nil
|
||||
}
|
||||
|
||||
// NewInt64Gauge creates an OTel Int64Gauge.
|
||||
func (b *Backend) NewInt64Gauge(name, desc string, labelNames ...string) (*Int64Gauge, error) {
|
||||
normalizedLabelNames, err := validateLabelNames(labelNames)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create int64 gauge %q: %w", name, err)
|
||||
}
|
||||
g, err := b.meter.Int64Gauge(name, metric.WithDescription(desc))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create int64 gauge %q: %w", name, err)
|
||||
}
|
||||
return &Int64Gauge{g: g, labelNames: normalizedLabelNames}, nil
|
||||
}
|
||||
|
||||
// NewFloat64Gauge creates an OTel Float64Gauge.
|
||||
func (b *Backend) NewFloat64Gauge(name, desc string, labelNames ...string) (*Float64Gauge, error) {
|
||||
normalizedLabelNames, err := validateLabelNames(labelNames)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create float64 gauge %q: %w", name, err)
|
||||
}
|
||||
g, err := b.meter.Float64Gauge(name, metric.WithDescription(desc))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create float64 gauge %q: %w", name, err)
|
||||
}
|
||||
return &Float64Gauge{g: g, labelNames: normalizedLabelNames}, nil
|
||||
}
|
||||
|
||||
// NewHistogram creates an OTel Float64Histogram with explicit bucket boundaries.
|
||||
func (b *Backend) NewHistogram(name, desc string, buckets []float64, labelNames ...string) (*Histogram, error) {
|
||||
normalizedLabelNames, err := validateLabelNames(labelNames)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create histogram %q: %w", name, err)
|
||||
}
|
||||
h, err := b.meter.Float64Histogram(name,
|
||||
metric.WithDescription(desc),
|
||||
metric.WithExplicitBucketBoundaries(buckets...),
|
||||
)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otel: create histogram %q: %w", name, err)
|
||||
}
|
||||
return &Histogram{h: h, labelNames: normalizedLabelNames}, nil
|
||||
}
|
||||
|
||||
func validateLabelNames(labelNames []string) ([]string, error) {
|
||||
if len(labelNames) == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
normalized := make([]string, len(labelNames))
|
||||
seen := make(map[string]struct{}, len(labelNames))
|
||||
for i, name := range labelNames {
|
||||
if !metricLabelNameRE.MatchString(name) {
|
||||
return nil, fmt.Errorf("invalid label name %q", name)
|
||||
}
|
||||
if _, exists := seen[name]; exists {
|
||||
return nil, fmt.Errorf("duplicate label name %q", name)
|
||||
}
|
||||
seen[name] = struct{}{}
|
||||
normalized[i] = name
|
||||
}
|
||||
|
||||
return normalized, nil
|
||||
}
|
||||
|
||||
func labelsToAttrs(labelNames []string, labels map[string]string) []attribute.KeyValue {
|
||||
if len(labelNames) == 0 {
|
||||
if len(labels) > 0 {
|
||||
log.Printf("WARN: dropping otel metric sample due to unexpected labels: got=%v expected=none", labels)
|
||||
return nil
|
||||
}
|
||||
return []attribute.KeyValue{}
|
||||
}
|
||||
|
||||
attrs := make([]attribute.KeyValue, 0, len(labelNames))
|
||||
for _, labelName := range labelNames {
|
||||
attrs = append(attrs, attribute.String(labelName, labels[labelName]))
|
||||
}
|
||||
|
||||
for got := range labels {
|
||||
found := false
|
||||
for _, expected := range labelNames {
|
||||
if got == expected {
|
||||
found = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
log.Printf("WARN: dropping otel metric sample due to unexpected label key %q (expected=%v)", got, labelNames)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
return attrs
|
||||
}
|
||||
|
||||
// Counter wraps an OTel Int64Counter.
|
||||
type Counter struct {
|
||||
c metric.Int64Counter
|
||||
labelNames []string
|
||||
}
|
||||
|
||||
// Add increments the counter by value.
|
||||
func (c *Counter) Add(ctx context.Context, value int64, labels map[string]string) {
|
||||
attrs := labelsToAttrs(c.labelNames, labels)
|
||||
if attrs == nil {
|
||||
return
|
||||
}
|
||||
c.c.Add(ctx, value, metric.WithAttributes(attrs...))
|
||||
}
|
||||
|
||||
// UpDownCounter wraps an OTel Int64UpDownCounter.
|
||||
type UpDownCounter struct {
|
||||
c metric.Int64UpDownCounter
|
||||
labelNames []string
|
||||
}
|
||||
|
||||
// Add adjusts the up-down counter by value.
|
||||
func (u *UpDownCounter) Add(ctx context.Context, value int64, labels map[string]string) {
|
||||
attrs := labelsToAttrs(u.labelNames, labels)
|
||||
if attrs == nil {
|
||||
return
|
||||
}
|
||||
u.c.Add(ctx, value, metric.WithAttributes(attrs...))
|
||||
}
|
||||
|
||||
// Int64Gauge wraps an OTel Int64Gauge.
|
||||
type Int64Gauge struct {
|
||||
g metric.Int64Gauge
|
||||
labelNames []string
|
||||
}
|
||||
|
||||
// Record sets the gauge to value.
|
||||
func (g *Int64Gauge) Record(ctx context.Context, value int64, labels map[string]string) {
|
||||
attrs := labelsToAttrs(g.labelNames, labels)
|
||||
if attrs == nil {
|
||||
return
|
||||
}
|
||||
g.g.Record(ctx, value, metric.WithAttributes(attrs...))
|
||||
}
|
||||
|
||||
// Float64Gauge wraps an OTel Float64Gauge.
|
||||
type Float64Gauge struct {
|
||||
g metric.Float64Gauge
|
||||
labelNames []string
|
||||
}
|
||||
|
||||
// Record sets the gauge to value.
|
||||
func (g *Float64Gauge) Record(ctx context.Context, value float64, labels map[string]string) {
|
||||
attrs := labelsToAttrs(g.labelNames, labels)
|
||||
if attrs == nil {
|
||||
return
|
||||
}
|
||||
g.g.Record(ctx, value, metric.WithAttributes(attrs...))
|
||||
}
|
||||
|
||||
// Histogram wraps an OTel Float64Histogram.
|
||||
type Histogram struct {
|
||||
h metric.Float64Histogram
|
||||
labelNames []string
|
||||
}
|
||||
|
||||
// Record observes value in the histogram.
|
||||
func (h *Histogram) Record(ctx context.Context, value float64, labels map[string]string) {
|
||||
attrs := labelsToAttrs(h.labelNames, labels)
|
||||
if attrs == nil {
|
||||
return
|
||||
}
|
||||
h.h.Record(ctx, value, metric.WithAttributes(attrs...))
|
||||
}
|
||||
175
internal/observability/otel/backend_test.go
Normal file
175
internal/observability/otel/backend_test.go
Normal file
@@ -0,0 +1,175 @@
|
||||
package otel_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
obsotel "github.com/fosrl/gerbil/internal/observability/otel"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultGRPCEndpoint = "localhost:4317"
|
||||
defaultServiceName = "gerbil-test"
|
||||
)
|
||||
|
||||
func newInMemoryBackend(t *testing.T) *obsotel.Backend {
|
||||
t.Helper()
|
||||
// Use a very short export interval; an in-process collector (noop exporter)
|
||||
// is used by pointing to a non-existent endpoint with insecure mode.
|
||||
// The backend itself should initialise without error since connection is lazy.
|
||||
b, err := obsotel.New(obsotel.Config{
|
||||
Protocol: "grpc",
|
||||
Endpoint: defaultGRPCEndpoint,
|
||||
Insecure: true,
|
||||
ExportInterval: 100 * time.Millisecond,
|
||||
ServiceName: defaultServiceName,
|
||||
ServiceVersion: "0.0.1",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create otel backend: %v", err)
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
func TestOtelBackendHTTPHandlerIsNil(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
if b.HTTPHandler() != nil {
|
||||
t.Error("OTel backend HTTPHandler should return nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtelBackendShutdown(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second)
|
||||
defer cancel()
|
||||
if err := b.Shutdown(ctx); err != nil {
|
||||
// Shutdown with unreachable collector may fail to flush; that's acceptable.
|
||||
// What matters is that Shutdown does not panic.
|
||||
t.Logf("Shutdown returned (expected with no collector): %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtelBackendCounter(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
c, err := b.NewCounter("gerbil_test_counter_total", "test counter", "result")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
// Should not panic
|
||||
c.Add(context.Background(), 1, map[string]string{"result": "ok"})
|
||||
c.Add(context.Background(), 5, nil)
|
||||
}
|
||||
|
||||
func TestOtelBackendUpDownCounter(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
u, err := b.NewUpDownCounter("gerbil_test_updown", "test updown", "state")
|
||||
if err != nil {
|
||||
t.Fatalf("NewUpDownCounter returned error: %v", err)
|
||||
}
|
||||
u.Add(context.Background(), 3, map[string]string{"state": "active"})
|
||||
u.Add(context.Background(), -1, map[string]string{"state": "active"})
|
||||
}
|
||||
|
||||
func TestOtelBackendInt64Gauge(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
g, err := b.NewInt64Gauge("gerbil_test_int_gauge", "test gauge")
|
||||
if err != nil {
|
||||
t.Fatalf("NewInt64Gauge returned error: %v", err)
|
||||
}
|
||||
g.Record(context.Background(), 42, nil)
|
||||
}
|
||||
|
||||
func TestOtelBackendFloat64Gauge(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
g, err := b.NewFloat64Gauge("gerbil_test_float_gauge", "test float gauge")
|
||||
if err != nil {
|
||||
t.Fatalf("NewFloat64Gauge returned error: %v", err)
|
||||
}
|
||||
g.Record(context.Background(), 3.14, nil)
|
||||
}
|
||||
|
||||
func TestOtelBackendHistogram(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
h, err := b.NewHistogram("gerbil_test_duration_seconds", "test histogram",
|
||||
[]float64{0.1, 0.5, 1.0}, "method")
|
||||
if err != nil {
|
||||
t.Fatalf("NewHistogram returned error: %v", err)
|
||||
}
|
||||
h.Record(context.Background(), 0.3, map[string]string{"method": "GET"})
|
||||
}
|
||||
|
||||
func TestOtelBackendHTTPProtocol(t *testing.T) {
|
||||
b, err := obsotel.New(obsotel.Config{
|
||||
Protocol: "http",
|
||||
Endpoint: "localhost:4318",
|
||||
Insecure: true,
|
||||
ExportInterval: 100 * time.Millisecond,
|
||||
ServiceName: defaultServiceName,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create otel http backend: %v", err)
|
||||
}
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
if b.HTTPHandler() != nil {
|
||||
t.Error("OTel HTTP backend should not expose a /metrics endpoint")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtelBackendInvalidProtocol(t *testing.T) {
|
||||
_, err := obsotel.New(obsotel.Config{
|
||||
Protocol: "tcp",
|
||||
Endpoint: defaultGRPCEndpoint,
|
||||
ExportInterval: 10 * time.Second,
|
||||
})
|
||||
if err == nil {
|
||||
t.Error("expected error for invalid protocol")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOtelBackendDeploymentEnvironment(t *testing.T) {
|
||||
b, err := obsotel.New(obsotel.Config{
|
||||
Protocol: "grpc",
|
||||
Endpoint: defaultGRPCEndpoint,
|
||||
Insecure: true,
|
||||
ExportInterval: 100 * time.Millisecond,
|
||||
ServiceName: defaultServiceName,
|
||||
ServiceVersion: "1.2.3",
|
||||
DeploymentEnvironment: "staging",
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
}
|
||||
|
||||
func TestOtelBackendRejectsInvalidLabelNames(t *testing.T) {
|
||||
b := newInMemoryBackend(t)
|
||||
defer b.Shutdown(context.Background()) //nolint:errcheck
|
||||
|
||||
t.Run("duplicate labels", func(t *testing.T) {
|
||||
_, err := b.NewCounter("gerbil_test_invalid_labels_total", "test counter", "result", "result")
|
||||
if err == nil {
|
||||
t.Fatal("expected error for duplicate label names")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("invalid label name", func(t *testing.T) {
|
||||
_, err := b.NewHistogram("gerbil_test_invalid_histogram", "test histogram", []float64{0.1, 1.0}, "status-code")
|
||||
if err == nil {
|
||||
t.Fatal("expected error for invalid label name")
|
||||
}
|
||||
})
|
||||
}
|
||||
68
internal/observability/otel/exporter.go
Normal file
68
internal/observability/otel/exporter.go
Normal file
@@ -0,0 +1,68 @@
|
||||
package otel
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"strings"
|
||||
|
||||
"go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc"
|
||||
"go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp"
|
||||
sdkmetric "go.opentelemetry.io/otel/sdk/metric"
|
||||
)
|
||||
|
||||
// newExporter creates the appropriate OTLP exporter based on cfg.Protocol.
|
||||
func newExporter(ctx context.Context, cfg Config) (sdkmetric.Exporter, error) {
|
||||
if strings.TrimSpace(cfg.Endpoint) == "" {
|
||||
return nil, fmt.Errorf("otel: cfg.Endpoint is empty")
|
||||
}
|
||||
|
||||
switch cfg.Protocol {
|
||||
case "grpc", "":
|
||||
return newGRPCExporter(ctx, cfg)
|
||||
case "http":
|
||||
return newHTTPExporter(ctx, cfg)
|
||||
default:
|
||||
return nil, fmt.Errorf("otel: unknown protocol %q (must be \"grpc\" or \"http\")", cfg.Protocol)
|
||||
}
|
||||
}
|
||||
|
||||
func newGRPCExporter(ctx context.Context, cfg Config) (sdkmetric.Exporter, error) {
|
||||
opts := []otlpmetricgrpc.Option{
|
||||
otlpmetricgrpc.WithEndpoint(cfg.Endpoint),
|
||||
}
|
||||
if cfg.Insecure {
|
||||
opts = append(opts, otlpmetricgrpc.WithInsecure())
|
||||
}
|
||||
exp, err := otlpmetricgrpc.New(ctx, opts...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otlp grpc exporter: %w", err)
|
||||
}
|
||||
return exp, nil
|
||||
}
|
||||
|
||||
func newHTTPExporter(ctx context.Context, cfg Config) (sdkmetric.Exporter, error) {
|
||||
endpoint := strings.TrimSpace(cfg.Endpoint)
|
||||
|
||||
opts := make([]otlpmetrichttp.Option, 0, 3)
|
||||
if strings.Contains(endpoint, "://") {
|
||||
parsed, err := url.Parse(endpoint)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otlp http exporter: parse endpoint URL %q: %w", endpoint, err)
|
||||
}
|
||||
opts = append(opts, otlpmetrichttp.WithEndpointURL(parsed.String()))
|
||||
} else {
|
||||
opts = append(opts,
|
||||
otlpmetrichttp.WithEndpoint(endpoint),
|
||||
otlpmetrichttp.WithURLPath("/v1/metrics"),
|
||||
)
|
||||
}
|
||||
if cfg.Insecure {
|
||||
opts = append(opts, otlpmetrichttp.WithInsecure())
|
||||
}
|
||||
exp, err := otlpmetrichttp.New(ctx, opts...)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("otlp http exporter: %w", err)
|
||||
}
|
||||
return exp, nil
|
||||
}
|
||||
25
internal/observability/otel/resource.go
Normal file
25
internal/observability/otel/resource.go
Normal file
@@ -0,0 +1,25 @@
|
||||
package otel
|
||||
|
||||
import (
|
||||
"go.opentelemetry.io/otel/attribute"
|
||||
"go.opentelemetry.io/otel/sdk/resource"
|
||||
semconv "go.opentelemetry.io/otel/semconv/v1.18.0"
|
||||
)
|
||||
|
||||
// newResource builds an OTel resource for the Gerbil service.
|
||||
func newResource(serviceName, serviceVersion, deploymentEnv string) (*resource.Resource, error) {
|
||||
attrs := []attribute.KeyValue{
|
||||
semconv.ServiceName(serviceName),
|
||||
}
|
||||
if serviceVersion != "" {
|
||||
attrs = append(attrs, semconv.ServiceVersion(serviceVersion))
|
||||
}
|
||||
if deploymentEnv != "" {
|
||||
attrs = append(attrs, semconv.DeploymentEnvironment(deploymentEnv))
|
||||
}
|
||||
|
||||
return resource.Merge(
|
||||
resource.Default(),
|
||||
resource.NewSchemaless(attrs...),
|
||||
)
|
||||
}
|
||||
310
internal/observability/prometheus/backend.go
Normal file
310
internal/observability/prometheus/backend.go
Normal file
@@ -0,0 +1,310 @@
|
||||
// Package prometheus implements the native Prometheus metrics backend for Gerbil.
|
||||
//
|
||||
// This backend uses the Prometheus Go client directly; it does NOT depend on the
|
||||
// OpenTelemetry SDK. A dedicated Prometheus registry is used so that default
|
||||
// Go/process metrics are not unintentionally included unless the caller opts in.
|
||||
package prometheus
|
||||
|
||||
import (
|
||||
"context"
|
||||
"log"
|
||||
"net/http"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/collectors"
|
||||
"github.com/prometheus/client_golang/prometheus/promhttp"
|
||||
)
|
||||
|
||||
// Config holds Prometheus-backend configuration.
|
||||
type Config struct {
|
||||
// Path is the HTTP endpoint path (e.g. "/metrics").
|
||||
Path string
|
||||
|
||||
// IncludeGoMetrics controls whether the standard Go runtime and process
|
||||
// collectors are registered on the dedicated registry.
|
||||
// Defaults to true if not explicitly set.
|
||||
IncludeGoMetrics *bool
|
||||
}
|
||||
|
||||
// Backend is the native Prometheus metrics backend.
|
||||
// Metric instruments are created via the New* family of methods and stored
|
||||
// in the backend-specific instrument types that implement the observability
|
||||
// instrument interfaces.
|
||||
type Backend struct {
|
||||
cfg Config
|
||||
registry *prometheus.Registry
|
||||
handler http.Handler
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// New creates and initialises a Prometheus backend.
|
||||
//
|
||||
// cfg.Path sets the HTTP endpoint path (defaults to "/metrics" if empty).
|
||||
// cfg.IncludeGoMetrics controls whether standard Go runtime and process metrics
|
||||
// are included; defaults to true when nil.
|
||||
//
|
||||
// Returns an error if the registry cannot be created.
|
||||
func New(cfg Config) (*Backend, error) {
|
||||
if cfg.Path == "" {
|
||||
cfg.Path = "/metrics"
|
||||
}
|
||||
|
||||
registry := prometheus.NewRegistry()
|
||||
droppedSamplesCounter := prometheus.NewCounter(prometheus.CounterOpts{
|
||||
Name: "gerbil_dropped_metric_samples_total",
|
||||
Help: "Total number of metric samples dropped due to invalid labels or unsupported label sets",
|
||||
})
|
||||
registry.MustRegister(droppedSamplesCounter)
|
||||
|
||||
// Include Go and process metrics by default.
|
||||
includeGo := cfg.IncludeGoMetrics == nil || *cfg.IncludeGoMetrics
|
||||
if includeGo {
|
||||
registry.MustRegister(
|
||||
collectors.NewGoCollector(),
|
||||
collectors.NewProcessCollector(collectors.ProcessCollectorOpts{}),
|
||||
)
|
||||
}
|
||||
|
||||
handler := promhttp.HandlerFor(registry, promhttp.HandlerOpts{
|
||||
EnableOpenMetrics: false,
|
||||
})
|
||||
|
||||
return &Backend{cfg: cfg, registry: registry, handler: handler, droppedSamplesCounter: droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// HTTPHandler returns the Prometheus /metrics HTTP handler.
|
||||
func (b *Backend) HTTPHandler() http.Handler {
|
||||
return b.handler
|
||||
}
|
||||
|
||||
// Shutdown is a no-op for the Prometheus backend.
|
||||
// The registry does not maintain background goroutines.
|
||||
func (b *Backend) Shutdown(_ context.Context) error {
|
||||
_ = b
|
||||
return nil
|
||||
}
|
||||
|
||||
// NewCounter creates a Prometheus CounterVec registered on the backend's registry.
|
||||
func (b *Backend) NewCounter(name, desc string, labelNames ...string) (*Counter, error) {
|
||||
vec := prometheus.NewCounterVec(prometheus.CounterOpts{
|
||||
Name: name,
|
||||
Help: desc,
|
||||
}, labelNames)
|
||||
if err := b.registry.Register(vec); err != nil {
|
||||
if are, ok := err.(prometheus.AlreadyRegisteredError); ok {
|
||||
existing, ok := are.ExistingCollector.(*prometheus.CounterVec)
|
||||
if !ok {
|
||||
return nil, err
|
||||
}
|
||||
return &Counter{vec: existing, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return &Counter{vec: vec, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// NewUpDownCounter creates a Prometheus GaugeVec (Prometheus gauges are
|
||||
// bidirectional) registered on the backend's registry.
|
||||
func (b *Backend) NewUpDownCounter(name, desc string, labelNames ...string) (*UpDownCounter, error) {
|
||||
vec := prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: name,
|
||||
Help: desc,
|
||||
}, labelNames)
|
||||
if err := b.registry.Register(vec); err != nil {
|
||||
if are, ok := err.(prometheus.AlreadyRegisteredError); ok {
|
||||
existing, ok := are.ExistingCollector.(*prometheus.GaugeVec)
|
||||
if !ok {
|
||||
return nil, err
|
||||
}
|
||||
return &UpDownCounter{vec: existing, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return &UpDownCounter{vec: vec, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// NewInt64Gauge creates a Prometheus GaugeVec registered on the backend's registry.
|
||||
func (b *Backend) NewInt64Gauge(name, desc string, labelNames ...string) (*Int64Gauge, error) {
|
||||
vec := prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: name,
|
||||
Help: desc,
|
||||
}, labelNames)
|
||||
if err := b.registry.Register(vec); err != nil {
|
||||
if are, ok := err.(prometheus.AlreadyRegisteredError); ok {
|
||||
existing, ok := are.ExistingCollector.(*prometheus.GaugeVec)
|
||||
if !ok {
|
||||
return nil, err
|
||||
}
|
||||
return &Int64Gauge{vec: existing, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return &Int64Gauge{vec: vec, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// NewFloat64Gauge creates a Prometheus GaugeVec registered on the backend's registry.
|
||||
func (b *Backend) NewFloat64Gauge(name, desc string, labelNames ...string) (*Float64Gauge, error) {
|
||||
vec := prometheus.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: name,
|
||||
Help: desc,
|
||||
}, labelNames)
|
||||
if err := b.registry.Register(vec); err != nil {
|
||||
if are, ok := err.(prometheus.AlreadyRegisteredError); ok {
|
||||
existing, ok := are.ExistingCollector.(*prometheus.GaugeVec)
|
||||
if !ok {
|
||||
return nil, err
|
||||
}
|
||||
return &Float64Gauge{vec: existing, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return &Float64Gauge{vec: vec, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// NewHistogram creates a Prometheus HistogramVec registered on the backend's registry.
|
||||
func (b *Backend) NewHistogram(name, desc string, buckets []float64, labelNames ...string) (*Histogram, error) {
|
||||
vec := prometheus.NewHistogramVec(prometheus.HistogramOpts{
|
||||
Name: name,
|
||||
Help: desc,
|
||||
Buckets: buckets,
|
||||
}, labelNames)
|
||||
if err := b.registry.Register(vec); err != nil {
|
||||
if are, ok := err.(prometheus.AlreadyRegisteredError); ok {
|
||||
existing, ok := are.ExistingCollector.(*prometheus.HistogramVec)
|
||||
if !ok {
|
||||
return nil, err
|
||||
}
|
||||
return &Histogram{vec: existing, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
return &Histogram{vec: vec, labelNames: append([]string(nil), labelNames...), droppedSamplesCounter: b.droppedSamplesCounter}, nil
|
||||
}
|
||||
|
||||
// Counter is a native Prometheus counter instrument.
|
||||
type Counter struct {
|
||||
vec *prometheus.CounterVec
|
||||
labelNames []string
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// Add increments the counter by value for the given labels.
|
||||
//
|
||||
// value must be non-negative. Negative values are ignored.
|
||||
func (c *Counter) Add(_ context.Context, value int64, labels map[string]string) {
|
||||
if value < 0 {
|
||||
log.Printf("WARN: counter add called with negative value=%d labels=%v expected_labels=%v", value, labels, c.labelNames)
|
||||
return
|
||||
}
|
||||
normalized, ok := normalizeLabels(c.labelNames, labels, c.droppedSamplesCounter)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
defer guardMetricPanic("counter", c.labelNames, labels)
|
||||
c.vec.With(normalized).Add(float64(value))
|
||||
}
|
||||
|
||||
// UpDownCounter is a native Prometheus gauge used as a bidirectional counter.
|
||||
type UpDownCounter struct {
|
||||
vec *prometheus.GaugeVec
|
||||
labelNames []string
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// Add adjusts the gauge by value for the given labels.
|
||||
func (u *UpDownCounter) Add(_ context.Context, value int64, labels map[string]string) {
|
||||
normalized, ok := normalizeLabels(u.labelNames, labels, u.droppedSamplesCounter)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
defer guardMetricPanic("updown", u.labelNames, labels)
|
||||
u.vec.With(normalized).Add(float64(value))
|
||||
}
|
||||
|
||||
// Int64Gauge is a native Prometheus gauge recording integer snapshot values.
|
||||
type Int64Gauge struct {
|
||||
vec *prometheus.GaugeVec
|
||||
labelNames []string
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// Record sets the gauge to value for the given labels.
|
||||
func (g *Int64Gauge) Record(_ context.Context, value int64, labels map[string]string) {
|
||||
normalized, ok := normalizeLabels(g.labelNames, labels, g.droppedSamplesCounter)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
defer guardMetricPanic("int64-gauge", g.labelNames, labels)
|
||||
g.vec.With(normalized).Set(float64(value))
|
||||
}
|
||||
|
||||
// Float64Gauge is a native Prometheus gauge recording float snapshot values.
|
||||
type Float64Gauge struct {
|
||||
vec *prometheus.GaugeVec
|
||||
labelNames []string
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// Record sets the gauge to value for the given labels.
|
||||
func (g *Float64Gauge) Record(_ context.Context, value float64, labels map[string]string) {
|
||||
normalized, ok := normalizeLabels(g.labelNames, labels, g.droppedSamplesCounter)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
defer guardMetricPanic("float64-gauge", g.labelNames, labels)
|
||||
g.vec.With(normalized).Set(value)
|
||||
}
|
||||
|
||||
// Histogram is a native Prometheus histogram instrument.
|
||||
type Histogram struct {
|
||||
vec *prometheus.HistogramVec
|
||||
labelNames []string
|
||||
droppedSamplesCounter prometheus.Counter
|
||||
}
|
||||
|
||||
// Record observes value for the given labels.
|
||||
func (h *Histogram) Record(_ context.Context, value float64, labels map[string]string) {
|
||||
normalized, ok := normalizeLabels(h.labelNames, labels, h.droppedSamplesCounter)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
defer guardMetricPanic("histogram", h.labelNames, labels)
|
||||
h.vec.With(normalized).Observe(value)
|
||||
}
|
||||
|
||||
func normalizeLabels(labelNames []string, labels map[string]string, droppedSamplesCounter prometheus.Counter) (prometheus.Labels, bool) {
|
||||
if len(labelNames) == 0 {
|
||||
if len(labels) > 0 {
|
||||
if droppedSamplesCounter != nil {
|
||||
droppedSamplesCounter.Inc()
|
||||
}
|
||||
log.Printf("WARN: dropping metric sample due to unexpected labels: got=%v expected=none", labels)
|
||||
return nil, false
|
||||
}
|
||||
return nil, true
|
||||
}
|
||||
|
||||
normalized := make(prometheus.Labels, len(labelNames))
|
||||
for _, name := range labelNames {
|
||||
normalized[name] = ""
|
||||
}
|
||||
|
||||
for k, v := range labels {
|
||||
if _, ok := normalized[k]; !ok {
|
||||
if droppedSamplesCounter != nil {
|
||||
droppedSamplesCounter.Inc()
|
||||
}
|
||||
log.Printf("WARN: dropping metric sample due to unexpected label key %q (expected=%v)", k, labelNames)
|
||||
return nil, false
|
||||
}
|
||||
normalized[k] = v
|
||||
}
|
||||
|
||||
return normalized, true
|
||||
}
|
||||
|
||||
func guardMetricPanic(kind string, expected []string, labels map[string]string) {
|
||||
if recovered := recover(); recovered != nil {
|
||||
log.Printf("WARN: dropped %s metric sample due to label panic: expected=%v got=%v err=%v", kind, expected, labels, recovered)
|
||||
}
|
||||
}
|
||||
231
internal/observability/prometheus/backend_test.go
Normal file
231
internal/observability/prometheus/backend_test.go
Normal file
@@ -0,0 +1,231 @@
|
||||
package prometheus_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
obsprom "github.com/fosrl/gerbil/internal/observability/prometheus"
|
||||
)
|
||||
|
||||
func newTestBackend(t *testing.T) *obsprom.Backend {
|
||||
t.Helper()
|
||||
b, err := obsprom.New(obsprom.Config{Path: "/metrics"})
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create prometheus backend: %v", err)
|
||||
}
|
||||
return b
|
||||
}
|
||||
|
||||
func TestPrometheusBackendHTTPHandler(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
if b.HTTPHandler() == nil {
|
||||
t.Error("HTTPHandler should not be nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendShutdown(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
if err := b.Shutdown(context.Background()); err != nil {
|
||||
t.Errorf("Shutdown returned error: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendCounter(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
c, err := b.NewCounter("test_counter_total", "A test counter", "result")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
c.Add(context.Background(), 3, map[string]string{"result": "ok"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `test_counter_total{result="ok"} 3`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendUpDownCounter(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
u, err := b.NewUpDownCounter("test_gauge_total", "A test up-down counter", "state")
|
||||
if err != nil {
|
||||
t.Fatalf("NewUpDownCounter returned error: %v", err)
|
||||
}
|
||||
u.Add(context.Background(), 5, map[string]string{"state": "active"})
|
||||
u.Add(context.Background(), -2, map[string]string{"state": "active"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `test_gauge_total{state="active"} 3`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendInt64Gauge(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
g, err := b.NewInt64Gauge("test_int_gauge", "An integer gauge", "ifname")
|
||||
if err != nil {
|
||||
t.Fatalf("NewInt64Gauge returned error: %v", err)
|
||||
}
|
||||
g.Record(context.Background(), 42, map[string]string{"ifname": "wg0"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `test_int_gauge{ifname="wg0"} 42`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendFloat64Gauge(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
g, err := b.NewFloat64Gauge("test_float_gauge", "A float gauge", "cert")
|
||||
if err != nil {
|
||||
t.Fatalf("NewFloat64Gauge returned error: %v", err)
|
||||
}
|
||||
g.Record(context.Background(), 7.5, map[string]string{"cert": "example.com"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `test_float_gauge{cert="example.com"} 7.5`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendHistogram(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
buckets := []float64{0.1, 0.5, 1.0, 5.0}
|
||||
h, err := b.NewHistogram("test_duration_seconds", "A test histogram", buckets, "method")
|
||||
if err != nil {
|
||||
t.Fatalf("NewHistogram returned error: %v", err)
|
||||
}
|
||||
h.Record(context.Background(), 0.3, map[string]string{"method": "GET"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
if !strings.Contains(body, "test_duration_seconds") {
|
||||
t.Errorf("expected histogram metric in output, body:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendMultipleLabels(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
c, err := b.NewCounter("multi_label_total", "Multi-label counter", "method", "route", "status_code")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
c.Add(context.Background(), 1, map[string]string{
|
||||
"method": "POST",
|
||||
"route": "/api/peers",
|
||||
"status_code": "200",
|
||||
})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
if !strings.Contains(body, "multi_label_total") {
|
||||
t.Errorf("expected multi_label_total in output, body:\n%s", body)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendGoMetrics(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
body := scrapeMetrics(t, b)
|
||||
// Default backend includes Go runtime metrics.
|
||||
if !strings.Contains(body, "go_goroutines") {
|
||||
t.Error("expected go_goroutines in default backend output")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendNoGoMetrics(t *testing.T) {
|
||||
f := false
|
||||
b, err := obsprom.New(obsprom.Config{IncludeGoMetrics: &f})
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected error: %v", err)
|
||||
}
|
||||
body := scrapeMetrics(t, b)
|
||||
if strings.Contains(body, "go_goroutines") {
|
||||
t.Error("expected no go_goroutines when IncludeGoMetrics=false")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrometheusBackendNilLabels(t *testing.T) {
|
||||
// Adding with nil labels should not panic (treated as empty map).
|
||||
b := newTestBackend(t)
|
||||
c, err := b.NewCounter("nil_labels_total", "counter with no labels")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
// nil labels with no label names declared should be safe
|
||||
c.Add(context.Background(), 1, nil)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendConcurrentAdd(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
c, err := b.NewCounter("concurrent_total", "concurrent counter", "worker")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
for i := 0; i < 10; i++ {
|
||||
go func() {
|
||||
for j := 0; j < 100; j++ {
|
||||
c.Add(context.Background(), 1, map[string]string{"worker": "w"})
|
||||
}
|
||||
done <- struct{}{}
|
||||
}()
|
||||
}
|
||||
for i := 0; i < 10; i++ {
|
||||
<-done
|
||||
}
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `concurrent_total{worker="w"} 1000`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendAlreadyRegisteredCounter(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
c1, err := b.NewCounter("dupe_counter_total", "duplicate counter", "result")
|
||||
if err != nil {
|
||||
t.Fatalf("first NewCounter returned error: %v", err)
|
||||
}
|
||||
c2, err := b.NewCounter("dupe_counter_total", "duplicate counter", "result")
|
||||
if err != nil {
|
||||
t.Fatalf("second NewCounter returned error: %v", err)
|
||||
}
|
||||
|
||||
c1.Add(context.Background(), 1, map[string]string{"result": "ok"})
|
||||
c2.Add(context.Background(), 2, map[string]string{"result": "ok"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
assertMetricPresent(t, body, `dupe_counter_total{result="ok"} 3`)
|
||||
}
|
||||
|
||||
func TestPrometheusBackendInvalidLabelsNoPanic(t *testing.T) {
|
||||
b := newTestBackend(t)
|
||||
c, err := b.NewCounter("invalid_labels_total", "invalid labels test", "result")
|
||||
if err != nil {
|
||||
t.Fatalf("NewCounter returned error: %v", err)
|
||||
}
|
||||
|
||||
// Extra label key should be dropped and must not panic.
|
||||
c.Add(context.Background(), 5, map[string]string{"result": "ok", "unexpected": "x"})
|
||||
|
||||
body := scrapeMetrics(t, b)
|
||||
if strings.Contains(body, `invalid_labels_total{result="ok"}`) {
|
||||
t.Error("invalid label sample should have been dropped")
|
||||
}
|
||||
}
|
||||
|
||||
// --- helpers ---
|
||||
|
||||
func scrapeMetrics(t *testing.T, b *obsprom.Backend) string {
|
||||
t.Helper()
|
||||
req := httptest.NewRequest(http.MethodGet, "/metrics", http.NoBody)
|
||||
rr := httptest.NewRecorder()
|
||||
b.HTTPHandler().ServeHTTP(rr, req)
|
||||
if rr.Code != http.StatusOK {
|
||||
t.Fatalf("metrics handler returned %d", rr.Code)
|
||||
}
|
||||
body, err := io.ReadAll(rr.Body)
|
||||
if err != nil {
|
||||
t.Fatalf("failed to read response body: %v", err)
|
||||
}
|
||||
return string(body)
|
||||
}
|
||||
|
||||
func assertMetricPresent(t *testing.T, body, expected string) {
|
||||
t.Helper()
|
||||
if !strings.Contains(body, expected) {
|
||||
t.Errorf("expected %q in metrics output\nbody:\n%s", expected, body)
|
||||
}
|
||||
}
|
||||
452
proxy/proxy.go
452
proxy/proxy.go
@@ -11,15 +11,58 @@ import (
|
||||
"log"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/metrics"
|
||||
"github.com/fosrl/gerbil/logger"
|
||||
"github.com/fosrl/gerbil/proxyproto"
|
||||
"github.com/patrickmn/go-cache"
|
||||
)
|
||||
|
||||
// defaultMaxSNIConnections caps the number of concurrent client connections
|
||||
// the SNI proxy will accept. Without a cap, a burst of connections (a
|
||||
// scanner sweep, a client reconnect storm) spawns unbounded goroutines each
|
||||
// holding copy buffers and sockets, exhausting container memory faster than
|
||||
// GC/backpressure can catch up.
|
||||
// Sized conservatively since each connection now holds up to two pooled
|
||||
// 32KB copy buffers once the buffer pool is actually honored (see
|
||||
// bufferedReader/bufferedWriter below). Overridable via
|
||||
// GERBIL_MAX_SNI_CONNECTIONS.
|
||||
const defaultMaxSNIConnections = 4096
|
||||
|
||||
var maxSNIConnections = loadMaxSNIConnections()
|
||||
|
||||
func loadMaxSNIConnections() int64 {
|
||||
if v := os.Getenv("GERBIL_MAX_SNI_CONNECTIONS"); v != "" {
|
||||
if n, err := strconv.ParseInt(v, 10, 64); err == nil && n > 0 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return defaultMaxSNIConnections
|
||||
}
|
||||
|
||||
// defaultMaxSNIConnectionsPerIP caps concurrent connections from a single
|
||||
// source IP, independent of the global maxSNIConnections budget. Without
|
||||
// this, one noisy/misbehaving client (or a single scanning host) can
|
||||
// consume the entire global budget and lock out every other customer
|
||||
// sharing this proxy. Overridable via GERBIL_MAX_SNI_CONNECTIONS_PER_IP.
|
||||
const defaultMaxSNIConnectionsPerIP = 256
|
||||
|
||||
var maxSNIConnectionsPerIP = loadMaxSNIConnectionsPerIP()
|
||||
|
||||
func loadMaxSNIConnectionsPerIP() int64 {
|
||||
if v := os.Getenv("GERBIL_MAX_SNI_CONNECTIONS_PER_IP"); v != "" {
|
||||
if n, err := strconv.ParseInt(v, 10, 64); err == nil && n > 0 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return defaultMaxSNIConnectionsPerIP
|
||||
}
|
||||
|
||||
// RouteRecord represents a routing configuration
|
||||
type RouteRecord struct {
|
||||
Hostname string
|
||||
@@ -32,6 +75,16 @@ type RouteAPIResponse struct {
|
||||
Endpoints []string `json:"endpoints"`
|
||||
}
|
||||
|
||||
// ProxyProtocolInfo holds information parsed from incoming PROXY protocol header
|
||||
type ProxyProtocolInfo struct {
|
||||
Protocol string // TCP4 or TCP6
|
||||
SrcIP string
|
||||
DestIP string
|
||||
SrcPort int
|
||||
DestPort int
|
||||
OriginalConn net.Conn // The original connection after PROXY protocol parsing
|
||||
}
|
||||
|
||||
// SNIProxy represents the main proxy server
|
||||
type SNIProxy struct {
|
||||
port int
|
||||
@@ -59,6 +112,24 @@ type SNIProxy struct {
|
||||
|
||||
// Trusted upstream proxies that can send PROXY protocol
|
||||
trustedUpstreams map[string]struct{}
|
||||
|
||||
// Reusable HTTP client for API requests
|
||||
httpClient *http.Client
|
||||
|
||||
// Buffer pool for connection piping
|
||||
bufferPool *sync.Pool
|
||||
|
||||
// activeConnections tracks concurrent client connections so
|
||||
// acceptConnections can enforce maxSNIConnections.
|
||||
activeConnections atomic.Int64
|
||||
|
||||
// perIPConnections tracks concurrent connections per source IP (map[string]*atomic.Int64)
|
||||
// so acceptConnections can enforce maxSNIConnectionsPerIP and stop one
|
||||
// client from starving the rest. Entries are removed once a given IP's
|
||||
// count returns to zero, so this stays bounded by currently-connected
|
||||
// distinct IPs (itself bounded by maxSNIConnections) rather than growing
|
||||
// with every IP ever seen.
|
||||
perIPConnections sync.Map
|
||||
}
|
||||
|
||||
type activeTunnel struct {
|
||||
@@ -79,6 +150,249 @@ func (conn readOnlyConn) SetDeadline(t time.Time) error { return nil }
|
||||
func (conn readOnlyConn) SetReadDeadline(t time.Time) error { return nil }
|
||||
func (conn readOnlyConn) SetWriteDeadline(t time.Time) error { return nil }
|
||||
|
||||
// parseProxyProtocolHeader parses a PROXY protocol v1 header from the connection
|
||||
func (p *SNIProxy) parseProxyProtocolHeader(conn net.Conn) (*ProxyProtocolInfo, net.Conn, error) {
|
||||
// Check if the connection comes from a trusted upstream
|
||||
remoteHost, _, err := net.SplitHostPort(conn.RemoteAddr().String())
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("failed to parse remote address: %w", err)
|
||||
}
|
||||
|
||||
// Resolve the remote IP to hostname to check if it's trusted
|
||||
// For simplicity, we'll check the IP directly in trusted upstreams
|
||||
// In production, you might want to do reverse DNS lookup
|
||||
if _, isTrusted := p.trustedUpstreams[remoteHost]; !isTrusted {
|
||||
// Not from trusted upstream, return original connection
|
||||
return nil, conn, nil
|
||||
}
|
||||
|
||||
// Set read timeout for PROXY protocol parsing
|
||||
if err := conn.SetReadDeadline(time.Now().Add(5 * time.Second)); err != nil {
|
||||
return nil, conn, fmt.Errorf("failed to set read deadline: %w", err)
|
||||
}
|
||||
|
||||
// Read the first line (PROXY protocol header)
|
||||
buffer := make([]byte, 512) // PROXY protocol header should be much smaller
|
||||
n, err := conn.Read(buffer)
|
||||
if err != nil {
|
||||
// If we can't read from trusted upstream, treat as regular connection
|
||||
logger.Debug("Could not read from trusted upstream %s, treating as regular connection: %v", remoteHost, err)
|
||||
// Clear read timeout before returning
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", clearErr)
|
||||
}
|
||||
return nil, conn, nil
|
||||
}
|
||||
|
||||
// Find the end of the first line (CRLF)
|
||||
headerEnd := bytes.Index(buffer[:n], []byte("\r\n"))
|
||||
if headerEnd == -1 {
|
||||
// No PROXY protocol header found, treat as regular TLS connection
|
||||
// Return the connection with the buffered data prepended
|
||||
logger.Debug("No PROXY protocol header from trusted upstream %s, treating as regular TLS connection", remoteHost)
|
||||
|
||||
// Clear read timeout
|
||||
if err := conn.SetReadDeadline(time.Time{}); err != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", err)
|
||||
}
|
||||
|
||||
// Create a reader that includes the buffered data + original connection
|
||||
newReader := io.MultiReader(bytes.NewReader(buffer[:n]), conn)
|
||||
wrappedConn := &proxyProtocolConn{
|
||||
Conn: conn,
|
||||
reader: newReader,
|
||||
}
|
||||
return nil, wrappedConn, nil
|
||||
}
|
||||
|
||||
headerLine := string(buffer[:headerEnd])
|
||||
remainingData := buffer[headerEnd+2 : n]
|
||||
|
||||
// Parse PROXY protocol line: "PROXY TCP4/TCP6 srcIP destIP srcPort destPort"
|
||||
parts := strings.Fields(headerLine)
|
||||
if len(parts) != 6 || parts[0] != "PROXY" {
|
||||
// Check for PROXY UNKNOWN
|
||||
if len(parts) == 2 && parts[0] == "PROXY" && parts[1] == "UNKNOWN" {
|
||||
// PROXY UNKNOWN - use original connection info
|
||||
return nil, conn, nil
|
||||
}
|
||||
// Invalid PROXY protocol, but might be regular TLS - treat as such
|
||||
logger.Debug("Invalid PROXY protocol from trusted upstream %s, treating as regular TLS connection: %s", remoteHost, headerLine)
|
||||
|
||||
// Clear read timeout
|
||||
if err := conn.SetReadDeadline(time.Time{}); err != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", err)
|
||||
}
|
||||
|
||||
// Return the connection with all buffered data prepended
|
||||
newReader := io.MultiReader(bytes.NewReader(buffer[:n]), conn)
|
||||
wrappedConn := &proxyProtocolConn{
|
||||
Conn: conn,
|
||||
reader: newReader,
|
||||
}
|
||||
return nil, wrappedConn, nil
|
||||
}
|
||||
|
||||
protocol := parts[1]
|
||||
srcIP := parts[2]
|
||||
destIP := parts[3]
|
||||
srcPort, err := strconv.Atoi(parts[4])
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("invalid source port in PROXY header: %s", parts[4])
|
||||
}
|
||||
destPort, err := strconv.Atoi(parts[5])
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("invalid destination port in PROXY header: %s", parts[5])
|
||||
}
|
||||
|
||||
// Create a new reader that includes remaining data + original connection
|
||||
var newReader io.Reader
|
||||
if len(remainingData) > 0 {
|
||||
newReader = io.MultiReader(bytes.NewReader(remainingData), conn)
|
||||
} else {
|
||||
newReader = conn
|
||||
}
|
||||
|
||||
// Create a wrapper connection that reads from the combined reader
|
||||
wrappedConn := &proxyProtocolConn{
|
||||
Conn: conn,
|
||||
reader: newReader,
|
||||
}
|
||||
|
||||
proxyInfo := &ProxyProtocolInfo{
|
||||
Protocol: protocol,
|
||||
SrcIP: srcIP,
|
||||
DestIP: destIP,
|
||||
SrcPort: srcPort,
|
||||
DestPort: destPort,
|
||||
OriginalConn: wrappedConn,
|
||||
}
|
||||
|
||||
// Clear read timeout
|
||||
if err := conn.SetReadDeadline(time.Time{}); err != nil {
|
||||
return nil, conn, fmt.Errorf("failed to clear read deadline: %w", err)
|
||||
}
|
||||
|
||||
return proxyInfo, wrappedConn, nil
|
||||
}
|
||||
|
||||
// proxyProtocolConn wraps a connection to read from a custom reader
|
||||
type proxyProtocolConn struct {
|
||||
net.Conn
|
||||
reader io.Reader
|
||||
}
|
||||
|
||||
func (c *proxyProtocolConn) Read(b []byte) (int, error) {
|
||||
return c.reader.Read(b)
|
||||
}
|
||||
|
||||
// buildProxyProtocolHeaderFromInfo creates a PROXY protocol v1 header using ProxyProtocolInfo
|
||||
func (p *SNIProxy) buildProxyProtocolHeaderFromInfo(proxyInfo *ProxyProtocolInfo, targetAddr net.Addr) string {
|
||||
targetTCP, ok := targetAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
// Fallback for unknown address types
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
// Use the original client information from the PROXY protocol
|
||||
var targetIP string
|
||||
var protocol string
|
||||
|
||||
// Parse source IP to determine protocol family
|
||||
srcIP := net.ParseIP(proxyInfo.SrcIP)
|
||||
if srcIP == nil {
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
if srcIP.To4() != nil {
|
||||
// Source is IPv4, use TCP4 protocol
|
||||
protocol = "TCP4"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
// Target is also IPv4, use as-is
|
||||
targetIP = targetTCP.IP.String()
|
||||
} else {
|
||||
// Target is IPv6, but we need IPv4 for consistent protocol family
|
||||
if targetTCP.IP.IsLoopback() {
|
||||
targetIP = "127.0.0.1"
|
||||
} else {
|
||||
targetIP = "127.0.0.1" // Safe fallback
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Source is IPv6, use TCP6 protocol
|
||||
protocol = "TCP6"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
// Target is IPv4, convert to IPv6 representation
|
||||
targetIP = "::ffff:" + targetTCP.IP.String()
|
||||
} else {
|
||||
// Target is also IPv6, use as-is
|
||||
targetIP = targetTCP.IP.String()
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Sprintf("PROXY %s %s %s %d %d\r\n",
|
||||
protocol,
|
||||
proxyInfo.SrcIP,
|
||||
targetIP,
|
||||
proxyInfo.SrcPort,
|
||||
targetTCP.Port)
|
||||
}
|
||||
|
||||
// buildProxyProtocolHeader creates a PROXY protocol v1 header
|
||||
func buildProxyProtocolHeader(clientAddr, targetAddr net.Addr) string {
|
||||
clientTCP, ok := clientAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
// Fallback for unknown address types
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
targetTCP, ok := targetAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
// Fallback for unknown address types
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
// Determine protocol family based on client IP and normalize target IP accordingly
|
||||
var protocol string
|
||||
var targetIP string
|
||||
|
||||
if clientTCP.IP.To4() != nil {
|
||||
// Client is IPv4, use TCP4 protocol
|
||||
protocol = "TCP4"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
// Target is also IPv4, use as-is
|
||||
targetIP = targetTCP.IP.String()
|
||||
} else {
|
||||
// Target is IPv6, but we need IPv4 for consistent protocol family
|
||||
// Use the IPv4 loopback if target is IPv6 loopback, otherwise use 127.0.0.1
|
||||
if targetTCP.IP.IsLoopback() {
|
||||
targetIP = "127.0.0.1"
|
||||
} else {
|
||||
// For non-loopback IPv6 targets, we could try to extract embedded IPv4
|
||||
// or fall back to a sensible IPv4 address based on the target
|
||||
targetIP = "127.0.0.1" // Safe fallback
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Client is IPv6, use TCP6 protocol
|
||||
protocol = "TCP6"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
// Target is IPv4, convert to IPv6 representation
|
||||
targetIP = "::ffff:" + targetTCP.IP.String()
|
||||
} else {
|
||||
// Target is also IPv6, use as-is
|
||||
targetIP = targetTCP.IP.String()
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Sprintf("PROXY %s %s %s %d %d\r\n",
|
||||
protocol,
|
||||
clientTCP.IP.String(),
|
||||
targetIP,
|
||||
clientTCP.Port,
|
||||
targetTCP.Port)
|
||||
}
|
||||
|
||||
// NewSNIProxy creates a new SNI proxy instance
|
||||
func NewSNIProxy(port int, remoteConfigURL, publicKey, localProxyAddr string, localProxyPort int, localOverrides []string, proxyProtocol bool, trustedUpstreams []string) (*SNIProxy, error) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
@@ -121,6 +435,20 @@ func NewSNIProxy(port int, remoteConfigURL, publicKey, localProxyAddr string, lo
|
||||
localOverrides: overridesMap,
|
||||
activeTunnels: make(map[string]*activeTunnel),
|
||||
trustedUpstreams: trustedMap,
|
||||
httpClient: &http.Client{
|
||||
Timeout: 5 * time.Second,
|
||||
Transport: &http.Transport{
|
||||
MaxIdleConns: 100,
|
||||
MaxIdleConnsPerHost: 10,
|
||||
IdleConnTimeout: 90 * time.Second,
|
||||
},
|
||||
},
|
||||
bufferPool: &sync.Pool{
|
||||
New: func() interface{} {
|
||||
buf := make([]byte, 32*1024)
|
||||
return &buf
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
return proxy, nil
|
||||
@@ -184,8 +512,46 @@ func (p *SNIProxy) acceptConnections() {
|
||||
}
|
||||
}
|
||||
|
||||
if p.activeConnections.Load() >= maxSNIConnections {
|
||||
logger.Debug("Max concurrent SNI connections (%d) reached, rejecting connection from %s", maxSNIConnections, conn.RemoteAddr())
|
||||
metrics.RecordSNIConnection("rejected_max_connections")
|
||||
conn.Close()
|
||||
continue
|
||||
}
|
||||
|
||||
remoteHost, _, err := net.SplitHostPort(conn.RemoteAddr().String())
|
||||
if err != nil {
|
||||
remoteHost = conn.RemoteAddr().String()
|
||||
}
|
||||
|
||||
counterVal, _ := p.perIPConnections.LoadOrStore(remoteHost, new(atomic.Int64))
|
||||
perIPCounter := counterVal.(*atomic.Int64)
|
||||
if perIPCounter.Load() >= maxSNIConnectionsPerIP {
|
||||
logger.Debug("Max concurrent SNI connections per IP (%d) reached for %s, rejecting connection", maxSNIConnectionsPerIP, remoteHost)
|
||||
metrics.RecordSNIConnection("rejected_per_ip_limit")
|
||||
conn.Close()
|
||||
continue
|
||||
}
|
||||
perIPCounter.Add(1)
|
||||
|
||||
p.activeConnections.Add(1)
|
||||
metrics.RecordSNIActiveConnection(1)
|
||||
p.wg.Add(1)
|
||||
go p.handleConnection(conn)
|
||||
go func() {
|
||||
defer func() {
|
||||
if perIPCounter.Add(-1) == 0 {
|
||||
// Best-effort cleanup: only remove the map entry if it
|
||||
// still holds this exact counter (a concurrent new
|
||||
// connection from the same IP may have already bumped
|
||||
// it back up via LoadOrStore, or replaced it after a
|
||||
// prior race). Undercounting in that narrow race window
|
||||
// just means one connection isn't rate-limited briefly,
|
||||
// never unbounded growth.
|
||||
p.perIPConnections.CompareAndDelete(remoteHost, counterVal)
|
||||
}
|
||||
}()
|
||||
p.handleConnection(conn)
|
||||
}()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -233,21 +599,29 @@ func (p *SNIProxy) extractSNI(conn net.Conn) (string, io.Reader, error) {
|
||||
func (p *SNIProxy) handleConnection(clientConn net.Conn) {
|
||||
defer p.wg.Done()
|
||||
defer clientConn.Close()
|
||||
defer func() {
|
||||
p.activeConnections.Add(-1)
|
||||
metrics.RecordSNIActiveConnection(-1)
|
||||
}()
|
||||
|
||||
metrics.RecordSNIConnection("accepted")
|
||||
|
||||
logger.Debug("Accepted connection from %s", clientConn.RemoteAddr())
|
||||
|
||||
// Check for PROXY protocol from trusted upstream
|
||||
var proxyInfo *proxyproto.Info
|
||||
var proxyInfo *ProxyProtocolInfo
|
||||
var actualClientConn net.Conn = clientConn
|
||||
|
||||
if len(p.trustedUpstreams) > 0 {
|
||||
var err error
|
||||
proxyInfo, actualClientConn, err = proxyproto.ParseV1Header(clientConn, p.trustedUpstreams)
|
||||
proxyInfo, actualClientConn, err = p.parseProxyProtocolHeader(clientConn)
|
||||
if err != nil {
|
||||
metrics.RecordSNIProxyProtocolParseError()
|
||||
logger.Debug("Failed to parse PROXY protocol: %v", err)
|
||||
return
|
||||
}
|
||||
if proxyInfo != nil {
|
||||
metrics.RecordSNITrustedProxyEvent("proxy_protocol_parsed")
|
||||
logger.Debug("Received PROXY protocol from trusted upstream: %s:%d -> %s:%d",
|
||||
proxyInfo.SrcIP, proxyInfo.SrcPort, proxyInfo.DestIP, proxyInfo.DestPort)
|
||||
} else {
|
||||
@@ -264,11 +638,13 @@ func (p *SNIProxy) handleConnection(clientConn net.Conn) {
|
||||
}
|
||||
|
||||
// Extract SNI hostname
|
||||
clientHelloStart := time.Now()
|
||||
hostname, clientReader, err := p.extractSNI(actualClientConn)
|
||||
if err != nil {
|
||||
logger.Debug("SNI extraction failed: %v", err)
|
||||
return
|
||||
}
|
||||
metrics.RecordProxyTLSHandshake(time.Since(clientHelloStart).Seconds())
|
||||
|
||||
if hostname == "" {
|
||||
log.Println("No SNI hostname found")
|
||||
@@ -316,16 +692,18 @@ func (p *SNIProxy) handleConnection(clientConn net.Conn) {
|
||||
defer targetConn.Close()
|
||||
|
||||
logger.Debug("Connected to target: %s:%d", route.TargetHost, route.TargetPort)
|
||||
metrics.RecordActiveProxyConnection(1)
|
||||
defer metrics.RecordActiveProxyConnection(-1)
|
||||
|
||||
// Send PROXY protocol header if enabled
|
||||
if p.proxyProtocol {
|
||||
var proxyHeader string
|
||||
if proxyInfo != nil {
|
||||
// Use original client info from PROXY protocol
|
||||
proxyHeader = proxyproto.BuildV1HeaderFromInfo(proxyInfo, targetConn.LocalAddr())
|
||||
proxyHeader = p.buildProxyProtocolHeaderFromInfo(proxyInfo, targetConn.LocalAddr())
|
||||
} else {
|
||||
// Use direct client connection info
|
||||
proxyHeader = proxyproto.BuildV1Header(clientConn.RemoteAddr(), targetConn.LocalAddr())
|
||||
proxyHeader = buildProxyProtocolHeader(clientConn.RemoteAddr(), targetConn.LocalAddr())
|
||||
}
|
||||
logger.Debug("Sending PROXY protocol header: %s", strings.TrimSpace(proxyHeader))
|
||||
|
||||
@@ -365,7 +743,7 @@ func (p *SNIProxy) handleConnection(clientConn net.Conn) {
|
||||
}()
|
||||
|
||||
// Start bidirectional data transfer
|
||||
p.pipe(actualClientConn, targetConn, clientReader)
|
||||
p.pipe(hostname, actualClientConn, targetConn, clientReader)
|
||||
}
|
||||
|
||||
// getRoute retrieves routing information for a hostname
|
||||
@@ -373,6 +751,7 @@ func (p *SNIProxy) getRoute(hostname, clientAddr string) (*RouteRecord, error) {
|
||||
// Check local overrides first
|
||||
if _, isOverride := p.localOverrides[hostname]; isOverride {
|
||||
logger.Debug("Local override matched for hostname: %s", hostname)
|
||||
metrics.RecordProxyRouteLookup("local_override")
|
||||
return &RouteRecord{
|
||||
Hostname: hostname,
|
||||
TargetHost: p.localProxyAddr,
|
||||
@@ -385,6 +764,7 @@ func (p *SNIProxy) getRoute(hostname, clientAddr string) (*RouteRecord, error) {
|
||||
_, isLocal := p.localSNIs[hostname]
|
||||
p.localSNIsLock.RUnlock()
|
||||
if isLocal {
|
||||
metrics.RecordProxyRouteLookup("local")
|
||||
return &RouteRecord{
|
||||
Hostname: hostname,
|
||||
TargetHost: p.localProxyAddr,
|
||||
@@ -395,13 +775,16 @@ func (p *SNIProxy) getRoute(hostname, clientAddr string) (*RouteRecord, error) {
|
||||
// Check cache first
|
||||
if cached, found := p.cache.Get(hostname); found {
|
||||
if cached == nil {
|
||||
metrics.RecordProxyRouteLookup("cached_not_found")
|
||||
return nil, nil // Cached negative result
|
||||
}
|
||||
logger.Debug("Cache hit for hostname: %s", hostname)
|
||||
metrics.RecordProxyRouteLookup("cache_hit")
|
||||
return cached.(*RouteRecord), nil
|
||||
}
|
||||
|
||||
logger.Debug("Cache miss for hostname: %s, querying API", hostname)
|
||||
metrics.RecordProxyRouteLookup("cache_miss")
|
||||
|
||||
// Query API with timeout
|
||||
ctx, cancel := context.WithTimeout(p.ctx, 5*time.Second)
|
||||
@@ -429,22 +812,28 @@ func (p *SNIProxy) getRoute(hostname, clientAddr string) (*RouteRecord, error) {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
|
||||
// Make HTTP request
|
||||
client := &http.Client{Timeout: 5 * time.Second}
|
||||
resp, err := client.Do(req)
|
||||
apiStart := time.Now()
|
||||
// Make HTTP request using reusable client
|
||||
resp, err := p.httpClient.Do(req)
|
||||
if err != nil {
|
||||
metrics.RecordSNIRouteAPIRequest("error")
|
||||
return nil, fmt.Errorf("API request failed: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
metrics.RecordSNIRouteAPILatency(time.Since(apiStart).Seconds())
|
||||
|
||||
if resp.StatusCode == http.StatusNotFound {
|
||||
metrics.RecordSNIRouteAPIRequest("not_found")
|
||||
// Cache negative result for shorter time (1 minute)
|
||||
p.cache.Set(hostname, nil, 1*time.Minute)
|
||||
return nil, nil
|
||||
}
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
metrics.RecordSNIRouteAPIRequest("error")
|
||||
return nil, fmt.Errorf("API returned status %d", resp.StatusCode)
|
||||
}
|
||||
metrics.RecordSNIRouteAPIRequest("success")
|
||||
|
||||
// Parse response
|
||||
var apiResponse RouteAPIResponse
|
||||
@@ -500,8 +889,25 @@ func (p *SNIProxy) selectStickyEndpoint(clientAddr string, endpoints []string) s
|
||||
return endpoints[index]
|
||||
}
|
||||
|
||||
// bufferedReader hides any io.WriterTo the wrapped reader implements (e.g.
|
||||
// io.MultiReader, used by peekClientHello to replay the buffered
|
||||
// ClientHello ahead of the raw connection). Without this, io.CopyBuffer
|
||||
// bypasses the caller-supplied buffer entirely and lets WriteTo drive its
|
||||
// own, uncapped allocations - defeating the point of bufferPool.
|
||||
type bufferedReader struct {
|
||||
io.Reader
|
||||
}
|
||||
|
||||
// bufferedWriter hides any io.ReaderFrom the wrapped writer implements (e.g.
|
||||
// *net.TCPConn's splice/sendfile fast path). Without this, io.CopyBuffer
|
||||
// bypasses the caller-supplied buffer here too, so each copy allocates and
|
||||
// manages its own buffer regardless of what's pooled.
|
||||
type bufferedWriter struct {
|
||||
io.Writer
|
||||
}
|
||||
|
||||
// pipe handles bidirectional data transfer between connections
|
||||
func (p *SNIProxy) pipe(clientConn, targetConn net.Conn, clientReader io.Reader) {
|
||||
func (p *SNIProxy) pipe(hostname string, clientConn, targetConn net.Conn, clientReader io.Reader) {
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(2)
|
||||
|
||||
@@ -520,9 +926,16 @@ func (p *SNIProxy) pipe(clientConn, targetConn net.Conn, clientReader io.Reader)
|
||||
defer wg.Done()
|
||||
defer closeConns()
|
||||
|
||||
// Use a large buffer for better performance
|
||||
buf := make([]byte, 32*1024)
|
||||
_, err := io.CopyBuffer(targetConn, clientReader, buf)
|
||||
// Get buffer from pool and return when done
|
||||
bufPtr := p.bufferPool.Get().(*[]byte)
|
||||
defer func() {
|
||||
// Clear buffer before returning to pool to prevent data leakage
|
||||
clear(*bufPtr)
|
||||
p.bufferPool.Put(bufPtr)
|
||||
}()
|
||||
|
||||
bytesCopied, err := io.CopyBuffer(bufferedWriter{targetConn}, bufferedReader{clientReader}, *bufPtr)
|
||||
metrics.RecordProxyBytesTransmitted("client_to_target", bytesCopied)
|
||||
if err != nil && err != io.EOF {
|
||||
logger.Debug("Copy client->target error: %v", err)
|
||||
}
|
||||
@@ -533,9 +946,16 @@ func (p *SNIProxy) pipe(clientConn, targetConn net.Conn, clientReader io.Reader)
|
||||
defer wg.Done()
|
||||
defer closeConns()
|
||||
|
||||
// Use a large buffer for better performance
|
||||
buf := make([]byte, 32*1024)
|
||||
_, err := io.CopyBuffer(clientConn, targetConn, buf)
|
||||
// Get buffer from pool and return when done
|
||||
bufPtr := p.bufferPool.Get().(*[]byte)
|
||||
defer func() {
|
||||
// Clear buffer before returning to pool to prevent data leakage
|
||||
clear(*bufPtr)
|
||||
p.bufferPool.Put(bufPtr)
|
||||
}()
|
||||
|
||||
bytesCopied, err := io.CopyBuffer(bufferedWriter{clientConn}, bufferedReader{targetConn}, *bufPtr)
|
||||
metrics.RecordProxyBytesTransmitted("target_to_client", bytesCopied)
|
||||
if err != nil && err != io.EOF {
|
||||
logger.Debug("Copy target->client error: %v", err)
|
||||
}
|
||||
|
||||
@@ -3,8 +3,6 @@ package proxy
|
||||
import (
|
||||
"net"
|
||||
"testing"
|
||||
|
||||
"github.com/fosrl/gerbil/proxyproto"
|
||||
)
|
||||
|
||||
func TestBuildProxyProtocolHeader(t *testing.T) {
|
||||
@@ -58,7 +56,7 @@ func TestBuildProxyProtocolHeader(t *testing.T) {
|
||||
t.Fatalf("Failed to resolve target address: %v", err)
|
||||
}
|
||||
|
||||
result := proxyproto.BuildV1Header(clientTCP, targetTCP)
|
||||
result := buildProxyProtocolHeader(clientTCP, targetTCP)
|
||||
if result != tt.expected {
|
||||
t.Errorf("Expected %q, got %q", tt.expected, result)
|
||||
}
|
||||
@@ -71,7 +69,7 @@ func TestBuildProxyProtocolHeaderUnknownType(t *testing.T) {
|
||||
clientAddr := &net.UDPAddr{IP: net.ParseIP("192.168.1.100"), Port: 12345}
|
||||
targetAddr := &net.UDPAddr{IP: net.ParseIP("10.0.0.1"), Port: 443}
|
||||
|
||||
result := proxyproto.BuildV1Header(clientAddr, targetAddr)
|
||||
result := buildProxyProtocolHeader(clientAddr, targetAddr)
|
||||
expected := "PROXY UNKNOWN\r\n"
|
||||
|
||||
if result != expected {
|
||||
@@ -80,8 +78,13 @@ func TestBuildProxyProtocolHeaderUnknownType(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestBuildProxyProtocolHeaderFromInfo(t *testing.T) {
|
||||
proxy, err := NewSNIProxy(8443, "", "", "127.0.0.1", 443, nil, true, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create SNI proxy: %v", err)
|
||||
}
|
||||
|
||||
// Test IPv4 case
|
||||
info := &proxyproto.Info{
|
||||
proxyInfo := &ProxyProtocolInfo{
|
||||
Protocol: "TCP4",
|
||||
SrcIP: "10.0.0.1",
|
||||
DestIP: "192.168.1.100",
|
||||
@@ -90,7 +93,7 @@ func TestBuildProxyProtocolHeaderFromInfo(t *testing.T) {
|
||||
}
|
||||
|
||||
targetAddr, _ := net.ResolveTCPAddr("tcp", "127.0.0.1:8080")
|
||||
header := proxyproto.BuildV1HeaderFromInfo(info, targetAddr)
|
||||
header := proxy.buildProxyProtocolHeaderFromInfo(proxyInfo, targetAddr)
|
||||
|
||||
expected := "PROXY TCP4 10.0.0.1 127.0.0.1 12345 8080\r\n"
|
||||
if header != expected {
|
||||
@@ -98,7 +101,7 @@ func TestBuildProxyProtocolHeaderFromInfo(t *testing.T) {
|
||||
}
|
||||
|
||||
// Test IPv6 case
|
||||
info = &proxyproto.Info{
|
||||
proxyInfo = &ProxyProtocolInfo{
|
||||
Protocol: "TCP6",
|
||||
SrcIP: "2001:db8::1",
|
||||
DestIP: "2001:db8::2",
|
||||
@@ -107,99 +110,10 @@ func TestBuildProxyProtocolHeaderFromInfo(t *testing.T) {
|
||||
}
|
||||
|
||||
targetAddr, _ = net.ResolveTCPAddr("tcp6", "[::1]:8080")
|
||||
header = proxyproto.BuildV1HeaderFromInfo(info, targetAddr)
|
||||
header = proxy.buildProxyProtocolHeaderFromInfo(proxyInfo, targetAddr)
|
||||
|
||||
expected = "PROXY TCP6 2001:db8::1 ::1 12345 8080\r\n"
|
||||
if header != expected {
|
||||
t.Errorf("Expected header '%s', got '%s'", expected, header)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseV2UDPHeader(t *testing.T) {
|
||||
// Build a minimal PROXY v2 header for IPv4 UDP
|
||||
// Magic (12) + ver/cmd (1) + fam/proto (1) + len (2) + src IP (4) + dst IP (4) + src port (2) + dst port (2) = 28 bytes
|
||||
header := []byte{
|
||||
// Magic signature
|
||||
0x0D, 0x0A, 0x0D, 0x0A, 0x00, 0x0D, 0x0A, 0x51, 0x55, 0x49, 0x54, 0x0A,
|
||||
// Version 2 (0x2x), PROXY command (0x01)
|
||||
0x21,
|
||||
// AF_INET (0x1x), DGRAM/UDP (0x02)
|
||||
0x12,
|
||||
// Address block length: 12 bytes (4+4+2+2)
|
||||
0x00, 0x0C,
|
||||
// Source IP: 192.168.1.100
|
||||
192, 168, 1, 100,
|
||||
// Destination IP: 10.0.0.1
|
||||
10, 0, 0, 1,
|
||||
// Source port: 4500
|
||||
0x11, 0x94,
|
||||
// Destination port: 21820
|
||||
0x55, 0x3C,
|
||||
}
|
||||
|
||||
// Append a fake application payload
|
||||
payload := []byte{0x01, 0x02, 0x03}
|
||||
data := append(header, payload...)
|
||||
|
||||
info, remaining, ok := proxyproto.ParseV2UDPHeader(data)
|
||||
if !ok {
|
||||
t.Fatal("Expected ParseV2UDPHeader to return ok=true")
|
||||
}
|
||||
if info == nil {
|
||||
t.Fatal("Expected non-nil Info")
|
||||
}
|
||||
if info.Protocol != "UDP4" {
|
||||
t.Errorf("Expected protocol UDP4, got %s", info.Protocol)
|
||||
}
|
||||
if info.SrcIP != "192.168.1.100" {
|
||||
t.Errorf("Expected SrcIP 192.168.1.100, got %s", info.SrcIP)
|
||||
}
|
||||
if info.DestIP != "10.0.0.1" {
|
||||
t.Errorf("Expected DestIP 10.0.0.1, got %s", info.DestIP)
|
||||
}
|
||||
if info.SrcPort != 4500 {
|
||||
t.Errorf("Expected SrcPort 4500, got %d", info.SrcPort)
|
||||
}
|
||||
if info.DestPort != 21820 {
|
||||
t.Errorf("Expected DestPort 21820, got %d", info.DestPort)
|
||||
}
|
||||
if len(remaining) != len(payload) {
|
||||
t.Errorf("Expected %d remaining bytes, got %d", len(payload), len(remaining))
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseV2UDPHeaderNoHeader(t *testing.T) {
|
||||
// Data that does NOT start with v2 magic should be returned as-is
|
||||
data := []byte{0x01, 0x02, 0x03}
|
||||
info, remaining, ok := proxyproto.ParseV2UDPHeader(data)
|
||||
if ok {
|
||||
t.Error("Expected ok=false for non-v2 data")
|
||||
}
|
||||
if info != nil {
|
||||
t.Error("Expected nil Info for non-v2 data")
|
||||
}
|
||||
if len(remaining) != len(data) {
|
||||
t.Errorf("Expected remaining to equal original data length %d, got %d", len(data), len(remaining))
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsV2Header(t *testing.T) {
|
||||
valid := []byte{
|
||||
0x0D, 0x0A, 0x0D, 0x0A, 0x00, 0x0D, 0x0A, 0x51, 0x55, 0x49, 0x54, 0x0A,
|
||||
// extra bytes beyond the magic
|
||||
0x21, 0x12,
|
||||
}
|
||||
if !proxyproto.IsV2Header(valid) {
|
||||
t.Error("Expected IsV2Header=true for valid magic")
|
||||
}
|
||||
|
||||
invalid := []byte{0x01, 0x02, 0x03}
|
||||
if proxyproto.IsV2Header(invalid) {
|
||||
t.Error("Expected IsV2Header=false for non-magic data")
|
||||
}
|
||||
|
||||
tooShort := []byte{0x0D, 0x0A}
|
||||
if proxyproto.IsV2Header(tooShort) {
|
||||
t.Error("Expected IsV2Header=false for too-short data")
|
||||
}
|
||||
}
|
||||
@@ -1,370 +0,0 @@
|
||||
// Package proxyproto provides shared PROXY protocol v1 (TCP) and v2 (UDP) parsing
|
||||
// and header building utilities used by both the SNI proxy and UDP relay components.
|
||||
package proxyproto
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/fosrl/gerbil/logger"
|
||||
)
|
||||
|
||||
// v2Signature is the 12-byte magic prefix for PROXY protocol v2 headers.
|
||||
var v2Signature = []byte{
|
||||
0x0D, 0x0A, 0x0D, 0x0A, 0x00, 0x0D, 0x0A, 0x51, 0x55, 0x49, 0x54, 0x0A,
|
||||
}
|
||||
|
||||
// Info holds information parsed from an incoming PROXY protocol header (v1 or v2).
|
||||
type Info struct {
|
||||
Protocol string // e.g. "TCP4", "TCP6", "UDP4", "UDP6"
|
||||
SrcIP string
|
||||
DestIP string
|
||||
SrcPort int
|
||||
DestPort int
|
||||
}
|
||||
|
||||
// Conn wraps a net.Conn so that reads are satisfied from a pre-pended buffered
|
||||
// reader first (remaining bytes after PROXY header parsing) and then from the
|
||||
// underlying connection. All other net.Conn methods are forwarded unchanged.
|
||||
type Conn struct {
|
||||
net.Conn
|
||||
Reader io.Reader
|
||||
}
|
||||
|
||||
// Read satisfies net.Conn, draining the buffered reader before falling through
|
||||
// to the underlying connection.
|
||||
func (c *Conn) Read(b []byte) (int, error) {
|
||||
return c.Reader.Read(b)
|
||||
}
|
||||
|
||||
// IsV2Header returns true when data begins with the 12-byte PROXY protocol v2
|
||||
// magic signature.
|
||||
func IsV2Header(data []byte) bool {
|
||||
if len(data) < 12 {
|
||||
return false
|
||||
}
|
||||
return bytes.Equal(data[:12], v2Signature)
|
||||
}
|
||||
|
||||
// ParseV2UDPHeader tries to parse a PROXY protocol v2 header from the front of
|
||||
// a UDP datagram payload.
|
||||
//
|
||||
// Three return values are provided:
|
||||
// - *Info – filled when a PROXY command header was parsed successfully; nil
|
||||
// for a LOCAL command or unrecognised address family.
|
||||
// - []byte – the remaining payload that follows the header (the actual
|
||||
// application data).
|
||||
// - bool – true when a v2 header was detected (and consumed), false when
|
||||
// no v2 magic is present and data should be treated as-is.
|
||||
func ParseV2UDPHeader(data []byte) (*Info, []byte, bool) {
|
||||
if !IsV2Header(data) {
|
||||
return nil, data, false
|
||||
}
|
||||
|
||||
// Minimum fixed header size: 12 (magic) + 1 (ver/cmd) + 1 (fam/proto) + 2 (len) = 16
|
||||
if len(data) < 16 {
|
||||
return nil, data, false
|
||||
}
|
||||
|
||||
// Byte 12: version (high nibble) + command (low nibble)
|
||||
versionCmd := data[12]
|
||||
version := (versionCmd >> 4) & 0x0F
|
||||
command := versionCmd & 0x0F
|
||||
|
||||
if version != 2 {
|
||||
return nil, data, false
|
||||
}
|
||||
|
||||
// Byte 13: address family (high nibble) + transport protocol (low nibble)
|
||||
familyProto := data[13]
|
||||
family := (familyProto >> 4) & 0x0F
|
||||
protocol := familyProto & 0x0F
|
||||
|
||||
// Bytes 14-15: length of the address block that follows, big-endian
|
||||
addrLen := int(binary.BigEndian.Uint16(data[14:16]))
|
||||
totalHeaderLen := 16 + addrLen
|
||||
|
||||
if len(data) < totalHeaderLen {
|
||||
// Truncated packet – signal that a header was detected but is malformed
|
||||
return nil, data, false
|
||||
}
|
||||
|
||||
payload := data[totalHeaderLen:]
|
||||
|
||||
// LOCAL command (0) carries no address information.
|
||||
if command == 0 {
|
||||
return nil, payload, true
|
||||
}
|
||||
|
||||
if command != 1 {
|
||||
// Unknown command – consume the header and return no info
|
||||
return nil, payload, true
|
||||
}
|
||||
|
||||
addrBlock := data[16:totalHeaderLen]
|
||||
|
||||
var (
|
||||
srcIP, destIP net.IP
|
||||
srcPort uint16
|
||||
destPort uint16
|
||||
protocolStr string
|
||||
)
|
||||
|
||||
switch {
|
||||
case family == 1 && protocol == 1: // AF_INET / STREAM (TCP over IPv4)
|
||||
if len(addrBlock) < 12 {
|
||||
return nil, payload, false
|
||||
}
|
||||
srcIP = net.IP(addrBlock[0:4])
|
||||
destIP = net.IP(addrBlock[4:8])
|
||||
srcPort = binary.BigEndian.Uint16(addrBlock[8:10])
|
||||
destPort = binary.BigEndian.Uint16(addrBlock[10:12])
|
||||
protocolStr = "TCP4"
|
||||
|
||||
case family == 1 && protocol == 2: // AF_INET / DGRAM (UDP over IPv4)
|
||||
if len(addrBlock) < 12 {
|
||||
return nil, payload, false
|
||||
}
|
||||
srcIP = net.IP(addrBlock[0:4])
|
||||
destIP = net.IP(addrBlock[4:8])
|
||||
srcPort = binary.BigEndian.Uint16(addrBlock[8:10])
|
||||
destPort = binary.BigEndian.Uint16(addrBlock[10:12])
|
||||
protocolStr = "UDP4"
|
||||
|
||||
case family == 2 && protocol == 1: // AF_INET6 / STREAM (TCP over IPv6)
|
||||
if len(addrBlock) < 36 {
|
||||
return nil, payload, false
|
||||
}
|
||||
srcIP = net.IP(addrBlock[0:16])
|
||||
destIP = net.IP(addrBlock[16:32])
|
||||
srcPort = binary.BigEndian.Uint16(addrBlock[32:34])
|
||||
destPort = binary.BigEndian.Uint16(addrBlock[34:36])
|
||||
protocolStr = "TCP6"
|
||||
|
||||
case family == 2 && protocol == 2: // AF_INET6 / DGRAM (UDP over IPv6)
|
||||
if len(addrBlock) < 36 {
|
||||
return nil, payload, false
|
||||
}
|
||||
srcIP = net.IP(addrBlock[0:16])
|
||||
destIP = net.IP(addrBlock[16:32])
|
||||
srcPort = binary.BigEndian.Uint16(addrBlock[32:34])
|
||||
destPort = binary.BigEndian.Uint16(addrBlock[34:36])
|
||||
protocolStr = "UDP6"
|
||||
|
||||
default:
|
||||
// UNSPEC or AF_UNIX – consume the header, no address info available
|
||||
return nil, payload, true
|
||||
}
|
||||
|
||||
info := &Info{
|
||||
Protocol: protocolStr,
|
||||
SrcIP: srcIP.String(),
|
||||
DestIP: destIP.String(),
|
||||
SrcPort: int(srcPort),
|
||||
DestPort: int(destPort),
|
||||
}
|
||||
return info, payload, true
|
||||
}
|
||||
|
||||
// ParseV1Header attempts to parse a PROXY protocol v1 (text) header from the
|
||||
// given TCP connection.
|
||||
//
|
||||
// The function first checks whether the remote address appears in
|
||||
// trustedUpstreams. If it does not, it returns (nil, conn, nil) and the caller
|
||||
// should treat the connection as a plain (non-proxied) connection.
|
||||
//
|
||||
// When a trusted upstream is detected the function reads up to 512 bytes,
|
||||
// locates the CRLF-terminated header line, and parses the proxy information.
|
||||
// Whatever bytes were consumed (including any data beyond the header line) are
|
||||
// re-prepended via a *Conn wrapper so that subsequent reads by the caller are
|
||||
// transparent.
|
||||
//
|
||||
// Return values:
|
||||
// - *Info – non-nil when a valid PROXY header was parsed.
|
||||
// - net.Conn – always a valid connection (possibly a *Conn wrapper).
|
||||
// - error – non-nil only on hard failures (e.g. bad port numbers).
|
||||
func ParseV1Header(conn net.Conn, trustedUpstreams map[string]struct{}) (*Info, net.Conn, error) {
|
||||
remoteHost, _, err := net.SplitHostPort(conn.RemoteAddr().String())
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("failed to parse remote address: %w", err)
|
||||
}
|
||||
|
||||
if _, isTrusted := trustedUpstreams[remoteHost]; !isTrusted {
|
||||
return nil, conn, nil
|
||||
}
|
||||
|
||||
// Give the upstream 5 s to deliver the PROXY header before timing out.
|
||||
if err := conn.SetReadDeadline(time.Now().Add(5 * time.Second)); err != nil {
|
||||
return nil, conn, fmt.Errorf("failed to set read deadline: %w", err)
|
||||
}
|
||||
|
||||
// The PROXY v1 spec mandates the header fits in 108 bytes; 512 is generous.
|
||||
buffer := make([]byte, 512)
|
||||
n, err := conn.Read(buffer)
|
||||
if err != nil {
|
||||
logger.Debug("Could not read from trusted upstream %s, treating as regular connection: %v", remoteHost, err)
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", clearErr)
|
||||
}
|
||||
return nil, conn, nil
|
||||
}
|
||||
|
||||
// Locate the CRLF that terminates the PROXY header line.
|
||||
headerEnd := bytes.Index(buffer[:n], []byte("\r\n"))
|
||||
if headerEnd == -1 {
|
||||
logger.Debug("No PROXY protocol header from trusted upstream %s, treating as regular TLS connection", remoteHost)
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", clearErr)
|
||||
}
|
||||
newReader := io.MultiReader(bytes.NewReader(buffer[:n]), conn)
|
||||
return nil, &Conn{Conn: conn, Reader: newReader}, nil
|
||||
}
|
||||
|
||||
headerLine := string(buffer[:headerEnd])
|
||||
remainingData := buffer[headerEnd+2 : n]
|
||||
|
||||
parts := strings.Fields(headerLine)
|
||||
|
||||
// Handle "PROXY UNKNOWN" – upstream knows the real source but we don't need it.
|
||||
if len(parts) == 2 && parts[0] == "PROXY" && parts[1] == "UNKNOWN" {
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", clearErr)
|
||||
}
|
||||
var newConn net.Conn
|
||||
if len(remainingData) > 0 {
|
||||
newConn = &Conn{Conn: conn, Reader: io.MultiReader(bytes.NewReader(remainingData), conn)}
|
||||
} else {
|
||||
newConn = conn
|
||||
}
|
||||
return nil, newConn, nil
|
||||
}
|
||||
|
||||
if len(parts) != 6 || parts[0] != "PROXY" {
|
||||
// Malformed line from a trusted upstream – re-prepend everything and
|
||||
// let the caller deal with it as a plain TLS connection.
|
||||
logger.Debug("Invalid PROXY protocol from trusted upstream %s, treating as regular TLS connection: %s", remoteHost, headerLine)
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
logger.Debug("Failed to clear read deadline: %v", clearErr)
|
||||
}
|
||||
newReader := io.MultiReader(bytes.NewReader(buffer[:n]), conn)
|
||||
return nil, &Conn{Conn: conn, Reader: newReader}, nil
|
||||
}
|
||||
|
||||
protocol := parts[1]
|
||||
srcIP := parts[2]
|
||||
destIP := parts[3]
|
||||
|
||||
srcPort, err := strconv.Atoi(parts[4])
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("invalid source port in PROXY header: %s", parts[4])
|
||||
}
|
||||
destPort, err := strconv.Atoi(parts[5])
|
||||
if err != nil {
|
||||
return nil, conn, fmt.Errorf("invalid destination port in PROXY header: %s", parts[5])
|
||||
}
|
||||
|
||||
// Re-assemble a reader that returns any bytes read beyond the header first.
|
||||
var newReader io.Reader
|
||||
if len(remainingData) > 0 {
|
||||
newReader = io.MultiReader(bytes.NewReader(remainingData), conn)
|
||||
} else {
|
||||
newReader = conn
|
||||
}
|
||||
wrappedConn := &Conn{Conn: conn, Reader: newReader}
|
||||
|
||||
if clearErr := conn.SetReadDeadline(time.Time{}); clearErr != nil {
|
||||
return nil, conn, fmt.Errorf("failed to clear read deadline: %w", clearErr)
|
||||
}
|
||||
|
||||
info := &Info{
|
||||
Protocol: protocol,
|
||||
SrcIP: srcIP,
|
||||
DestIP: destIP,
|
||||
SrcPort: srcPort,
|
||||
DestPort: destPort,
|
||||
}
|
||||
return info, wrappedConn, nil
|
||||
}
|
||||
|
||||
// BuildV1Header constructs a PROXY protocol v1 header string from two TCP
|
||||
// addresses, normalising the protocol family so that v1's constraint of a
|
||||
// single family per header is satisfied.
|
||||
func BuildV1Header(clientAddr, targetAddr net.Addr) string {
|
||||
clientTCP, ok := clientAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
targetTCP, ok := targetAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
var protocol, targetIP string
|
||||
|
||||
if clientTCP.IP.To4() != nil {
|
||||
// IPv4 client
|
||||
protocol = "TCP4"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
targetIP = targetTCP.IP.String()
|
||||
} else if targetTCP.IP.IsLoopback() {
|
||||
targetIP = "127.0.0.1"
|
||||
} else {
|
||||
targetIP = "127.0.0.1" // safe fallback for mixed-family
|
||||
}
|
||||
} else {
|
||||
// IPv6 client
|
||||
protocol = "TCP6"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
targetIP = "::ffff:" + targetTCP.IP.String()
|
||||
} else {
|
||||
targetIP = targetTCP.IP.String()
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Sprintf("PROXY %s %s %s %d %d\r\n",
|
||||
protocol, clientTCP.IP.String(), targetIP, clientTCP.Port, targetTCP.Port)
|
||||
}
|
||||
|
||||
// BuildV1HeaderFromInfo constructs a PROXY protocol v1 header string using a
|
||||
// previously-parsed *Info (i.e. when this server itself sits behind an
|
||||
// upstream proxy) and the target TCP address.
|
||||
func BuildV1HeaderFromInfo(info *Info, targetAddr net.Addr) string {
|
||||
targetTCP, ok := targetAddr.(*net.TCPAddr)
|
||||
if !ok {
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
srcIP := net.ParseIP(info.SrcIP)
|
||||
if srcIP == nil {
|
||||
return "PROXY UNKNOWN\r\n"
|
||||
}
|
||||
|
||||
var protocol, targetIP string
|
||||
|
||||
if srcIP.To4() != nil {
|
||||
protocol = "TCP4"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
targetIP = targetTCP.IP.String()
|
||||
} else if targetTCP.IP.IsLoopback() {
|
||||
targetIP = "127.0.0.1"
|
||||
} else {
|
||||
targetIP = "127.0.0.1"
|
||||
}
|
||||
} else {
|
||||
protocol = "TCP6"
|
||||
if targetTCP.IP.To4() != nil {
|
||||
targetIP = "::ffff:" + targetTCP.IP.String()
|
||||
} else {
|
||||
targetIP = targetTCP.IP.String()
|
||||
}
|
||||
}
|
||||
|
||||
return fmt.Sprintf("PROXY %s %s %s %d %d\r\n",
|
||||
protocol, info.SrcIP, targetIP, info.SrcPort, targetTCP.Port)
|
||||
}
|
||||
569
relay/relay.go
569
relay/relay.go
@@ -1,6 +1,7 @@
|
||||
package relay
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/binary"
|
||||
@@ -9,18 +10,22 @@ import (
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"runtime"
|
||||
"strings"
|
||||
"strconv"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/fosrl/gerbil/internal/metrics"
|
||||
"github.com/fosrl/gerbil/logger"
|
||||
"github.com/fosrl/gerbil/proxyproto"
|
||||
"golang.org/x/crypto/chacha20poly1305"
|
||||
"golang.org/x/crypto/curve25519"
|
||||
"golang.zx2c4.com/wireguard/wgctrl/wgtypes"
|
||||
)
|
||||
|
||||
const relayIfname = "relay"
|
||||
|
||||
type EncryptedHolePunchMessage struct {
|
||||
EphemeralPublicKey string `json:"ephemeralPublicKey"`
|
||||
Nonce []byte `json:"nonce"`
|
||||
@@ -58,8 +63,31 @@ type PeerDestination struct {
|
||||
}
|
||||
|
||||
type DestinationConn struct {
|
||||
conn *net.UDPConn
|
||||
lastUsed time.Time
|
||||
conn *net.UDPConn
|
||||
// lastUsed is unix nanoseconds, read/written via atomic ops since it's
|
||||
// touched from packet workers and the response goroutine concurrently
|
||||
// with no lock, and is also scanned for LRU eviction below.
|
||||
lastUsed atomic.Int64
|
||||
}
|
||||
|
||||
// defaultMaxUDPConnections caps the number of concurrent per-peer outbound
|
||||
// UDP sockets the relay will keep open in s.connections. Without a cap, a
|
||||
// burst of peer churn creates sockets faster than the 5-minute idle cleanup
|
||||
// can reap them, exhausting the host's ephemeral port range or fd ulimit.
|
||||
// That surfaces as "dial udp ...: resource temporarily unavailable" on every
|
||||
// subsequent packet and pegs the CPU logging the flood (outage 2026-07-03,
|
||||
// recurrence 2026-07-05). Overridable via GERBIL_MAX_UDP_CONNECTIONS.
|
||||
const defaultMaxUDPConnections = 8192
|
||||
|
||||
var maxUDPConnections = loadMaxUDPConnections()
|
||||
|
||||
func loadMaxUDPConnections() int64 {
|
||||
if v := os.Getenv("GERBIL_MAX_UDP_CONNECTIONS"); v != "" {
|
||||
if n, err := strconv.ParseInt(v, 10, 64); err == nil && n > 0 {
|
||||
return n
|
||||
}
|
||||
}
|
||||
return defaultMaxUDPConnections
|
||||
}
|
||||
|
||||
// Type for storing WireGuard handshake information
|
||||
@@ -136,6 +164,24 @@ const (
|
||||
WireGuardMessageTypeTransportData = 4
|
||||
)
|
||||
|
||||
// cachedEndpointState holds the last-known endpoint fields used for change detection.
|
||||
// Timestamp is intentionally excluded since it always changes.
|
||||
type cachedEndpointState struct {
|
||||
OlmID string
|
||||
NewtID string
|
||||
Token string
|
||||
IP string
|
||||
Port int
|
||||
PublicKey string
|
||||
}
|
||||
|
||||
// cachedEndpointEntry wraps cachedEndpointState with a timestamp so the cache
|
||||
// can be expired after a short TTL even when the endpoint fields are unchanged.
|
||||
type cachedEndpointEntry struct {
|
||||
state cachedEndpointState
|
||||
cachedAt time.Time
|
||||
}
|
||||
|
||||
// --- End Types ---
|
||||
|
||||
// bufferPool allows reusing buffers to reduce allocations.
|
||||
@@ -152,14 +198,20 @@ type UDPProxyServer struct {
|
||||
conn *net.UDPConn
|
||||
proxyMappings sync.Map // map[string]ProxyMapping where key is "ip:port"
|
||||
connections sync.Map // map[string]*DestinationConn where key is destination "ip:port"
|
||||
privateKey wgtypes.Key
|
||||
packetChan chan Packet
|
||||
ctx context.Context
|
||||
cancel context.CancelFunc
|
||||
// connectionCount mirrors len(connections) without an O(n) sync.Map walk,
|
||||
// so the cap check in getOrCreateConnection is cheap on the hot path.
|
||||
connectionCount atomic.Int64
|
||||
privateKey wgtypes.Key
|
||||
packetChan chan Packet
|
||||
ctx context.Context
|
||||
cancel context.CancelFunc
|
||||
|
||||
// Session tracking for WireGuard peers
|
||||
// Key format: "senderIndex:receiverIndex"
|
||||
wgSessions sync.Map
|
||||
// Session index for O(1) lookup by receiver index
|
||||
// Key: receiverIndex (uint32), Value: *WireGuardSession
|
||||
sessionsByReceiverIndex sync.Map
|
||||
// Communication pattern tracking for rebuilding sessions
|
||||
// Key format: "clientIP:clientPort-destIP:destPort"
|
||||
commPatterns sync.Map
|
||||
@@ -168,61 +220,42 @@ type UDPProxyServer struct {
|
||||
// Cache for resolved UDP addresses to avoid per-packet DNS lookups
|
||||
// Key: "ip:port" string, Value: *net.UDPAddr
|
||||
addrCache sync.Map
|
||||
// lastEndpointCache stores the last-known endpoint state per client (key: olmId:newtId)
|
||||
// used to skip redundant HTTP notifications when nothing has changed.
|
||||
lastEndpointCache sync.Map
|
||||
// notifyChan is the async queue for hole-punch endpoint notifications.
|
||||
// Dedicated notifier workers drain this channel and perform the HTTP call.
|
||||
notifyChan chan ClientEndpoint
|
||||
// ReachableAt is the URL where this server can be reached
|
||||
ReachableAt string
|
||||
|
||||
// proxyProtocol enables PROXY protocol v2 header parsing for incoming UDP packets.
|
||||
// When enabled, packets from trustedUpstreams that carry a v2 header will have
|
||||
// their source address overridden with the address reported in the header.
|
||||
proxyProtocol bool
|
||||
trustedUpstreams map[string]struct{}
|
||||
}
|
||||
|
||||
// NewUDPProxyServer initializes the server with a buffered packet channel and derived context.
|
||||
//
|
||||
// proxyProtocol enables PROXY protocol v2 parsing for datagrams arriving from
|
||||
// any address listed in trustedUpstreams (plain IPs or resolvable hostnames).
|
||||
// When a trusted datagram carries a v2 header its source address is replaced
|
||||
// with the address carried inside the header before further processing, so that
|
||||
// hole-punch endpoints reflect the original client IP rather than the load
|
||||
// balancer's address.
|
||||
func NewUDPProxyServer(parentCtx context.Context, addr, serverURL string, privateKey wgtypes.Key, reachableAt string, proxyProtocol bool, trustedUpstreams []string) *UDPProxyServer {
|
||||
func NewUDPProxyServer(parentCtx context.Context, addr, serverURL string, privateKey wgtypes.Key, reachableAt string) *UDPProxyServer {
|
||||
ctx, cancel := context.WithCancel(parentCtx)
|
||||
|
||||
trustedMap := make(map[string]struct{})
|
||||
for _, upstream := range trustedUpstreams {
|
||||
upstream = strings.TrimSpace(upstream)
|
||||
if upstream == "" {
|
||||
continue
|
||||
}
|
||||
trustedMap[upstream] = struct{}{}
|
||||
// Also resolve any hostnames to their current IPs so we can match by IP.
|
||||
if ips, err := net.LookupIP(upstream); err == nil {
|
||||
for _, ip := range ips {
|
||||
trustedMap[ip.String()] = struct{}{}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return &UDPProxyServer{
|
||||
addr: addr,
|
||||
serverURL: serverURL,
|
||||
privateKey: privateKey,
|
||||
packetChan: make(chan Packet, 50000), // Increased from 1000 to handle high throughput
|
||||
ReachableAt: reachableAt,
|
||||
ctx: ctx,
|
||||
cancel: cancel,
|
||||
proxyProtocol: proxyProtocol,
|
||||
trustedUpstreams: trustedMap,
|
||||
addr: addr,
|
||||
serverURL: serverURL,
|
||||
privateKey: privateKey,
|
||||
packetChan: make(chan Packet, 50000), // Increased from 1000 to handle high throughput
|
||||
notifyChan: make(chan ClientEndpoint, 1000),
|
||||
ReachableAt: reachableAt,
|
||||
ctx: ctx,
|
||||
cancel: cancel,
|
||||
}
|
||||
}
|
||||
|
||||
// Start sets up the UDP listener, worker pool, and begins reading packets.
|
||||
func (s *UDPProxyServer) Start() error {
|
||||
// Fetch initial mappings.
|
||||
if err := s.fetchInitialMappings(); err != nil {
|
||||
return fmt.Errorf("failed to fetch initial mappings: %v", err)
|
||||
}
|
||||
// Fetch initial mappings asynchronously so a large (potentially 100MB+)
|
||||
// response does not block the UDP listener from coming up. Any packets
|
||||
// arriving for unknown mappings before the load completes will simply
|
||||
// log and be repopulated via the hole-punch path.
|
||||
go func() {
|
||||
if err := s.fetchInitialMappings(); err != nil {
|
||||
logger.Error("Failed to fetch initial mappings: %v", err)
|
||||
}
|
||||
}()
|
||||
|
||||
udpAddr, err := net.ResolveUDPAddr("udp", s.addr)
|
||||
if err != nil {
|
||||
@@ -264,6 +297,11 @@ func (s *UDPProxyServer) Start() error {
|
||||
// Start the hole punch rate limiter cleanup routine
|
||||
go s.cleanupHolePunchRateLimiter()
|
||||
|
||||
// Start async endpoint notifier workers (HTTP calls off the hot path)
|
||||
for i := 0; i < 5; i++ {
|
||||
go s.endpointNotifierWorker()
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -321,48 +359,17 @@ func (s *UDPProxyServer) readPackets() {
|
||||
// packetWorker processes incoming packets from the channel.
|
||||
func (s *UDPProxyServer) packetWorker() {
|
||||
for packet := range s.packetChan {
|
||||
// effectiveData and effectiveAddr represent the application-layer payload
|
||||
// and the true originating address. They start as the raw UDP values and
|
||||
// may be updated below when a PROXY protocol v2 header is present.
|
||||
effectiveData := packet.data[:packet.n]
|
||||
effectiveAddr := packet.remoteAddr
|
||||
|
||||
// ---------- PROXY protocol v2 (UDP) ------------------------------------
|
||||
// If proxy protocol is enabled and this datagram arrives from a trusted
|
||||
// upstream (e.g. a load balancer), attempt to parse the v2 header so
|
||||
// that we use the original client address for hole-punch registration and
|
||||
// WireGuard session tracking rather than the load balancer's address.
|
||||
if s.proxyProtocol && len(s.trustedUpstreams) > 0 {
|
||||
remoteHost := packet.remoteAddr.IP.String()
|
||||
if _, trusted := s.trustedUpstreams[remoteHost]; trusted {
|
||||
if info, payload, ok := proxyproto.ParseV2UDPHeader(effectiveData); ok {
|
||||
if info != nil {
|
||||
// Override source address with what the proxy reported.
|
||||
if srcIP := net.ParseIP(info.SrcIP); srcIP != nil {
|
||||
effectiveAddr = &net.UDPAddr{
|
||||
IP: srcIP,
|
||||
Port: info.SrcPort,
|
||||
}
|
||||
logger.Debug("PROXY protocol v2: overriding source %s → %s:%d",
|
||||
packet.remoteAddr, info.SrcIP, info.SrcPort)
|
||||
}
|
||||
}
|
||||
// Always advance past the header so the remainder is treated
|
||||
// as the real application payload.
|
||||
effectiveData = payload
|
||||
}
|
||||
}
|
||||
}
|
||||
// -----------------------------------------------------------------------
|
||||
|
||||
// Determine packet type by inspecting the first byte of the (possibly
|
||||
// stripped) application payload.
|
||||
if len(effectiveData) > 0 && effectiveData[0] >= 1 && effectiveData[0] <= 4 {
|
||||
// Determine packet type by inspecting the first byte.
|
||||
if packet.n > 0 && packet.data[0] >= 1 && packet.data[0] <= 4 {
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "in")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(packet.n))
|
||||
// Process as a WireGuard packet.
|
||||
s.handleWireGuardPacket(effectiveData, effectiveAddr)
|
||||
s.handleWireGuardPacket(packet.data, packet.remoteAddr)
|
||||
} else {
|
||||
metrics.RecordUDPPacket(relayIfname, "hole_punch", "in")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "hole_punch", float64(packet.n))
|
||||
// Rate limit: allow at most 2 hole punch messages per IP:Port per second
|
||||
rateLimitKey := effectiveAddr.String()
|
||||
rateLimitKey := packet.remoteAddr.String()
|
||||
entryVal, _ := s.holePunchRateLimiter.LoadOrStore(rateLimitKey, &holePunchRateLimitEntry{
|
||||
windowStart: time.Now(),
|
||||
})
|
||||
@@ -378,14 +385,16 @@ func (s *UDPProxyServer) packetWorker() {
|
||||
rlEntry.mu.Unlock()
|
||||
if !allowed {
|
||||
// logger.Debug("Rate limiting hole punch message from %s", rateLimitKey)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "rate_limited")
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
}
|
||||
|
||||
// Process as an encrypted hole punch message
|
||||
var encMsg EncryptedHolePunchMessage
|
||||
if err := json.Unmarshal(effectiveData, &encMsg); err != nil {
|
||||
if err := json.Unmarshal(packet.data, &encMsg); err != nil {
|
||||
logger.Error("Error unmarshaling encrypted message: %v", err)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "error")
|
||||
// Return the buffer to the pool for reuse and continue with next packet
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
@@ -393,6 +402,7 @@ func (s *UDPProxyServer) packetWorker() {
|
||||
|
||||
if encMsg.EphemeralPublicKey == "" {
|
||||
logger.Error("Received malformed message without ephemeral key")
|
||||
metrics.RecordHolePunchEvent(relayIfname, "error")
|
||||
// Return the buffer to the pool for reuse and continue with next packet
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
@@ -402,6 +412,7 @@ func (s *UDPProxyServer) packetWorker() {
|
||||
decryptedData, err := s.decryptMessage(encMsg)
|
||||
if err != nil {
|
||||
// logger.Error("Failed to decrypt message: %v", err)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "error")
|
||||
// Return the buffer to the pool for reuse and continue with next packet
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
@@ -411,6 +422,7 @@ func (s *UDPProxyServer) packetWorker() {
|
||||
var msg HolePunchMessage
|
||||
if err := json.Unmarshal(decryptedData, &msg); err != nil {
|
||||
logger.Error("Error unmarshaling decrypted message: %v", err)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "error")
|
||||
// Return the buffer to the pool for reuse and continue with next packet
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
@@ -420,22 +432,75 @@ func (s *UDPProxyServer) packetWorker() {
|
||||
NewtID: msg.NewtID,
|
||||
OlmID: msg.OlmID,
|
||||
Token: msg.Token,
|
||||
IP: effectiveAddr.IP.String(),
|
||||
Port: effectiveAddr.Port,
|
||||
IP: packet.remoteAddr.IP.String(),
|
||||
Port: packet.remoteAddr.Port,
|
||||
Timestamp: time.Now().Unix(),
|
||||
ReachableAt: s.ReachableAt,
|
||||
ExitNodePublicKey: s.privateKey.PublicKey().String(),
|
||||
ClientPublicKey: msg.PublicKey,
|
||||
}
|
||||
logger.Debug("Created endpoint from packet remoteAddr %s: IP=%s, Port=%d", effectiveAddr.String(), endpoint.IP, endpoint.Port)
|
||||
s.notifyServer(endpoint)
|
||||
logger.Debug("Created endpoint from packet remoteAddr %s: IP=%s, Port=%d", packet.remoteAddr.String(), endpoint.IP, endpoint.Port)
|
||||
|
||||
// Check if anything meaningful changed before queuing an HTTP notification.
|
||||
// The cache expires after 2.5 s so the server always receives a fresh
|
||||
// timestamp within its 5-second staleness window.
|
||||
const endpointCacheTTL = 2500 * time.Millisecond
|
||||
cacheKey := endpoint.OlmID + ":" + endpoint.NewtID
|
||||
newState := cachedEndpointState{
|
||||
OlmID: endpoint.OlmID,
|
||||
NewtID: endpoint.NewtID,
|
||||
Token: endpoint.Token,
|
||||
IP: endpoint.IP,
|
||||
Port: endpoint.Port,
|
||||
PublicKey: endpoint.ClientPublicKey,
|
||||
}
|
||||
if cached, ok := s.lastEndpointCache.Load(cacheKey); ok {
|
||||
entry := cached.(cachedEndpointEntry)
|
||||
if entry.state == newState && time.Since(entry.cachedAt) < endpointCacheTTL {
|
||||
// Endpoint unchanged and cache still fresh - skip the HTTP call.
|
||||
logger.Debug("Endpoint unchanged for %s, skipping notification", cacheKey)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "deduplicated")
|
||||
s.clearSessionsForIP(endpoint.IP)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "success")
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
continue
|
||||
}
|
||||
}
|
||||
s.lastEndpointCache.Store(cacheKey, cachedEndpointEntry{state: newState, cachedAt: time.Now()})
|
||||
|
||||
// Queue the notification asynchronously so the hot path is not blocked by HTTP.
|
||||
select {
|
||||
case s.notifyChan <- endpoint:
|
||||
case <-s.ctx.Done():
|
||||
// shutting down
|
||||
default:
|
||||
logger.Debug("Notification queue full, dropping hole punch notification for %s:%d", endpoint.IP, endpoint.Port)
|
||||
metrics.RecordHolePunchEvent(relayIfname, "queue_full")
|
||||
}
|
||||
s.clearSessionsForIP(endpoint.IP) // Clear sessions for this IP to allow re-establishment
|
||||
metrics.RecordHolePunchEvent(relayIfname, "success")
|
||||
}
|
||||
// Return the buffer to the pool for reuse.
|
||||
bufferPool.Put(packet.data[:1500])
|
||||
}
|
||||
}
|
||||
|
||||
// endpointNotifierWorker drains the notifyChan and performs the HTTP notification for each
|
||||
// hole-punch endpoint. Running several of these keeps latency low even when the server is slow.
|
||||
func (s *UDPProxyServer) endpointNotifierWorker() {
|
||||
for {
|
||||
select {
|
||||
case endpoint, ok := <-s.notifyChan:
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
s.notifyServer(endpoint)
|
||||
case <-s.ctx.Done():
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// decryptMessage decrypts the message using the server's private key
|
||||
func (s *UDPProxyServer) decryptMessage(encMsg EncryptedHolePunchMessage) ([]byte, error) {
|
||||
// Parse the ephemeral public key
|
||||
@@ -471,6 +536,7 @@ func (s *UDPProxyServer) decryptMessage(encMsg EncryptedHolePunchMessage) ([]byt
|
||||
}
|
||||
|
||||
func (s *UDPProxyServer) fetchInitialMappings() error {
|
||||
logger.Info("Requesting initial proxy mappings")
|
||||
body := bytes.NewBuffer([]byte(fmt.Sprintf(`{"publicKey": "%s"}`, s.privateKey.PublicKey().String())))
|
||||
resp, err := http.Post(s.serverURL+"/gerbil/get-all-relays", "application/json", body)
|
||||
if err != nil {
|
||||
@@ -482,22 +548,82 @@ func (s *UDPProxyServer) fetchInitialMappings() error {
|
||||
return fmt.Errorf("server returned non-OK status: %d, body: %s",
|
||||
resp.StatusCode, string(body))
|
||||
}
|
||||
data, err := io.ReadAll(resp.Body)
|
||||
logger.Info("Received initial mappings, streaming decode")
|
||||
|
||||
// Stream-decode the response instead of buffering the entire body
|
||||
// (which can be 100MB+) and then re-walking it with json.Unmarshal.
|
||||
// This both lowers peak memory and lets us start populating the
|
||||
// sync.Map as entries arrive.
|
||||
dec := json.NewDecoder(bufio.NewReaderSize(resp.Body, 1<<20))
|
||||
|
||||
// Expect opening '{' of the top-level object.
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to read response body: %v", err)
|
||||
return fmt.Errorf("failed to read opening token: %v", err)
|
||||
}
|
||||
logger.Info("Received initial mappings: %s", string(data))
|
||||
var initialMappings InitialMappings
|
||||
if err := json.Unmarshal(data, &initialMappings); err != nil {
|
||||
return fmt.Errorf("failed to unmarshal initial mappings: %v", err)
|
||||
if d, ok := tok.(json.Delim); !ok || d != '{' {
|
||||
return fmt.Errorf("expected '{' at top level, got %v", tok)
|
||||
}
|
||||
// Store mappings in our sync.Map.
|
||||
for key, mapping := range initialMappings.Mappings {
|
||||
// Initialize LastUsed timestamp for initial mappings
|
||||
mapping.LastUsed = time.Now()
|
||||
s.proxyMappings.Store(key, mapping)
|
||||
|
||||
count := 0
|
||||
now := time.Now()
|
||||
|
||||
for dec.More() {
|
||||
keyTok, err := dec.Token()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to read top-level key: %v", err)
|
||||
}
|
||||
key, ok := keyTok.(string)
|
||||
if !ok {
|
||||
return fmt.Errorf("expected string key at top level, got %T", keyTok)
|
||||
}
|
||||
|
||||
if key != "mappings" {
|
||||
// Skip unknown top-level fields without materializing them.
|
||||
var skip json.RawMessage
|
||||
if err := dec.Decode(&skip); err != nil {
|
||||
return fmt.Errorf("failed to skip field %q: %v", key, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Expect opening '{' of the mappings object.
|
||||
tok, err := dec.Token()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to read mappings open: %v", err)
|
||||
}
|
||||
if d, ok := tok.(json.Delim); !ok || d != '{' {
|
||||
return fmt.Errorf("expected '{' for mappings, got %v", tok)
|
||||
}
|
||||
|
||||
for dec.More() {
|
||||
mapKeyTok, err := dec.Token()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to read mapping key: %v", err)
|
||||
}
|
||||
mapKey, ok := mapKeyTok.(string)
|
||||
if !ok {
|
||||
return fmt.Errorf("expected string mapping key, got %T", mapKeyTok)
|
||||
}
|
||||
|
||||
var mapping ProxyMapping
|
||||
if err := dec.Decode(&mapping); err != nil {
|
||||
return fmt.Errorf("failed to decode mapping %q: %v", mapKey, err)
|
||||
}
|
||||
mapping.LastUsed = now
|
||||
s.proxyMappings.Store(mapKey, mapping)
|
||||
count++
|
||||
}
|
||||
|
||||
// Consume closing '}' of mappings object.
|
||||
if _, err := dec.Token(); err != nil {
|
||||
return fmt.Errorf("failed to read mappings close: %v", err)
|
||||
}
|
||||
}
|
||||
logger.Info("Loaded %d initial proxy mappings", len(initialMappings.Mappings))
|
||||
|
||||
metrics.RecordProxyInitialMappings(relayIfname, int64(count))
|
||||
metrics.RecordProxyMapping(relayIfname, int64(count))
|
||||
logger.Info("Loaded %d initial proxy mappings", count)
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -612,7 +738,11 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
_, err = conn.Write(packet)
|
||||
if err != nil {
|
||||
logger.Debug("Failed to forward handshake initiation: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
continue
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(len(packet)))
|
||||
}
|
||||
|
||||
case WireGuardMessageTypeHandshakeResponse:
|
||||
@@ -624,12 +754,19 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
sessionKey := fmt.Sprintf("%d:%d", receiverIndex, senderIndex)
|
||||
|
||||
// Store the session information
|
||||
s.wgSessions.Store(sessionKey, &WireGuardSession{
|
||||
session := &WireGuardSession{
|
||||
ReceiverIndex: receiverIndex,
|
||||
SenderIndex: senderIndex,
|
||||
DestAddr: remoteAddr,
|
||||
LastSeen: time.Now(),
|
||||
})
|
||||
}
|
||||
if _, loaded := s.wgSessions.LoadOrStore(sessionKey, session); loaded {
|
||||
s.wgSessions.Store(sessionKey, session)
|
||||
} else {
|
||||
metrics.RecordSession(relayIfname, 1)
|
||||
}
|
||||
// Also index by sender index for O(1) lookup in transport data path
|
||||
s.sessionsByReceiverIndex.Store(senderIndex, session)
|
||||
|
||||
// Forward the response to the original sender
|
||||
for _, dest := range proxyMapping.Destinations {
|
||||
@@ -648,28 +785,26 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
_, err = conn.Write(packet)
|
||||
if err != nil {
|
||||
logger.Error("Failed to forward handshake response: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
continue
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(len(packet)))
|
||||
}
|
||||
|
||||
case WireGuardMessageTypeTransportData:
|
||||
// Data packet: forward only to the established session peer
|
||||
// logger.Debug("Received transport data with receiver index %d from %s", receiverIndex, remoteAddr)
|
||||
|
||||
// Look up the session based on the receiver index
|
||||
// Look up the session based on the receiver index - O(1) lookup instead of O(n) Range
|
||||
var destAddr *net.UDPAddr
|
||||
|
||||
// First check for existing sessions to see if we know where to send this packet
|
||||
s.wgSessions.Range(func(k, v interface{}) bool {
|
||||
session := v.(*WireGuardSession)
|
||||
// Check if session matches (read lock for check)
|
||||
if session.GetSenderIndex() == receiverIndex {
|
||||
// Found matching session - get dest addr and update last seen
|
||||
destAddr = session.GetDestAddr()
|
||||
session.UpdateLastSeen()
|
||||
return false // stop iteration
|
||||
}
|
||||
return true // continue iteration
|
||||
})
|
||||
// Fast path: direct index lookup by receiver index
|
||||
if sessionObj, ok := s.sessionsByReceiverIndex.Load(receiverIndex); ok {
|
||||
session := sessionObj.(*WireGuardSession)
|
||||
destAddr = session.GetDestAddr()
|
||||
session.UpdateLastSeen()
|
||||
}
|
||||
|
||||
if destAddr != nil {
|
||||
// We found a specific peer to forward to
|
||||
@@ -685,7 +820,11 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
_, err = conn.Write(packet)
|
||||
if err != nil {
|
||||
logger.Debug("Failed to forward transport data: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
return
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(len(packet)))
|
||||
} else {
|
||||
// No known session, fall back to forwarding to all peers
|
||||
logger.Debug("No session found for receiver index %d, forwarding to all destinations", receiverIndex)
|
||||
@@ -708,7 +847,11 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
_, err = conn.Write(packet)
|
||||
if err != nil {
|
||||
logger.Debug("Failed to forward transport data: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
continue
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(len(packet)))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -733,7 +876,11 @@ func (s *UDPProxyServer) handleWireGuardPacket(packet []byte, remoteAddr *net.UD
|
||||
_, err = conn.Write(packet)
|
||||
if err != nil {
|
||||
logger.Error("Failed to forward WireGuard packet: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
continue
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(len(packet)))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -744,21 +891,35 @@ func (s *UDPProxyServer) getOrCreateConnection(destAddr *net.UDPAddr, remoteAddr
|
||||
// Check if we have an existing connection
|
||||
if conn, ok := s.connections.Load(key); ok {
|
||||
destConn := conn.(*DestinationConn)
|
||||
destConn.lastUsed = time.Now()
|
||||
destConn.lastUsed.Store(time.Now().UnixNano())
|
||||
return destConn.conn, nil
|
||||
}
|
||||
|
||||
// Enforce a hard cap on concurrent sockets so a burst of peer churn can't
|
||||
// exhaust the host's ephemeral ports/fds. Evict the least-recently-used
|
||||
// connection to make room instead of growing unbounded.
|
||||
if s.connectionCount.Load() >= maxUDPConnections {
|
||||
s.evictLRUConnection()
|
||||
}
|
||||
|
||||
// Create new connection
|
||||
newConn, err := net.DialUDP("udp", nil, destAddr)
|
||||
if err != nil {
|
||||
metrics.RecordProxyConnectionError(relayIfname, "dial_udp")
|
||||
return nil, fmt.Errorf("failed to create UDP connection: %v", err)
|
||||
}
|
||||
|
||||
// Store the new connection
|
||||
s.connections.Store(key, &DestinationConn{
|
||||
conn: newConn,
|
||||
lastUsed: time.Now(),
|
||||
})
|
||||
destConn := &DestinationConn{conn: newConn}
|
||||
destConn.lastUsed.Store(time.Now().UnixNano())
|
||||
|
||||
// Store the new connection. If another goroutine raced us and already
|
||||
// created one for this key, close ours and use theirs instead.
|
||||
if existing, loaded := s.connections.LoadOrStore(key, destConn); loaded {
|
||||
newConn.Close()
|
||||
return existing.(*DestinationConn).conn, nil
|
||||
}
|
||||
s.connectionCount.Add(1)
|
||||
metrics.RecordUDPConnection(relayIfname, 1)
|
||||
|
||||
// Start a goroutine to handle responses
|
||||
go s.handleResponses(newConn, destAddr, remoteAddr)
|
||||
@@ -766,6 +927,33 @@ func (s *UDPProxyServer) getOrCreateConnection(destAddr *net.UDPAddr, remoteAddr
|
||||
return newConn, nil
|
||||
}
|
||||
|
||||
// evictLRUConnection closes and removes the least-recently-used destination
|
||||
// connection so a new one can be created under the concurrent connection cap.
|
||||
func (s *UDPProxyServer) evictLRUConnection() {
|
||||
var oldestKey interface{}
|
||||
var oldestConn *DestinationConn
|
||||
var oldestTime int64
|
||||
|
||||
s.connections.Range(func(key, value interface{}) bool {
|
||||
destConn := value.(*DestinationConn)
|
||||
lu := destConn.lastUsed.Load()
|
||||
if oldestKey == nil || lu < oldestTime {
|
||||
oldestKey = key
|
||||
oldestConn = destConn
|
||||
oldestTime = lu
|
||||
}
|
||||
return true
|
||||
})
|
||||
|
||||
if oldestKey != nil {
|
||||
s.connections.Delete(oldestKey)
|
||||
oldestConn.conn.Close()
|
||||
s.connectionCount.Add(-1)
|
||||
metrics.RecordUDPConnection(relayIfname, -1)
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "conn_evicted", 1)
|
||||
}
|
||||
}
|
||||
|
||||
func (s *UDPProxyServer) handleResponses(conn *net.UDPConn, destAddr *net.UDPAddr, remoteAddr *net.UDPAddr) {
|
||||
buffer := make([]byte, 1500)
|
||||
for {
|
||||
@@ -774,6 +962,8 @@ func (s *UDPProxyServer) handleResponses(conn *net.UDPConn, destAddr *net.UDPAdd
|
||||
logger.Debug("Error reading response from %s: %v", destAddr.String(), err)
|
||||
return
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "in")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(n))
|
||||
|
||||
// Process the response to track sessions if it's a WireGuard packet
|
||||
if n > 0 && buffer[0] >= 1 && buffer[0] <= 4 {
|
||||
@@ -781,12 +971,19 @@ func (s *UDPProxyServer) handleResponses(conn *net.UDPConn, destAddr *net.UDPAdd
|
||||
if ok && buffer[0] == WireGuardMessageTypeHandshakeResponse {
|
||||
// Store the session mapping for the handshake response
|
||||
sessionKey := fmt.Sprintf("%d:%d", senderIndex, receiverIndex)
|
||||
s.wgSessions.Store(sessionKey, &WireGuardSession{
|
||||
session := &WireGuardSession{
|
||||
ReceiverIndex: receiverIndex,
|
||||
SenderIndex: senderIndex,
|
||||
DestAddr: destAddr,
|
||||
LastSeen: time.Now(),
|
||||
})
|
||||
}
|
||||
if _, loaded := s.wgSessions.LoadOrStore(sessionKey, session); loaded {
|
||||
s.wgSessions.Store(sessionKey, session)
|
||||
} else {
|
||||
metrics.RecordSession(relayIfname, 1)
|
||||
}
|
||||
// Also index by sender index for O(1) lookup
|
||||
s.sessionsByReceiverIndex.Store(senderIndex, session)
|
||||
logger.Debug("Stored session mapping: %s -> %s", sessionKey, destAddr.String())
|
||||
} else if ok && buffer[0] == WireGuardMessageTypeTransportData {
|
||||
// Track communication pattern for session rebuilding (reverse direction)
|
||||
@@ -798,26 +995,43 @@ func (s *UDPProxyServer) handleResponses(conn *net.UDPConn, destAddr *net.UDPAdd
|
||||
_, err = s.conn.WriteToUDP(buffer[:n], remoteAddr)
|
||||
if err != nil {
|
||||
logger.Error("Failed to forward response: %v", err)
|
||||
metrics.RecordProxyConnectionError(relayIfname, "write_udp")
|
||||
continue
|
||||
}
|
||||
metrics.RecordUDPPacket(relayIfname, "wireguard", "out")
|
||||
metrics.RecordUDPPacketSize(relayIfname, "wireguard", float64(n))
|
||||
}
|
||||
}
|
||||
|
||||
// Add a cleanup method to periodically remove idle connections
|
||||
func (s *UDPProxyServer) cleanupIdleConnections() {
|
||||
ticker := time.NewTicker(5 * time.Minute)
|
||||
// Ticker interval and idle threshold were previously 5min/10min, meaning
|
||||
// a socket could sit open for up to 15 minutes after going idle. Under a
|
||||
// reconnect/churn burst that lag is enough to exhaust ephemeral ports
|
||||
// before cleanup catches up, so both are tightened here.
|
||||
ticker := time.NewTicker(1 * time.Minute)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
now := time.Now()
|
||||
cleanupStart := time.Now()
|
||||
now := time.Now().UnixNano()
|
||||
removed := int64(0)
|
||||
s.connections.Range(func(key, value interface{}) bool {
|
||||
destConn := value.(*DestinationConn)
|
||||
if now.Sub(destConn.lastUsed) > 10*time.Minute {
|
||||
if now-destConn.lastUsed.Load() > int64(5*time.Minute) {
|
||||
destConn.conn.Close()
|
||||
s.connections.Delete(key)
|
||||
removed++
|
||||
}
|
||||
return true
|
||||
})
|
||||
if removed > 0 {
|
||||
s.connectionCount.Add(-removed)
|
||||
metrics.RecordUDPConnection(relayIfname, -removed)
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "conn", removed)
|
||||
}
|
||||
metrics.RecordProxyIdleCleanupDuration(relayIfname, "conn", time.Since(cleanupStart).Seconds())
|
||||
case <-s.ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -832,16 +1046,20 @@ func (s *UDPProxyServer) cleanupIdleSessions() {
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
cleanupStart := time.Now()
|
||||
now := time.Now()
|
||||
s.wgSessions.Range(func(key, value interface{}) bool {
|
||||
session := value.(*WireGuardSession)
|
||||
// Use thread-safe method to read LastSeen
|
||||
if now.Sub(session.GetLastSeen()) > 15*time.Minute {
|
||||
s.wgSessions.Delete(key)
|
||||
metrics.RecordSession(relayIfname, -1)
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "session", 1)
|
||||
logger.Debug("Removed idle session: %s", key)
|
||||
}
|
||||
return true
|
||||
})
|
||||
metrics.RecordProxyIdleCleanupDuration(relayIfname, "session", time.Since(cleanupStart).Seconds())
|
||||
case <-s.ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -855,16 +1073,20 @@ func (s *UDPProxyServer) cleanupIdleProxyMappings() {
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
cleanupStart := time.Now()
|
||||
now := time.Now()
|
||||
s.proxyMappings.Range(func(key, value interface{}) bool {
|
||||
mapping := value.(ProxyMapping)
|
||||
// Remove mappings that haven't been used in 30 minutes
|
||||
if now.Sub(mapping.LastUsed) > 30*time.Minute {
|
||||
s.proxyMappings.Delete(key)
|
||||
metrics.RecordProxyMapping(relayIfname, -1)
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "proxy_mapping", 1)
|
||||
logger.Debug("Removed idle proxy mapping: %s", key)
|
||||
}
|
||||
return true
|
||||
})
|
||||
metrics.RecordProxyIdleCleanupDuration(relayIfname, "proxy_mapping", time.Since(cleanupStart).Seconds())
|
||||
case <-s.ctx.Done():
|
||||
return
|
||||
}
|
||||
@@ -907,6 +1129,11 @@ func (s *UDPProxyServer) notifyServer(endpoint ClientEndpoint) {
|
||||
key := fmt.Sprintf("%s:%d", endpoint.IP, endpoint.Port)
|
||||
logger.Debug("About to store proxy mapping with key: %s (from endpoint IP=%s, Port=%d)", key, endpoint.IP, endpoint.Port)
|
||||
mapping.LastUsed = time.Now()
|
||||
if _, existed := s.proxyMappings.Load(key); existed {
|
||||
metrics.RecordProxyMappingUpdate(relayIfname)
|
||||
} else {
|
||||
metrics.RecordProxyMapping(relayIfname, 1)
|
||||
}
|
||||
s.proxyMappings.Store(key, mapping)
|
||||
|
||||
logger.Debug("Stored proxy mapping for %s with %d destinations (timestamp: %v)", key, len(mapping.Destinations), mapping.LastUsed)
|
||||
@@ -919,6 +1146,11 @@ func (s *UDPProxyServer) UpdateProxyMapping(sourceIP string, sourcePort int, des
|
||||
Destinations: destinations,
|
||||
LastUsed: time.Now(),
|
||||
}
|
||||
if _, existed := s.proxyMappings.Load(key); existed {
|
||||
metrics.RecordProxyMappingUpdate(relayIfname)
|
||||
} else {
|
||||
metrics.RecordProxyMapping(relayIfname, 1)
|
||||
}
|
||||
s.proxyMappings.Store(key, mapping)
|
||||
}
|
||||
|
||||
@@ -960,6 +1192,10 @@ func (s *UDPProxyServer) clearConnectionsForWGIP(wgIP string) {
|
||||
for _, key := range keysToDelete {
|
||||
s.connections.Delete(key)
|
||||
}
|
||||
if len(keysToDelete) > 0 {
|
||||
s.connectionCount.Add(-int64(len(keysToDelete)))
|
||||
metrics.RecordUDPConnection(relayIfname, -int64(len(keysToDelete)))
|
||||
}
|
||||
|
||||
logger.Info("Cleared %d connections for WG IP: %s", len(keysToDelete), wgIP)
|
||||
}
|
||||
@@ -985,6 +1221,10 @@ func (s *UDPProxyServer) clearSessionsForIP(ip string) {
|
||||
for _, key := range keysToDelete {
|
||||
s.wgSessions.Delete(key)
|
||||
}
|
||||
if len(keysToDelete) > 0 {
|
||||
metrics.RecordSession(relayIfname, -int64(len(keysToDelete)))
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "session", int64(len(keysToDelete)))
|
||||
}
|
||||
|
||||
logger.Debug("Cleared %d sessions for WG IP: %s", len(keysToDelete), ip)
|
||||
}
|
||||
@@ -1145,35 +1385,64 @@ func (s *UDPProxyServer) trackCommunicationPattern(fromAddr, toAddr *net.UDPAddr
|
||||
pattern.LastFromDest = now
|
||||
}
|
||||
|
||||
s.commPatterns.Store(patternKey, pattern)
|
||||
if _, loaded := s.commPatterns.LoadOrStore(patternKey, pattern); !loaded {
|
||||
metrics.RecordCommPattern(relayIfname, 1)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// tryRebuildSession attempts to rebuild a WireGuard session from communication patterns
|
||||
func (s *UDPProxyServer) tryRebuildSession(pattern *CommunicationPattern) {
|
||||
// Require both indices and a minimum amount of bidirectional traffic
|
||||
if pattern.ClientIndex == 0 || pattern.DestIndex == 0 || pattern.PacketCount < 4 {
|
||||
return
|
||||
}
|
||||
|
||||
// Check if we have bidirectional communication within a reasonable time window
|
||||
timeDiff := pattern.LastFromClient.Sub(pattern.LastFromDest)
|
||||
if timeDiff < 0 {
|
||||
timeDiff = -timeDiff
|
||||
}
|
||||
if timeDiff >= 30*time.Second {
|
||||
return
|
||||
}
|
||||
|
||||
// Only rebuild if we have recent bidirectional communication and both indices
|
||||
if timeDiff < 30*time.Second && pattern.ClientIndex != 0 && pattern.DestIndex != 0 && pattern.PacketCount >= 4 {
|
||||
// Create session mapping: client's index maps to destination
|
||||
sessionKey := fmt.Sprintf("%d:%d", pattern.DestIndex, pattern.ClientIndex)
|
||||
sessionKey := fmt.Sprintf("%d:%d", pattern.DestIndex, pattern.ClientIndex)
|
||||
destStr := pattern.ToDestination.String()
|
||||
|
||||
// Check if we already have this session
|
||||
if _, exists := s.wgSessions.Load(sessionKey); !exists {
|
||||
s.wgSessions.Store(sessionKey, &WireGuardSession{
|
||||
ReceiverIndex: pattern.DestIndex,
|
||||
SenderIndex: pattern.ClientIndex,
|
||||
DestAddr: pattern.ToDestination,
|
||||
LastSeen: time.Now(),
|
||||
})
|
||||
logger.Info("Rebuilt WireGuard session from communication pattern: %s -> %s (packets: %d)",
|
||||
sessionKey, pattern.ToDestination.String(), pattern.PacketCount)
|
||||
// Fast path: if a matching session already exists, just refresh LastSeen and bail out.
|
||||
// This prevents log spam and repeated work for every packet of an established flow.
|
||||
if existing, ok := s.wgSessions.Load(sessionKey); ok {
|
||||
sess := existing.(*WireGuardSession)
|
||||
if da := sess.GetDestAddr(); da != nil && da.String() == destStr {
|
||||
sess.UpdateLastSeen()
|
||||
// Make sure the receiver-index fast-path is populated so future packets
|
||||
// don't keep falling back to broadcast + pattern tracking.
|
||||
if _, indexed := s.sessionsByReceiverIndex.Load(pattern.ClientIndex); !indexed {
|
||||
s.sessionsByReceiverIndex.Store(pattern.ClientIndex, sess)
|
||||
}
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// Create or replace the session mapping
|
||||
session := &WireGuardSession{
|
||||
ReceiverIndex: pattern.DestIndex,
|
||||
SenderIndex: pattern.ClientIndex,
|
||||
DestAddr: pattern.ToDestination,
|
||||
LastSeen: time.Now(),
|
||||
}
|
||||
if _, loaded := s.wgSessions.LoadOrStore(sessionKey, session); loaded {
|
||||
s.wgSessions.Store(sessionKey, session)
|
||||
} else {
|
||||
metrics.RecordSession(relayIfname, 1)
|
||||
metrics.RecordSessionRebuilt(relayIfname)
|
||||
}
|
||||
// Index by client receiver index so the transport-data fast path can find it.
|
||||
s.sessionsByReceiverIndex.Store(pattern.ClientIndex, session)
|
||||
|
||||
logger.Info("Rebuilt WireGuard session from communication pattern: %s -> %s (packets: %d)",
|
||||
sessionKey, destStr, pattern.PacketCount)
|
||||
}
|
||||
|
||||
// cleanupIdleCommunicationPatterns periodically removes idle communication patterns
|
||||
@@ -1207,6 +1476,7 @@ func (s *UDPProxyServer) cleanupIdleCommunicationPatterns() {
|
||||
for {
|
||||
select {
|
||||
case <-ticker.C:
|
||||
cleanupStart := time.Now()
|
||||
now := time.Now()
|
||||
s.commPatterns.Range(func(key, value interface{}) bool {
|
||||
pattern := value.(*CommunicationPattern)
|
||||
@@ -1220,10 +1490,13 @@ func (s *UDPProxyServer) cleanupIdleCommunicationPatterns() {
|
||||
// Remove patterns that haven't had activity in 20 minutes
|
||||
if now.Sub(lastActivity) > 20*time.Minute {
|
||||
s.commPatterns.Delete(key)
|
||||
metrics.RecordCommPattern(relayIfname, -1)
|
||||
metrics.RecordProxyCleanupRemoved(relayIfname, "comm_pattern", 1)
|
||||
logger.Debug("Removed idle communication pattern: %s", key)
|
||||
}
|
||||
return true
|
||||
})
|
||||
metrics.RecordProxyIdleCleanupDuration(relayIfname, "comm_pattern", time.Since(cleanupStart).Seconds())
|
||||
case <-s.ctx.Done():
|
||||
return
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user