Archived
migrated to beszel
This commit is contained in:
@@ -32,6 +32,63 @@ http:
|
|||||||
# - tracing-middleware@file
|
# - tracing-middleware@file
|
||||||
- error-pages-middleware@docker
|
- error-pages-middleware@docker
|
||||||
|
|
||||||
|
serversTransports:
|
||||||
|
insecureTransport:
|
||||||
|
insecureSkipVerify: true
|
||||||
|
|
||||||
|
routers:
|
||||||
|
pve:
|
||||||
|
rule: "Host(`pve.lan.ddnsgeek.com`)"
|
||||||
|
entryPoints:
|
||||||
|
- websecure
|
||||||
|
service: pve
|
||||||
|
middlewares:
|
||||||
|
- authelia@docker
|
||||||
|
tls:
|
||||||
|
certResolver: myresolver
|
||||||
|
options: mtls-private-admin@file
|
||||||
|
|
||||||
|
pbs:
|
||||||
|
rule: "Host(`pbs.lan.ddnsgeek.com`)"
|
||||||
|
entryPoints:
|
||||||
|
- websecure
|
||||||
|
service: pbs
|
||||||
|
middlewares:
|
||||||
|
- authelia@docker
|
||||||
|
tls:
|
||||||
|
certResolver: myresolver
|
||||||
|
options: mtls-private-admin@file
|
||||||
|
|
||||||
|
pdm:
|
||||||
|
rule: "Host(`pdm.lan.ddnsgeek.com`)"
|
||||||
|
entryPoints:
|
||||||
|
- websecure
|
||||||
|
service: pdm
|
||||||
|
middlewares:
|
||||||
|
- authelia@docker
|
||||||
|
tls:
|
||||||
|
certResolver: myresolver
|
||||||
|
options: mtls-private-admin@file
|
||||||
|
|
||||||
|
services:
|
||||||
|
pve:
|
||||||
|
loadBalancer:
|
||||||
|
serversTransport: insecureTransport
|
||||||
|
servers:
|
||||||
|
- url: "https://pve.sweet.home:8006"
|
||||||
|
|
||||||
|
pbs:
|
||||||
|
loadBalancer:
|
||||||
|
serversTransport: insecureTransport
|
||||||
|
servers:
|
||||||
|
- url: "https://pbs.sweet.home:8007"
|
||||||
|
|
||||||
|
pdm:
|
||||||
|
loadBalancer:
|
||||||
|
serversTransport: insecureTransport
|
||||||
|
servers:
|
||||||
|
- url: "https://pdm.sweet.home:8443"
|
||||||
|
|
||||||
tls:
|
tls:
|
||||||
options:
|
options:
|
||||||
mtls-private-admin:
|
mtls-private-admin:
|
||||||
|
|||||||
@@ -1,5 +1,5 @@
|
|||||||
log:
|
log:
|
||||||
level: INFO
|
level: DEBUG
|
||||||
|
|
||||||
accessLog:
|
accessLog:
|
||||||
filePath: /logs/access.log
|
filePath: /logs/access.log
|
||||||
|
|||||||
+1
-1
@@ -4,4 +4,4 @@ networks:
|
|||||||
ipam:
|
ipam:
|
||||||
config:
|
config:
|
||||||
- subnet: 172.21.0.0/16
|
- subnet: 172.21.0.0/16
|
||||||
monitor:
|
# monitor:
|
||||||
|
|||||||
@@ -0,0 +1,58 @@
|
|||||||
|
services:
|
||||||
|
beszel:
|
||||||
|
profiles: ["monitoring","all","beszel"]
|
||||||
|
image: henrygd/beszel:latest
|
||||||
|
container_name: beszel
|
||||||
|
restart: always
|
||||||
|
environment:
|
||||||
|
APP_URL: https://beszel.lan.ddnsgeek.com
|
||||||
|
ports:
|
||||||
|
- 8090:8090
|
||||||
|
volumes:
|
||||||
|
- ${DOCKER_VOLUMES}/beszel/data:/beszel_data
|
||||||
|
- ${DOCKER_VOLUMES}/beszel/socket:/beszel_socket
|
||||||
|
healthcheck:
|
||||||
|
test: ['CMD', '/beszel', 'health', '--url', 'http://localhost:8090']
|
||||||
|
interval: 120s
|
||||||
|
start_period: 10s
|
||||||
|
timeout: 5s
|
||||||
|
networks:
|
||||||
|
- traefik
|
||||||
|
# - beszel
|
||||||
|
# - monitor
|
||||||
|
labels:
|
||||||
|
- "traefik.enable=true"
|
||||||
|
- "traefik.http.routers.beszel.rule=Host(`beszel.lan.ddnsgeek.com`)"
|
||||||
|
- "traefik.http.routers.beszel.entrypoints=websecure"
|
||||||
|
- "traefik.http.routers.beszel.tls=true"
|
||||||
|
- "traefik.http.routers.beszel.tls.certresolver=myresolver"
|
||||||
|
- "traefik.http.services.beszel.loadbalancer.server.port=8090"
|
||||||
|
- "io.portainer.accesscontrol.public"
|
||||||
|
- "traefik.docker.network=core_traefik"
|
||||||
|
- "traefik.http.routers.beszel.tls.options=mtls-private-admin@file"
|
||||||
|
|
||||||
|
beszel-agent:
|
||||||
|
profiles: ["monitoring","all","beszel"]
|
||||||
|
image: henrygd/beszel-agent:latest
|
||||||
|
container_name: beszel-agent
|
||||||
|
restart: always
|
||||||
|
network_mode: host
|
||||||
|
volumes:
|
||||||
|
- ./beszel/agent:/var/lib/beszel-agent
|
||||||
|
- ./beszel/socket:/beszel_socket
|
||||||
|
# - /var/run/docker.sock:/var/run/docker.sock:ro
|
||||||
|
environment:
|
||||||
|
LISTEN: /beszel_socket/beszel.sock
|
||||||
|
HUB_URL: http://localhost:8090
|
||||||
|
KEY: 'ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIFPR9kwtC4TAeTRu46A7+opZsYpxqkRJ+x/ZyB2GWCeG'
|
||||||
|
TOKEN: ee2f2df3-d79b-4695-8840-bc936dd7b121
|
||||||
|
DOCKER_HOST: tcp://localhost:2375
|
||||||
|
healthcheck:
|
||||||
|
test: ['CMD', '/agent', 'health']
|
||||||
|
interval: 120s
|
||||||
|
# networks:
|
||||||
|
# - beszel
|
||||||
|
# - monitor
|
||||||
|
|
||||||
|
#networks:
|
||||||
|
# beszel:
|
||||||
@@ -1,15 +0,0 @@
|
|||||||
FROM python:3.11-slim
|
|
||||||
|
|
||||||
#RUN groupadd -g 1000 appuser || true && \
|
|
||||||
# useradd -m -u 1000 -g 1000 -s /bin/bash appuser
|
|
||||||
#RUN groupadd -g 999 docker || true && usermod -aG docker appuser
|
|
||||||
WORKDIR /app
|
|
||||||
COPY exporter.py .
|
|
||||||
|
|
||||||
RUN mkdir -p /data
|
|
||||||
|
|
||||||
RUN pip install --no-cache-dir docker prometheus_client requests pyyaml
|
|
||||||
|
|
||||||
#USER appuser
|
|
||||||
|
|
||||||
CMD ["python", "-u", "exporter.py"]
|
|
||||||
@@ -1,8 +0,0 @@
|
|||||||
FROM python:3.11-slim
|
|
||||||
|
|
||||||
WORKDIR /app
|
|
||||||
COPY exporter.py .
|
|
||||||
|
|
||||||
RUN pip install docker prometheus_client requests pyyaml
|
|
||||||
|
|
||||||
CMD ["python", "exporter.py"]
|
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
# Exporter image-mapping test example
|
|
||||||
|
|
||||||
Run the exporter in dry-run mode to print the `service -> image:tag` mapping without starting the metrics loop:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
SERVICES_UP_SCRIPT=/workspace/docker/services-up.sh python monitoring/docker-exporter/exporter.py --dry-run
|
|
||||||
```
|
|
||||||
|
|
||||||
Example output excerpt:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"crowdsec": "crowdsecurity/crowdsec:latest",
|
|
||||||
"docker-update-exporter": "python:3.11-slim",
|
|
||||||
"nextcloud-webapp": "nextcloud:production",
|
|
||||||
"node-red": "nodered/node-red:latest",
|
|
||||||
"prometheus": "prom/prometheus:latest",
|
|
||||||
"traefik": "traefik:3"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
This confirms the exporter now reports images for both:
|
|
||||||
- services with explicit `image:` values, and
|
|
||||||
- services using `build:` contexts.
|
|
||||||
@@ -1,59 +0,0 @@
|
|||||||
services:
|
|
||||||
docker-update-exporter:
|
|
||||||
profiles: ["monitoring","all","docker-exporter", "prometheus"]
|
|
||||||
build:
|
|
||||||
context: ${PROJECT_ROOT}/monitoring/docker-exporter
|
|
||||||
container_name: docker-update-exporter
|
|
||||||
# volumes:
|
|
||||||
# - /var/run/docker.sock:/var/run/docker.sock
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/docker-exporter/data:/data:rw
|
|
||||||
# - ${PROJECT_ROOT}/services-up.sh:/app/services-up.sh:ro
|
|
||||||
environment:
|
|
||||||
LOG_LEVEL: ${DOCKER_EXPORTER_LOG_LEVEL}
|
|
||||||
DOCKER_HOST: ${DOCKER_SOCKET_PROXY_HOST}
|
|
||||||
depends_on:
|
|
||||||
- docker-socket-proxy
|
|
||||||
|
|
||||||
volumes:
|
|
||||||
- ~/.docker/config.json:/root/.docker/config.json:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/docker-exporter/data:/data:rw
|
|
||||||
- ${DOCKER_VOLUMES}/docker-update-exporter-data:/data:rw
|
|
||||||
- ${PROJECT_ROOT}:/compose:ro
|
|
||||||
# - ${PROJECT_ROOT}/default-environment.env:/compose/default-environment.env:ro
|
|
||||||
# - ${PROJECT_ROOT}/default-network.yml:/compose/default-network.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/core/docker-compose.yml:/compose/core/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/prometheus/docker-compose.yml:/compose/monitoring/prometheus/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/gotify/docker-compose.yml:/compose/monitoring/gotify/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/grafana/docker-compose.yml:/compose/monitoring/grafana/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/portainer/docker-compose.yml:/compose/monitoring/portainer/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/uptime-kuma/docker-compose.yml:/compose/monitoring/uptime-kuma/docker-compose.yml:>
|
|
||||||
# - ${PROJECT_ROOT}/apps/gitea/docker-compose.yml:/compose/apps/gitea/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/gramps/docker-compose.yml:/compose/apps/gramps/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/nextcloud/docker-compose.yml:/compose/apps/nextcloud/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/passbolt/docker-compose.yml:/compose/apps/passbolt/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/searxng/docker-compose.yml:/compose/apps/searxng/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/shift-recorder/docker-compose.yml:/compose/apps/shift-recorder/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/stockfill/docker-compose.yml:/compose/apps/stockfill/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/node-red/docker-compose.yml:/compose/monitoring/node-red/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/core/test/docker-compose.yml:/compose/core/test/docker-compose.yml:ro
|
|
||||||
|
|
||||||
|
|
||||||
# ports:
|
|
||||||
# - "9105:9105"
|
|
||||||
restart: unless-stopped
|
|
||||||
cap_drop:
|
|
||||||
- ALL
|
|
||||||
security_opt:
|
|
||||||
- no-new-privileges:true
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
- monitor
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:9105/metrics')"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 5s
|
|
||||||
retries: 3
|
|
||||||
start_period: 10s
|
|
||||||
#volumes:
|
|
||||||
# docker-update-exporter-data:
|
|
||||||
# external: true
|
|
||||||
@@ -1,579 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
import argparse
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import time
|
|
||||||
import json
|
|
||||||
import logging
|
|
||||||
import docker
|
|
||||||
import yaml
|
|
||||||
from prometheus_client import Gauge, start_http_server
|
|
||||||
|
|
||||||
# --- Logging ---
|
|
||||||
LOG_LEVEL = os.getenv("LOG_LEVEL", "DEBUG").upper()
|
|
||||||
logging.basicConfig(
|
|
||||||
level=getattr(logging, LOG_LEVEL, logging.DEBUG),
|
|
||||||
format="%(asctime)s [%(levelname)s] %(message)s"
|
|
||||||
)
|
|
||||||
logger = logging.getLogger("docker-update-exporter")
|
|
||||||
|
|
||||||
# --- Config ---
|
|
||||||
EXPORTER_PORT = 9105
|
|
||||||
CHECK_INTERVAL = 3600
|
|
||||||
CACHE_TTL = int(os.getenv("CACHE_TTL", "300"))
|
|
||||||
SERVICES_UP_SCRIPT = os.getenv("SERVICES_UP_SCRIPT", "/compose/services-up.sh")
|
|
||||||
CACHE_FILE = os.getenv("CACHE_FILE", "/data/remote_digest_cache.json")
|
|
||||||
DRY_RUN = os.getenv("DRY_RUN", "false").lower() in ("1", "true", "yes")
|
|
||||||
|
|
||||||
try:
|
|
||||||
client = docker.from_env()
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Docker client unavailable at startup: {e}")
|
|
||||||
client = None
|
|
||||||
|
|
||||||
# --- Metrics ---
|
|
||||||
CONTAINER_UPDATE = Gauge(
|
|
||||||
"docker_container_update_available",
|
|
||||||
"1 if container image is out of date (compose drift or registry), 0 otherwise",
|
|
||||||
["container", "compose_image", "running_image", "com_docker_compose_project"]
|
|
||||||
)
|
|
||||||
|
|
||||||
LAST_CHECK = Gauge(
|
|
||||||
"docker_image_update_last_check_timestamp",
|
|
||||||
"Last time the update check ran (unix timestamp)"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def set_container_update_metric(container_name, compose_image, running_image, project_name, update_flag):
|
|
||||||
"""Set update metric for a container and log the emitted metric payload."""
|
|
||||||
metric_labels = {
|
|
||||||
"container": container_name,
|
|
||||||
"compose_image": compose_image or "unknown",
|
|
||||||
"running_image": running_image,
|
|
||||||
"com_docker_compose_project": project_name,
|
|
||||||
}
|
|
||||||
CONTAINER_UPDATE.labels(**metric_labels).set(update_flag)
|
|
||||||
logger.info(
|
|
||||||
"Metric emitted: docker_container_update_available=%s labels=%s",
|
|
||||||
update_flag,
|
|
||||||
metric_labels,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def set_last_check_metric():
|
|
||||||
"""Set and log the timestamp for the most recent check cycle."""
|
|
||||||
ts = time.time()
|
|
||||||
LAST_CHECK.set(ts)
|
|
||||||
logger.info("Metric emitted: docker_image_update_last_check_timestamp=%s", ts)
|
|
||||||
|
|
||||||
# --- Persistent Cache ---
|
|
||||||
def load_cache():
|
|
||||||
if not os.path.exists(CACHE_FILE):
|
|
||||||
logger.info(f"Cache file does not exist yet: {CACHE_FILE}")
|
|
||||||
return {}
|
|
||||||
try:
|
|
||||||
with open(CACHE_FILE, "r") as f:
|
|
||||||
cache = json.load(f)
|
|
||||||
logger.info(f"Loaded {len(cache)} cached remote digests")
|
|
||||||
return cache
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to load cache: {e}")
|
|
||||||
return {}
|
|
||||||
|
|
||||||
def save_cache():
|
|
||||||
try:
|
|
||||||
os.makedirs(os.path.dirname(CACHE_FILE), exist_ok=True)
|
|
||||||
with open(CACHE_FILE, "w") as f:
|
|
||||||
json.dump(REMOTE_DIGEST_CACHE, f)
|
|
||||||
logger.debug(f"Saved {len(REMOTE_DIGEST_CACHE)} remote digests to cache")
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to save cache: {e}")
|
|
||||||
|
|
||||||
REMOTE_DIGEST_CACHE = load_cache()
|
|
||||||
now = time.time()
|
|
||||||
REMOTE_DIGEST_CACHE = {
|
|
||||||
image: (digest, ts)
|
|
||||||
for image, (digest, ts) in REMOTE_DIGEST_CACHE.items()
|
|
||||||
if now - ts < CACHE_TTL
|
|
||||||
}
|
|
||||||
|
|
||||||
# --- Helpers ---
|
|
||||||
def get_project_prefix_from_script(script_path):
|
|
||||||
prefix = "core-"
|
|
||||||
if not os.path.exists(script_path):
|
|
||||||
return prefix
|
|
||||||
try:
|
|
||||||
with open(script_path) as f:
|
|
||||||
for line in f:
|
|
||||||
m = re.match(r'PROJECT\s*=\s*["\']?([^"\']+)', line)
|
|
||||||
if m:
|
|
||||||
return m.group(1) + "-"
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Failed reading project prefix: {e}")
|
|
||||||
return prefix
|
|
||||||
|
|
||||||
def get_local_digest(image_name):
|
|
||||||
"""
|
|
||||||
Return the local digest for the specific image reference.
|
|
||||||
"""
|
|
||||||
if client is None:
|
|
||||||
return None
|
|
||||||
|
|
||||||
try:
|
|
||||||
img = client.images.get(image_name)
|
|
||||||
digests = img.attrs.get("RepoDigests", [])
|
|
||||||
|
|
||||||
logger.debug(f"RepoDigests for {image_name}: {digests}")
|
|
||||||
|
|
||||||
for entry in digests:
|
|
||||||
if "@" in entry:
|
|
||||||
digest = entry.split("@", 1)[1]
|
|
||||||
logger.debug(f"Local digest for {image_name}: {digest}")
|
|
||||||
return digest
|
|
||||||
|
|
||||||
logger.debug(f"No RepoDigest found for {image_name}")
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.debug(f"Could not get local digest for {image_name}: {e}")
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
def get_remote_digest(image_name):
|
|
||||||
"""
|
|
||||||
Return the upstream digest for the exact platform-specific image that Docker
|
|
||||||
would pull on this host. This avoids false positives with multi-arch images
|
|
||||||
where the registry manifest-list digest differs from the pulled image digest.
|
|
||||||
"""
|
|
||||||
now = time.time()
|
|
||||||
|
|
||||||
cached = REMOTE_DIGEST_CACHE.get(image_name)
|
|
||||||
if cached:
|
|
||||||
digest, ts = cached
|
|
||||||
if now - ts < CACHE_TTL:
|
|
||||||
logger.debug(f"Using cached remote digest for {image_name}: {digest}")
|
|
||||||
return digest
|
|
||||||
|
|
||||||
if client is None:
|
|
||||||
return None
|
|
||||||
|
|
||||||
try:
|
|
||||||
registry_data = client.images.get_registry_data(image_name)
|
|
||||||
|
|
||||||
digest = None
|
|
||||||
|
|
||||||
# docker SDK versions differ; try the common fields in order
|
|
||||||
if hasattr(registry_data, "id") and registry_data.id:
|
|
||||||
digest = registry_data.id
|
|
||||||
elif hasattr(registry_data, "attrs"):
|
|
||||||
digest = (
|
|
||||||
registry_data.attrs.get("Descriptor", {}).get("digest")
|
|
||||||
or registry_data.attrs.get("digest")
|
|
||||||
)
|
|
||||||
|
|
||||||
if digest:
|
|
||||||
REMOTE_DIGEST_CACHE[image_name] = (digest, now)
|
|
||||||
save_cache()
|
|
||||||
logger.debug(f"Remote digest for {image_name}: {digest}")
|
|
||||||
return digest
|
|
||||||
|
|
||||||
logger.warning(f"No remote digest found for {image_name}")
|
|
||||||
return None
|
|
||||||
except Exception as e:
|
|
||||||
logger.debug(f"Error fetching remote digest for {image_name}: {e}")
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
# --- Dockerfile Image Extraction ---
|
|
||||||
def parse_dockerfile_for_image(dockerfile_path):
|
|
||||||
if not os.path.exists(dockerfile_path):
|
|
||||||
return None
|
|
||||||
|
|
||||||
try:
|
|
||||||
arg_defaults = {}
|
|
||||||
last_from = None
|
|
||||||
with open(dockerfile_path) as df:
|
|
||||||
for line in df:
|
|
||||||
line = line.strip()
|
|
||||||
if not line or line.startswith("#"):
|
|
||||||
continue
|
|
||||||
|
|
||||||
if line.upper().startswith("ARG "):
|
|
||||||
arg_body = line[4:].strip()
|
|
||||||
if "=" in arg_body:
|
|
||||||
key, value = arg_body.split("=", 1)
|
|
||||||
arg_defaults[key.strip()] = value.strip()
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Prefer LABEL with image if present.
|
|
||||||
if "LABEL" in line and "image=" in line:
|
|
||||||
match = re.search(r'image=["\']?([^"\']+)["\']?', line)
|
|
||||||
if match:
|
|
||||||
image_name = normalize_image_name(substitute_dockerfile_args(match.group(1), arg_defaults))
|
|
||||||
logger.debug(f"Found LABEL image={image_name} in {dockerfile_path}")
|
|
||||||
return image_name
|
|
||||||
|
|
||||||
if line.upper().startswith("FROM "):
|
|
||||||
from_clause = line[5:].strip()
|
|
||||||
if from_clause.startswith("--"):
|
|
||||||
split_clause = from_clause.split(None, 1)
|
|
||||||
if len(split_clause) < 2:
|
|
||||||
continue
|
|
||||||
from_clause = split_clause[1]
|
|
||||||
|
|
||||||
parts = from_clause.split()
|
|
||||||
if not parts:
|
|
||||||
continue
|
|
||||||
|
|
||||||
candidate = substitute_dockerfile_args(parts[0], arg_defaults)
|
|
||||||
if candidate and candidate.lower() != "scratch":
|
|
||||||
last_from = normalize_image_name(candidate)
|
|
||||||
|
|
||||||
if last_from:
|
|
||||||
logger.debug(f"Found base FROM {last_from} in {dockerfile_path}")
|
|
||||||
return last_from
|
|
||||||
except Exception as e:
|
|
||||||
logger.debug(f"Error reading Dockerfile {dockerfile_path}: {e}")
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
def normalize_image_name(image_name):
|
|
||||||
if not image_name:
|
|
||||||
return None
|
|
||||||
if "@" in image_name:
|
|
||||||
return image_name
|
|
||||||
if ":" in image_name.rsplit("/", 1)[-1]:
|
|
||||||
return image_name
|
|
||||||
return f"{image_name}:latest"
|
|
||||||
|
|
||||||
def is_compose_build_placeholder(image_name, project_name):
|
|
||||||
if not image_name:
|
|
||||||
return False
|
|
||||||
candidate = str(image_name)
|
|
||||||
project_prefix = f"{project_name}-"
|
|
||||||
if candidate.startswith(project_prefix):
|
|
||||||
return True
|
|
||||||
# Keep backward-compatible behavior for historical default project prefix.
|
|
||||||
return candidate.startswith("core-")
|
|
||||||
|
|
||||||
def substitute_dockerfile_args(value, arg_defaults):
|
|
||||||
if not value:
|
|
||||||
return value
|
|
||||||
|
|
||||||
pattern = re.compile(r"\$\{([^}]+)\}|\$([A-Za-z_][A-Za-z0-9_]*)")
|
|
||||||
|
|
||||||
def replacer(match):
|
|
||||||
expr = match.group(1)
|
|
||||||
simple = match.group(2)
|
|
||||||
if simple:
|
|
||||||
return arg_defaults.get(simple, "")
|
|
||||||
|
|
||||||
if ":-" in expr:
|
|
||||||
var_name, default_value = expr.split(":-", 1)
|
|
||||||
return arg_defaults.get(var_name, default_value)
|
|
||||||
if "-" in expr:
|
|
||||||
var_name, default_value = expr.split("-", 1)
|
|
||||||
return arg_defaults.get(var_name, default_value)
|
|
||||||
return arg_defaults.get(expr, "")
|
|
||||||
|
|
||||||
return pattern.sub(replacer, value)
|
|
||||||
|
|
||||||
def expand_compose_path(path_value, project_root):
|
|
||||||
raw = str(path_value)
|
|
||||||
raw = raw.replace("${PROJECT_ROOT}", project_root).replace("$PROJECT_ROOT", project_root)
|
|
||||||
return os.path.expandvars(raw)
|
|
||||||
|
|
||||||
def get_project_root_from_script(script_path):
|
|
||||||
if not script_path:
|
|
||||||
return os.getcwd()
|
|
||||||
return os.path.dirname(os.path.abspath(script_path))
|
|
||||||
|
|
||||||
# --- Compose parsing ---
|
|
||||||
def get_compose_files_from_script(script_path):
|
|
||||||
files = []
|
|
||||||
if not os.path.exists(script_path):
|
|
||||||
return files
|
|
||||||
base_dir = get_project_root_from_script(script_path)
|
|
||||||
|
|
||||||
def _clean_compose_path(raw_path):
|
|
||||||
cleaned = str(raw_path).strip().strip(",")
|
|
||||||
if (cleaned.startswith('"') and cleaned.endswith('"')) or (
|
|
||||||
cleaned.startswith("'") and cleaned.endswith("'")
|
|
||||||
):
|
|
||||||
cleaned = cleaned[1:-1]
|
|
||||||
expanded = expand_compose_path(cleaned, base_dir)
|
|
||||||
if os.path.isabs(expanded):
|
|
||||||
return os.path.normpath(expanded)
|
|
||||||
return os.path.normpath(os.path.join(base_dir, expanded))
|
|
||||||
|
|
||||||
try:
|
|
||||||
with open(script_path) as f:
|
|
||||||
content = f.read()
|
|
||||||
match = re.search(r'FILES\s*=\s*\((.*?)\)', content, re.DOTALL)
|
|
||||||
if match:
|
|
||||||
for line in match.group(1).splitlines():
|
|
||||||
line = line.strip()
|
|
||||||
if line.startswith("-f"):
|
|
||||||
path = line[2:].strip()
|
|
||||||
if path:
|
|
||||||
full = _clean_compose_path(path)
|
|
||||||
files.append(full)
|
|
||||||
|
|
||||||
# services-up.sh can append many compose files at runtime via:
|
|
||||||
# FILES+=(-f "$file") done < <(find "$PROJECT_ROOT/apps" ...)
|
|
||||||
# Mirror that behavior here so we can map service->compose image.
|
|
||||||
root_dirs = []
|
|
||||||
find_match = re.search(r'find\s+(.*?)\s+\\\s*\n', content)
|
|
||||||
if find_match:
|
|
||||||
for token in re.findall(r'"([^"]+)"|\'([^\']+)\'', find_match.group(1)):
|
|
||||||
candidate = token[0] or token[1]
|
|
||||||
if candidate:
|
|
||||||
root_dirs.append(_clean_compose_path(candidate))
|
|
||||||
else:
|
|
||||||
root_dirs = [
|
|
||||||
os.path.join(base_dir, "apps"),
|
|
||||||
os.path.join(base_dir, "monitoring"),
|
|
||||||
os.path.join(base_dir, "core"),
|
|
||||||
]
|
|
||||||
|
|
||||||
for root_dir in root_dirs:
|
|
||||||
if not os.path.isdir(root_dir):
|
|
||||||
continue
|
|
||||||
for candidate in sorted(os.listdir(root_dir)):
|
|
||||||
svc_dir = os.path.join(root_dir, candidate)
|
|
||||||
if not os.path.isdir(svc_dir):
|
|
||||||
continue
|
|
||||||
for compose_name in ("docker-compose.yml", "docker-compose.yaml"):
|
|
||||||
compose_path = os.path.join(svc_dir, compose_name)
|
|
||||||
if os.path.exists(compose_path):
|
|
||||||
files.append(compose_path)
|
|
||||||
|
|
||||||
# Preserve order while removing duplicates.
|
|
||||||
deduped = []
|
|
||||||
seen = set()
|
|
||||||
for path in files:
|
|
||||||
if path in seen:
|
|
||||||
continue
|
|
||||||
seen.add(path)
|
|
||||||
deduped.append(path)
|
|
||||||
files = deduped
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Failed parsing services-up.sh: {e}")
|
|
||||||
return files
|
|
||||||
|
|
||||||
def parse_project_name_from_script(script_path):
|
|
||||||
project = "core"
|
|
||||||
if not os.path.exists(script_path):
|
|
||||||
return project
|
|
||||||
try:
|
|
||||||
with open(script_path) as f:
|
|
||||||
for line in f:
|
|
||||||
m = re.match(r'PROJECT\s*=\s*["\']?([^"\']+)', line)
|
|
||||||
if m:
|
|
||||||
project = m.group(1)
|
|
||||||
break
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Failed reading project name: {e}")
|
|
||||||
return project
|
|
||||||
|
|
||||||
def resolve_local_build_image(service_name, project_name):
|
|
||||||
if client is None:
|
|
||||||
return None
|
|
||||||
try:
|
|
||||||
images = client.images.list(filters={"label": f"com.docker.compose.service={service_name}"})
|
|
||||||
for image in images:
|
|
||||||
labels = image.attrs.get("Config", {}).get("Labels", {}) or {}
|
|
||||||
if labels.get("com.docker.compose.project") != project_name:
|
|
||||||
continue
|
|
||||||
for tag in image.tags:
|
|
||||||
if tag and "<none>" not in tag:
|
|
||||||
logger.debug(f"Resolved local compose image for {service_name}: {tag}")
|
|
||||||
return normalize_image_name(tag)
|
|
||||||
except Exception as e:
|
|
||||||
logger.debug(f"Could not inspect local build metadata for {service_name}: {e}")
|
|
||||||
return None
|
|
||||||
|
|
||||||
def parse_compose_services(compose_files, project_name, project_root):
|
|
||||||
svc_map = {}
|
|
||||||
for f in compose_files:
|
|
||||||
if not os.path.exists(f):
|
|
||||||
logger.warning(f"Compose file from services-up.sh is missing: {f}")
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
with open(f) as stream:
|
|
||||||
data = yaml.safe_load(stream) or {}
|
|
||||||
for svc_name, svc_def in data.get("services", {}).items():
|
|
||||||
image = normalize_image_name(svc_def.get("image"))
|
|
||||||
profiles = svc_def.get("profiles", [])
|
|
||||||
build_ctx = svc_def.get("build")
|
|
||||||
dockerfile_path = None
|
|
||||||
from_dockerfile = None
|
|
||||||
local_built_image = None
|
|
||||||
|
|
||||||
if build_ctx:
|
|
||||||
if isinstance(build_ctx, dict):
|
|
||||||
context = build_ctx.get("context", ".")
|
|
||||||
dockerfile = build_ctx.get("dockerfile", "Dockerfile")
|
|
||||||
else:
|
|
||||||
context = build_ctx
|
|
||||||
dockerfile = "Dockerfile"
|
|
||||||
|
|
||||||
compose_dir = os.path.dirname(f)
|
|
||||||
context_expanded = expand_compose_path(context, project_root)
|
|
||||||
if os.path.isabs(context_expanded):
|
|
||||||
context_path = context_expanded
|
|
||||||
else:
|
|
||||||
context_path = os.path.normpath(os.path.join(compose_dir, context_expanded))
|
|
||||||
dockerfile_expanded = expand_compose_path(dockerfile, project_root)
|
|
||||||
dockerfile_path = os.path.normpath(os.path.join(context_path, dockerfile_expanded))
|
|
||||||
from_dockerfile = normalize_image_name(parse_dockerfile_for_image(dockerfile_path))
|
|
||||||
local_built_image = resolve_local_build_image(svc_name, project_name)
|
|
||||||
|
|
||||||
placeholder_image = is_compose_build_placeholder(image, project_name) or is_compose_build_placeholder(local_built_image, project_name)
|
|
||||||
if placeholder_image:
|
|
||||||
resolved_image = from_dockerfile or image or local_built_image or f"{project_name}-{svc_name}:latest"
|
|
||||||
else:
|
|
||||||
resolved_image = image or local_built_image or from_dockerfile or f"{project_name}-{svc_name}:latest"
|
|
||||||
|
|
||||||
svc_map[svc_name] = {
|
|
||||||
"image": resolved_image,
|
|
||||||
"profiles": profiles,
|
|
||||||
"build_context": build_ctx,
|
|
||||||
"compose_file": f,
|
|
||||||
"dockerfile": dockerfile_path
|
|
||||||
}
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Failed parsing {f}: {e}")
|
|
||||||
|
|
||||||
logger.debug(f"Service image mapping: {svc_map}")
|
|
||||||
return svc_map
|
|
||||||
|
|
||||||
# --- Main check ---
|
|
||||||
def check_containers():
|
|
||||||
if client is None:
|
|
||||||
logger.error("Docker client is unavailable; skipping check cycle")
|
|
||||||
return
|
|
||||||
|
|
||||||
set_last_check_metric()
|
|
||||||
CONTAINER_UPDATE.clear()
|
|
||||||
|
|
||||||
project_name = parse_project_name_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
project_root = get_project_root_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
compose_files = get_compose_files_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
svc_map = parse_compose_services(compose_files, project_name, project_root)
|
|
||||||
|
|
||||||
containers = client.containers.list()
|
|
||||||
pending_metrics = []
|
|
||||||
remote_targets = set()
|
|
||||||
|
|
||||||
for container in containers:
|
|
||||||
proj = container.labels.get("com.docker.compose.project")
|
|
||||||
if not proj:
|
|
||||||
continue
|
|
||||||
|
|
||||||
svc = container.labels.get("com.docker.compose.service")
|
|
||||||
running = container.attrs["Config"]["Image"]
|
|
||||||
|
|
||||||
compose_image = None
|
|
||||||
if svc in svc_map:
|
|
||||||
compose_image = svc_map[svc]["image"]
|
|
||||||
|
|
||||||
local_digest = get_local_digest(running)
|
|
||||||
remote_target = compose_image or running
|
|
||||||
|
|
||||||
# If we cannot determine a local digest, we cannot compare and should
|
|
||||||
# avoid spending a registry lookup for this container.
|
|
||||||
if local_digest:
|
|
||||||
remote_targets.add(remote_target)
|
|
||||||
|
|
||||||
pending_metrics.append({
|
|
||||||
"container_name": container.name,
|
|
||||||
"service": svc,
|
|
||||||
"compose_image": compose_image,
|
|
||||||
"running_image": running,
|
|
||||||
"project_name": proj,
|
|
||||||
"remote_target": remote_target,
|
|
||||||
"local_digest": local_digest,
|
|
||||||
})
|
|
||||||
|
|
||||||
remote_digests = {target: get_remote_digest(target) for target in remote_targets}
|
|
||||||
|
|
||||||
for payload in pending_metrics:
|
|
||||||
local_digest = payload["local_digest"]
|
|
||||||
remote_target = payload["remote_target"]
|
|
||||||
remote_digest = remote_digests.get(remote_target)
|
|
||||||
update_flag = 1 if (local_digest and remote_digest and local_digest != remote_digest) else 0
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
"Digest comparison: container=%s service=%s running=%s target=%s local=%s remote=%s",
|
|
||||||
payload["container_name"],
|
|
||||||
payload["service"],
|
|
||||||
payload["running_image"],
|
|
||||||
remote_target,
|
|
||||||
local_digest,
|
|
||||||
remote_digest,
|
|
||||||
)
|
|
||||||
|
|
||||||
set_container_update_metric(
|
|
||||||
container_name=payload["container_name"],
|
|
||||||
compose_image=payload["compose_image"],
|
|
||||||
running_image=payload["running_image"],
|
|
||||||
project_name=payload["project_name"],
|
|
||||||
update_flag=update_flag,
|
|
||||||
)
|
|
||||||
|
|
||||||
def dump_service_image_mapping():
|
|
||||||
project_name = parse_project_name_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
project_root = get_project_root_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
compose_files = get_compose_files_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
svc_map = parse_compose_services(compose_files, project_name, project_root)
|
|
||||||
mapping = {name: data["image"] for name, data in sorted(svc_map.items())}
|
|
||||||
logger.info("Service to image mapping:")
|
|
||||||
logger.info(json.dumps(mapping, indent=2, sort_keys=True))
|
|
||||||
return mapping
|
|
||||||
|
|
||||||
# --- Runner ---
|
|
||||||
if __name__ == "__main__":
|
|
||||||
parser = argparse.ArgumentParser(description="Docker image update exporter")
|
|
||||||
parser.add_argument("--dry-run", action="store_true", help="Only print service->image mapping and exit")
|
|
||||||
parser.add_argument(
|
|
||||||
"--services-up-script",
|
|
||||||
default=SERVICES_UP_SCRIPT,
|
|
||||||
help=f"Path to services-up script (default: {SERVICES_UP_SCRIPT})",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--cache-file",
|
|
||||||
default=CACHE_FILE,
|
|
||||||
help=f"Path to digest cache file (default: {CACHE_FILE})",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
"--log-level",
|
|
||||||
default=LOG_LEVEL,
|
|
||||||
help=f"Logging level (default: {LOG_LEVEL})",
|
|
||||||
)
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
effective_log_level = str(args.log_level).upper()
|
|
||||||
logging.getLogger().setLevel(getattr(logging, effective_log_level, logging.DEBUG))
|
|
||||||
logger.setLevel(getattr(logging, effective_log_level, logging.DEBUG))
|
|
||||||
|
|
||||||
SERVICES_UP_SCRIPT = args.services_up_script
|
|
||||||
CACHE_FILE = args.cache_file
|
|
||||||
REMOTE_DIGEST_CACHE = load_cache()
|
|
||||||
now = time.time()
|
|
||||||
REMOTE_DIGEST_CACHE = {
|
|
||||||
image: (digest, ts)
|
|
||||||
for image, (digest, ts) in REMOTE_DIGEST_CACHE.items()
|
|
||||||
if now - ts < CACHE_TTL
|
|
||||||
}
|
|
||||||
|
|
||||||
if DRY_RUN or args.dry_run:
|
|
||||||
dump_service_image_mapping()
|
|
||||||
raise SystemExit(0)
|
|
||||||
|
|
||||||
start_http_server(EXPORTER_PORT)
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
check_containers()
|
|
||||||
except Exception as e:
|
|
||||||
logger.exception(f"update check failed: {e}")
|
|
||||||
time.sleep(CHECK_INTERVAL)
|
|
||||||
@@ -1,514 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import time
|
|
||||||
import json
|
|
||||||
import logging
|
|
||||||
import docker
|
|
||||||
import requests
|
|
||||||
import yaml
|
|
||||||
from prometheus_client import Gauge, start_http_server
|
|
||||||
|
|
||||||
# --- Logging ---
|
|
||||||
LOG_LEVEL = os.getenv("LOG_LEVEL", "INFO").upper()
|
|
||||||
|
|
||||||
logging.basicConfig(
|
|
||||||
level=getattr(logging, LOG_LEVEL, logging.INFO),
|
|
||||||
format="%(asctime)s [%(levelname)s] %(message)s"
|
|
||||||
)
|
|
||||||
|
|
||||||
logger = logging.getLogger("docker-update-exporter")
|
|
||||||
|
|
||||||
# --- Config ---
|
|
||||||
EXPORTER_PORT = 9105
|
|
||||||
CHECK_INTERVAL = 60
|
|
||||||
CACHE_TTL = 6 * 3600
|
|
||||||
SERVICES_UP_SCRIPT = "/compose/services-up.sh"
|
|
||||||
CACHE_FILE = "/data/remote_digest_cache.json"
|
|
||||||
|
|
||||||
client = docker.from_env()
|
|
||||||
|
|
||||||
# --- Metrics ---
|
|
||||||
CONTAINER_UPDATE = Gauge(
|
|
||||||
"docker_container_update_available",
|
|
||||||
"1 if container image is out of date (compose drift or registry), 0 otherwise",
|
|
||||||
["container", "compose_image", "running_image", "com_docker_compose_project"]
|
|
||||||
)
|
|
||||||
|
|
||||||
LAST_CHECK = Gauge(
|
|
||||||
"docker_image_update_last_check_timestamp",
|
|
||||||
"Last time the update check ran (unix timestamp)"
|
|
||||||
)
|
|
||||||
|
|
||||||
# --- Persistent Cache ---
|
|
||||||
|
|
||||||
def load_cache():
|
|
||||||
if not os.path.exists(CACHE_FILE):
|
|
||||||
logger.info(f"Cache file does not exist yet: {CACHE_FILE}")
|
|
||||||
return {}
|
|
||||||
|
|
||||||
try:
|
|
||||||
with open(CACHE_FILE, "r") as f:
|
|
||||||
cache = json.load(f)
|
|
||||||
logger.info(f"Loaded {len(cache)} cached remote digests")
|
|
||||||
logger.debug(f"Cache contents: {cache}")
|
|
||||||
return cache
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to load cache from {CACHE_FILE}: {e}")
|
|
||||||
return {}
|
|
||||||
|
|
||||||
def save_cache():
|
|
||||||
try:
|
|
||||||
os.makedirs(os.path.dirname(CACHE_FILE), exist_ok=True)
|
|
||||||
with open(CACHE_FILE, "w") as f:
|
|
||||||
json.dump(REMOTE_DIGEST_CACHE, f)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Saved {len(REMOTE_DIGEST_CACHE)} entries to cache file {CACHE_FILE}"
|
|
||||||
)
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to save cache to {CACHE_FILE}: {e}")
|
|
||||||
|
|
||||||
REMOTE_DIGEST_CACHE = load_cache()
|
|
||||||
|
|
||||||
# --- Helpers ---
|
|
||||||
|
|
||||||
def get_project_prefix_from_script(script_path):
|
|
||||||
project_prefix = "core-" # fallback
|
|
||||||
|
|
||||||
if not os.path.exists(script_path):
|
|
||||||
logger.warning(
|
|
||||||
f"services-up script not found at {script_path}, using fallback project prefix {project_prefix}"
|
|
||||||
)
|
|
||||||
return project_prefix
|
|
||||||
|
|
||||||
try:
|
|
||||||
with open(script_path, "r") as f:
|
|
||||||
for line in f:
|
|
||||||
line = line.strip()
|
|
||||||
m = re.match(r'PROJECT\s*=\s*["\']?([^"\']+)["\']?', line)
|
|
||||||
if m:
|
|
||||||
project_prefix = m.group(1) + "-"
|
|
||||||
logger.debug(
|
|
||||||
f"Detected compose project prefix from script: {project_prefix}"
|
|
||||||
)
|
|
||||||
break
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed reading project prefix from {script_path}: {e}")
|
|
||||||
|
|
||||||
return project_prefix
|
|
||||||
|
|
||||||
def get_local_digest(image_name):
|
|
||||||
try:
|
|
||||||
img = client.images.get(image_name)
|
|
||||||
digests = img.attrs.get("RepoDigests", [])
|
|
||||||
|
|
||||||
logger.debug(f"Local RepoDigests for {image_name}: {digests}")
|
|
||||||
|
|
||||||
if digests:
|
|
||||||
digest = digests[0].split("@")[1]
|
|
||||||
logger.debug(f"Local digest for {image_name}: {digest}")
|
|
||||||
return digest
|
|
||||||
|
|
||||||
logger.info(f"No local digest found for image {image_name}")
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(f"Failed to retrieve local digest for {image_name}: {e}")
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
def get_remote_digest(image_name):
|
|
||||||
now = time.time()
|
|
||||||
original = image_name
|
|
||||||
|
|
||||||
# Cache hit
|
|
||||||
if original in REMOTE_DIGEST_CACHE:
|
|
||||||
digest, ts = REMOTE_DIGEST_CACHE[original]
|
|
||||||
age = now - ts
|
|
||||||
|
|
||||||
if age < CACHE_TTL:
|
|
||||||
logger.debug(
|
|
||||||
f"Using cached remote digest for {original} "
|
|
||||||
f"(age={int(age)}s, ttl={CACHE_TTL}s): {digest}"
|
|
||||||
)
|
|
||||||
return digest
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
f"Cache entry expired for {original} "
|
|
||||||
f"(age={int(age)}s > ttl={CACHE_TTL}s)"
|
|
||||||
)
|
|
||||||
|
|
||||||
try:
|
|
||||||
if "/" not in image_name:
|
|
||||||
registry = "docker.io"
|
|
||||||
repo = "library/" + image_name
|
|
||||||
else:
|
|
||||||
parts = image_name.split("/")
|
|
||||||
if "." in parts[0] or ":" in parts[0]:
|
|
||||||
registry = parts[0]
|
|
||||||
repo = "/".join(parts[1:])
|
|
||||||
else:
|
|
||||||
registry = "docker.io"
|
|
||||||
repo = image_name
|
|
||||||
|
|
||||||
if ":" in repo:
|
|
||||||
repo, tag = repo.rsplit(":", 1)
|
|
||||||
else:
|
|
||||||
tag = "latest"
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Resolving remote digest for {original}: "
|
|
||||||
f"registry={registry}, repo={repo}, tag={tag}"
|
|
||||||
)
|
|
||||||
|
|
||||||
token = None
|
|
||||||
manifest_url = None
|
|
||||||
|
|
||||||
if registry in ["docker.io", "registry-1.docker.io"]:
|
|
||||||
logger.debug(f"Requesting Docker Hub token for {repo}")
|
|
||||||
|
|
||||||
token_res = requests.get(
|
|
||||||
"https://auth.docker.io/token",
|
|
||||||
params={
|
|
||||||
"service": "registry.docker.io",
|
|
||||||
"scope": f"repository:{repo}:pull"
|
|
||||||
},
|
|
||||||
timeout=10
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Docker Hub token response for {repo}: "
|
|
||||||
f"status={token_res.status_code}"
|
|
||||||
)
|
|
||||||
|
|
||||||
token = token_res.json().get("token")
|
|
||||||
manifest_url = (
|
|
||||||
f"https://registry-1.docker.io/v2/{repo}/manifests/{tag}"
|
|
||||||
)
|
|
||||||
|
|
||||||
elif registry == "ghcr.io":
|
|
||||||
logger.debug(f"Requesting GHCR token for {repo}")
|
|
||||||
|
|
||||||
token_res = requests.get(
|
|
||||||
"https://ghcr.io/token",
|
|
||||||
params={
|
|
||||||
"service": "ghcr.io",
|
|
||||||
"scope": f"repository:{repo}:pull"
|
|
||||||
},
|
|
||||||
timeout=10
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"GHCR token response for {repo}: "
|
|
||||||
f"status={token_res.status_code}"
|
|
||||||
)
|
|
||||||
|
|
||||||
token = token_res.json().get("token")
|
|
||||||
manifest_url = f"https://ghcr.io/v2/{repo}/manifests/{tag}"
|
|
||||||
|
|
||||||
else:
|
|
||||||
logger.warning(
|
|
||||||
f"Unsupported registry '{registry}' for image {original}"
|
|
||||||
)
|
|
||||||
return None
|
|
||||||
|
|
||||||
if not token:
|
|
||||||
logger.warning(
|
|
||||||
f"No authentication token returned for {original}"
|
|
||||||
)
|
|
||||||
return None
|
|
||||||
|
|
||||||
logger.debug(f"Requesting manifest for {original}: {manifest_url}")
|
|
||||||
|
|
||||||
res = requests.get(
|
|
||||||
manifest_url,
|
|
||||||
headers={
|
|
||||||
"Authorization": f"Bearer {token}",
|
|
||||||
"Accept": "application/vnd.docker.distribution.manifest.v2+json"
|
|
||||||
},
|
|
||||||
timeout=10
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Manifest response for {original}: "
|
|
||||||
f"status={res.status_code}"
|
|
||||||
)
|
|
||||||
|
|
||||||
if res.status_code == 200:
|
|
||||||
digest = res.headers.get("Docker-Content-Digest")
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
f"Fetched remote digest for {original}: {digest}"
|
|
||||||
)
|
|
||||||
|
|
||||||
REMOTE_DIGEST_CACHE[original] = (digest, now)
|
|
||||||
save_cache()
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Cached remote digest for {original}: {digest}"
|
|
||||||
)
|
|
||||||
|
|
||||||
return digest
|
|
||||||
|
|
||||||
if res.status_code == 429:
|
|
||||||
logger.warning(
|
|
||||||
f"Registry rate limit hit while fetching {original}"
|
|
||||||
)
|
|
||||||
elif res.status_code in [401, 403]:
|
|
||||||
logger.warning(
|
|
||||||
f"Authentication failed while fetching {original}: "
|
|
||||||
f"status={res.status_code}"
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
logger.warning(
|
|
||||||
f"Unexpected manifest response for {original}: "
|
|
||||||
f"status={res.status_code}, body={res.text[:250]}"
|
|
||||||
)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed to fetch remote digest for {original}: {e}")
|
|
||||||
|
|
||||||
return None
|
|
||||||
|
|
||||||
def get_compose_files_from_script(script_path):
|
|
||||||
files = []
|
|
||||||
|
|
||||||
if not os.path.exists(script_path):
|
|
||||||
logger.error(f"services-up script not found: {script_path}")
|
|
||||||
return files
|
|
||||||
|
|
||||||
base_dir = os.path.dirname(script_path)
|
|
||||||
|
|
||||||
try:
|
|
||||||
with open(script_path, "r") as f:
|
|
||||||
content = f.read()
|
|
||||||
|
|
||||||
match = re.search(r'FILES\s*=\s*\((.*?)\)', content, re.DOTALL)
|
|
||||||
|
|
||||||
if not match:
|
|
||||||
logger.warning(
|
|
||||||
f"No FILES=(...) block found in {script_path}"
|
|
||||||
)
|
|
||||||
return files
|
|
||||||
|
|
||||||
lines = match.group(1).splitlines()
|
|
||||||
|
|
||||||
for line in lines:
|
|
||||||
line = line.strip()
|
|
||||||
|
|
||||||
if line.startswith("-f"):
|
|
||||||
rel_path = line[2:].strip()
|
|
||||||
|
|
||||||
if rel_path:
|
|
||||||
full_path = os.path.normpath(
|
|
||||||
os.path.join(base_dir, rel_path)
|
|
||||||
)
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Resolved compose file: {rel_path} -> {full_path}"
|
|
||||||
)
|
|
||||||
|
|
||||||
files.append(full_path)
|
|
||||||
|
|
||||||
logger.info(f"Found {len(files)} compose files")
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed parsing compose files from {script_path}: {e}")
|
|
||||||
|
|
||||||
return files
|
|
||||||
|
|
||||||
def parse_compose_files(compose_files):
|
|
||||||
service_to_image = {}
|
|
||||||
|
|
||||||
for f in compose_files:
|
|
||||||
if not os.path.exists(f):
|
|
||||||
logger.warning(f"Compose file missing: {f}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
try:
|
|
||||||
with open(f, "r") as stream:
|
|
||||||
data = yaml.safe_load(stream) or {}
|
|
||||||
services = data.get("services", {})
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Parsing {len(services)} services from compose file {f}"
|
|
||||||
)
|
|
||||||
|
|
||||||
for service_name, service_def in services.items():
|
|
||||||
image = service_def.get("image")
|
|
||||||
is_built = False
|
|
||||||
|
|
||||||
if not image:
|
|
||||||
is_built = True
|
|
||||||
build_ctx = service_def.get("build")
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Service {service_name} is build-based, build config={build_ctx}"
|
|
||||||
)
|
|
||||||
|
|
||||||
if isinstance(build_ctx, dict):
|
|
||||||
context_path = build_ctx.get("context", ".")
|
|
||||||
dockerfile_path = os.path.join(
|
|
||||||
context_path,
|
|
||||||
build_ctx.get("dockerfile", "Dockerfile")
|
|
||||||
)
|
|
||||||
elif isinstance(build_ctx, str):
|
|
||||||
context_path = build_ctx
|
|
||||||
dockerfile_path = os.path.join(
|
|
||||||
context_path, "Dockerfile"
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
dockerfile_path = None
|
|
||||||
|
|
||||||
if dockerfile_path and os.path.exists(dockerfile_path):
|
|
||||||
try:
|
|
||||||
with open(dockerfile_path, "r") as df:
|
|
||||||
for line in df:
|
|
||||||
line = line.strip()
|
|
||||||
if (
|
|
||||||
line.upper().startswith("LABEL")
|
|
||||||
and "image=" in line
|
|
||||||
):
|
|
||||||
m = re.search(
|
|
||||||
r'image=["\']?([^"\']+)["\']?',
|
|
||||||
line
|
|
||||||
)
|
|
||||||
if m:
|
|
||||||
image = m.group(1)
|
|
||||||
logger.debug(
|
|
||||||
f"Found upstream image label for {service_name}: {image}"
|
|
||||||
)
|
|
||||||
break
|
|
||||||
except Exception as e:
|
|
||||||
logger.warning(
|
|
||||||
f"Failed reading Dockerfile {dockerfile_path}: {e}"
|
|
||||||
)
|
|
||||||
|
|
||||||
if not image:
|
|
||||||
image = f"{service_name}:latest"
|
|
||||||
logger.info(
|
|
||||||
f"No image label found for build service {service_name}, "
|
|
||||||
f"defaulting to {image}"
|
|
||||||
)
|
|
||||||
|
|
||||||
service_to_image[service_name] = (image, is_built)
|
|
||||||
|
|
||||||
except Exception as e:
|
|
||||||
logger.error(f"Failed parsing compose file {f}: {e}")
|
|
||||||
|
|
||||||
logger.info(f"Mapped {len(service_to_image)} compose services to images")
|
|
||||||
logger.debug(f"Service/image mapping: {service_to_image}")
|
|
||||||
|
|
||||||
return service_to_image
|
|
||||||
|
|
||||||
def check_containers():
|
|
||||||
logger.info("Starting container update check")
|
|
||||||
|
|
||||||
CONTAINER_UPDATE.clear()
|
|
||||||
|
|
||||||
project_prefix = get_project_prefix_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
compose_files = get_compose_files_from_script(SERVICES_UP_SCRIPT)
|
|
||||||
service_to_image = parse_compose_files(compose_files)
|
|
||||||
|
|
||||||
containers = client.containers.list()
|
|
||||||
logger.info(f"Checking {len(containers)} running containers")
|
|
||||||
|
|
||||||
for container in containers:
|
|
||||||
project_label = container.labels.get("com.docker.compose.project")
|
|
||||||
|
|
||||||
if not project_label:
|
|
||||||
logger.debug(
|
|
||||||
f"Skipping non-compose container {container.name}"
|
|
||||||
)
|
|
||||||
continue
|
|
||||||
|
|
||||||
service_label = container.labels.get("com.docker.compose.service")
|
|
||||||
running_image = container.attrs["Config"]["Image"]
|
|
||||||
|
|
||||||
logger.debug(
|
|
||||||
f"Evaluating container={container.name}, "
|
|
||||||
f"service={service_label}, project={project_label}, "
|
|
||||||
f"running_image={running_image}"
|
|
||||||
)
|
|
||||||
|
|
||||||
compose_image = None
|
|
||||||
is_built = False
|
|
||||||
|
|
||||||
if service_label and service_label in service_to_image:
|
|
||||||
compose_image, is_built = service_to_image[service_label]
|
|
||||||
|
|
||||||
if is_built:
|
|
||||||
compose_image_name, _, _ = compose_image.partition(":")
|
|
||||||
compose_image = f"{project_prefix}{compose_image_name}"
|
|
||||||
|
|
||||||
update_flag = 0
|
|
||||||
|
|
||||||
if is_built:
|
|
||||||
if running_image != compose_image:
|
|
||||||
logger.info(
|
|
||||||
f"Update detected for build-based container {container.name}: "
|
|
||||||
f"running image {running_image} != expected {compose_image}"
|
|
||||||
)
|
|
||||||
update_flag = 1
|
|
||||||
else:
|
|
||||||
local_digest = get_local_digest(running_image)
|
|
||||||
remote_digest = get_remote_digest(
|
|
||||||
service_to_image[service_label][0]
|
|
||||||
)
|
|
||||||
|
|
||||||
if local_digest and remote_digest and local_digest != remote_digest:
|
|
||||||
logger.info(
|
|
||||||
f"Remote image update available for {container.name}: "
|
|
||||||
f"{local_digest} != {remote_digest}"
|
|
||||||
)
|
|
||||||
update_flag = 1
|
|
||||||
else:
|
|
||||||
if running_image != compose_image:
|
|
||||||
logger.info(
|
|
||||||
f"Compose drift detected for {container.name}: "
|
|
||||||
f"running image {running_image} != compose image {compose_image}"
|
|
||||||
)
|
|
||||||
update_flag = 1
|
|
||||||
else:
|
|
||||||
local_digest = get_local_digest(running_image)
|
|
||||||
remote_digest = get_remote_digest(running_image)
|
|
||||||
|
|
||||||
if local_digest and remote_digest and local_digest != remote_digest:
|
|
||||||
logger.info(
|
|
||||||
f"Registry update available for {container.name}: "
|
|
||||||
f"{local_digest} != {remote_digest}"
|
|
||||||
)
|
|
||||||
update_flag = 1
|
|
||||||
|
|
||||||
CONTAINER_UPDATE.labels(
|
|
||||||
container=container.name,
|
|
||||||
compose_image=compose_image if compose_image else "unknown",
|
|
||||||
running_image=running_image,
|
|
||||||
com_docker_compose_project=project_label
|
|
||||||
).set(update_flag)
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
f"Container {container.name}: "
|
|
||||||
f"running={running_image}, "
|
|
||||||
f"compose={compose_image}, "
|
|
||||||
f"update_available={update_flag}"
|
|
||||||
)
|
|
||||||
|
|
||||||
LAST_CHECK.set(time.time())
|
|
||||||
logger.info("Container update check complete")
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
logger.info(
|
|
||||||
f"Starting Docker update exporter on port {EXPORTER_PORT} "
|
|
||||||
f"with LOG_LEVEL={LOG_LEVEL}"
|
|
||||||
)
|
|
||||||
|
|
||||||
start_http_server(EXPORTER_PORT)
|
|
||||||
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
check_containers()
|
|
||||||
except Exception as e:
|
|
||||||
logger.exception(f"Unhandled error during update check: {e}")
|
|
||||||
|
|
||||||
time.sleep(CHECK_INTERVAL)
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
services:
|
services:
|
||||||
docker-socket-proxy:
|
docker-socket-proxy:
|
||||||
profiles: ["monitoring","all","docker-socket-proxy", "core", "traefik", "prometheus"]
|
profiles: ["monitoring","all","docker-socket-proxy", "core", "traefik", "uptime-kuma", "beszel"]
|
||||||
image: tecnativa/docker-socket-proxy:latest
|
image: tecnativa/docker-socket-proxy:latest
|
||||||
container_name: docker-socket-proxy
|
container_name: docker-socket-proxy
|
||||||
hostname: docker-socket-proxy
|
hostname: docker-socket-proxy
|
||||||
@@ -43,5 +43,7 @@ services:
|
|||||||
security_opt:
|
security_opt:
|
||||||
- no-new-privileges:true
|
- no-new-privileges:true
|
||||||
networks:
|
networks:
|
||||||
- monitor
|
# - monitor
|
||||||
- traefik
|
- traefik
|
||||||
|
ports:
|
||||||
|
- 2375:2375
|
||||||
|
|||||||
@@ -1,55 +0,0 @@
|
|||||||
services:
|
|
||||||
grafana:
|
|
||||||
profiles: ["monitoring","all","grafana"]
|
|
||||||
image: grafana/grafana:latest
|
|
||||||
container_name: grafana
|
|
||||||
restart: unless-stopped
|
|
||||||
environment:
|
|
||||||
- GF_SERVER_ROOT_URL=${GRAFANA_ROOT_URL}
|
|
||||||
volumes:
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/grafana/data:/var/lib/grafana
|
|
||||||
- ${DOCKER_VOLUMES}/grafana-data:/var/lib/grafana
|
|
||||||
networks:
|
|
||||||
# - traefik_reverse_proxy
|
|
||||||
# - prometheus_edge
|
|
||||||
- traefik
|
|
||||||
- monitor
|
|
||||||
labels:
|
|
||||||
- "traefik.http.routers.grafana.rule=Host(`grafana.lan.ddnsgeek.com`)"
|
|
||||||
- "traefik.enable=true"
|
|
||||||
- "traefik.http.routers.grafana.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.grafana.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.grafana.tls.options=mtls-private-admin@file"
|
|
||||||
- "io.portainer.accesscontrol.public"
|
|
||||||
- "traefik.http.services.grafana.loadbalancer.server.port=3000"
|
|
||||||
- "traefik.docker.network=core_traefik"
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD", "wget", "--spider", "-q", "http://localhost:3000/api/health"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 10s
|
|
||||||
retries: 3
|
|
||||||
start_period: 30s
|
|
||||||
|
|
||||||
# tempo:
|
|
||||||
# image: grafana/tempo:latest
|
|
||||||
# container_name: tempo
|
|
||||||
# command:
|
|
||||||
# - "-config.file=/etc/tempo/config.yaml"
|
|
||||||
# volumes:
|
|
||||||
# - ./tempo/config.yaml:/etc/tempo/config.yaml
|
|
||||||
# - ./tempo/data:/var/lib/tempo
|
|
||||||
# ports:
|
|
||||||
# - "4317:4317" # OTLP gRPC endpoint for Traefik
|
|
||||||
# - "3200:3200" # optional: HTTP endpoint
|
|
||||||
# networks:
|
|
||||||
# - prometheus_edge
|
|
||||||
|
|
||||||
#networks:
|
|
||||||
# traefik_reverse_proxy:
|
|
||||||
# external: true
|
|
||||||
# prometheus_edge:
|
|
||||||
# external: true
|
|
||||||
|
|
||||||
#volumes:
|
|
||||||
# grafana-data:
|
|
||||||
# external: true
|
|
||||||
@@ -1,50 +0,0 @@
|
|||||||
services:
|
|
||||||
influxdb:
|
|
||||||
profiles: ["monitoring","all","influxdb", "prometheus"]
|
|
||||||
image: influxdb:2
|
|
||||||
container_name: influxdb
|
|
||||||
restart: unless-stopped
|
|
||||||
# env_file:
|
|
||||||
# - ${PROJECT_ROOT}/secrets/stack-secrets.env
|
|
||||||
volumes:
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/influxdb:/var/lib/influxdb2
|
|
||||||
- ${DOCKER_DATABASES}/influxdb-data:/var/lib/influxdb2
|
|
||||||
environment:
|
|
||||||
DOCKER_INFLUXDB_INIT_MODE: ${INFLUXDB_INIT_MODE}
|
|
||||||
DOCKER_INFLUXDB_INIT_USERNAME: ${INFLUXDB_INIT_USERNAME}
|
|
||||||
DOCKER_INFLUXDB_INIT_PASSWORD_FILE: /run/secrets/influxdb_init_password
|
|
||||||
DOCKER_INFLUXDB_INIT_ORG: ${INFLUXDB_INIT_ORG}
|
|
||||||
DOCKER_INFLUXDB_INIT_BUCKET: ${INFLUXDB_INIT_BUCKET}
|
|
||||||
secrets:
|
|
||||||
- influxdb_init_password
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
# - traefik_reverse_proxy
|
|
||||||
- traefik
|
|
||||||
- monitor
|
|
||||||
ports:
|
|
||||||
- 8086:8086
|
|
||||||
labels:
|
|
||||||
- "traefik.http.routers.influxdb.rule=Host(`influxdb.lan.ddnsgeek.com`)"
|
|
||||||
- "traefik.enable=true"
|
|
||||||
- "traefik.http.routers.influxdb.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.influxdb.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.influxdb.tls.options=mtls-private-admin@file"
|
|
||||||
- "io.portainer.accesscontrol.public"
|
|
||||||
- "traefik.http.services.influxdb.loadbalancer.server.port=8086"
|
|
||||||
- "traefik.http.routers.influxdb.middlewares=authelia"
|
|
||||||
- "traefik.docker.network=core_traefik"
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD-SHELL", "curl -f http://localhost:8086/health || exit 1"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 5s
|
|
||||||
retries: 3
|
|
||||||
start_period: 10s
|
|
||||||
|
|
||||||
secrets:
|
|
||||||
influxdb_init_password:
|
|
||||||
file: ${PROJECT_ROOT}/secrets/influxdb_init_password.txt
|
|
||||||
|
|
||||||
#volumes:
|
|
||||||
# influxdb-data:
|
|
||||||
# external: true
|
|
||||||
@@ -1,12 +0,0 @@
|
|||||||
FROM python:3.11-slim
|
|
||||||
|
|
||||||
WORKDIR /app
|
|
||||||
|
|
||||||
COPY requirements.txt .
|
|
||||||
RUN pip install --no-cache-dir -r requirements.txt
|
|
||||||
|
|
||||||
COPY app.py .
|
|
||||||
|
|
||||||
EXPOSE 8080
|
|
||||||
|
|
||||||
CMD ["python", "app.py"]
|
|
||||||
@@ -1,60 +0,0 @@
|
|||||||
# mTLS Bridge Service
|
|
||||||
|
|
||||||
Internal HTTP-to-mTLS bridge for services that cannot present client certificates directly (for example, Grafana webhooks).
|
|
||||||
|
|
||||||
## How it works
|
|
||||||
|
|
||||||
1. Accepts plain HTTP requests inside the Docker network.
|
|
||||||
2. Forwards requests to an upstream base URL.
|
|
||||||
3. Preserves the incoming request path/method/body/query string.
|
|
||||||
4. Presents a client certificate/key pair for mTLS authentication.
|
|
||||||
|
|
||||||
## Environment variables
|
|
||||||
|
|
||||||
- `TARGET_URL` (required): upstream base URL (for example `http://node-red:1880`).
|
|
||||||
- `CLIENT_CERT` (default `/certs/client.crt`): client certificate path.
|
|
||||||
- `CLIENT_KEY` (default `/certs/client.key`): client private key path.
|
|
||||||
- `UPSTREAM_CA_CERT` (optional, alias: `CA_CERT`): CA bundle path to verify upstream TLS. Use `false`/`0`/`no` to disable verification.
|
|
||||||
- `TIMEOUT` (default `5`): request timeout in seconds.
|
|
||||||
- `LOG_LEVEL` (default `INFO`): Python logging level.
|
|
||||||
- `HEALTH_ENDPOINT` (default `/_mtls_bridge/health`): local container health endpoint path.
|
|
||||||
- `ALLOWED_PATHS_FILE` (optional): file path containing one allowed endpoint path per line (for example `/health`). Blank lines and `#` comments are ignored. If unset, all paths are allowed.
|
|
||||||
- `MTLS_BRIDGE_BASIC_AUTH_USERS` (required for Traefik auth): value for `traefik.http.middlewares.*.basicauth.users` (e.g. `user:$$apr1$$...`).
|
|
||||||
- `MTLS_BRIDGE_CORS_ALLOW_ORIGIN` (default `https://grafana.lan.ddnsgeek.com`): origin allowed for browser-based panel actions.
|
|
||||||
|
|
||||||
## Endpoints
|
|
||||||
|
|
||||||
- `GET /_mtls_bridge/health` returns `200 OK` for container health checks.
|
|
||||||
- `/*` proxies requests to `${TARGET_URL}/*` with method/body/headers/query string preserved (subject to optional allow-list checks).
|
|
||||||
|
|
||||||
Examples with `TARGET_URL=http://node-red:1880`:
|
|
||||||
|
|
||||||
- `https://mtls-bridge.../docker-update-lockouts/clear` -> `http://node-red:1880/docker-update-lockouts/clear`
|
|
||||||
- `https://mtls-bridge.../health` -> `http://node-red:1880/health`
|
|
||||||
- `https://mtls-bridge.../uptime-kuma` -> `http://node-red:1880/uptime-kuma`
|
|
||||||
|
|
||||||
## Compose integration
|
|
||||||
|
|
||||||
This repository includes `monitoring/mtls-bridge/docker-compose.yml`:
|
|
||||||
|
|
||||||
- No public port exposure.
|
|
||||||
- Read-only cert mount (`${PROJECT_ROOT}/core/traefik/certs:/certs:ro`).
|
|
||||||
- Joined to internal monitoring/traefik networks.
|
|
||||||
|
|
||||||
## Example test
|
|
||||||
|
|
||||||
```bash
|
|
||||||
curl http://mtls-bridge:8080/_mtls_bridge/health
|
|
||||||
curl -X POST http://mtls-bridge:8080/docker-update-lockouts/clear
|
|
||||||
```
|
|
||||||
|
|
||||||
## Allow-list file example
|
|
||||||
|
|
||||||
```text
|
|
||||||
# one path per line
|
|
||||||
/docker-update-lockouts/clear
|
|
||||||
/health
|
|
||||||
/uptime-kuma
|
|
||||||
```
|
|
||||||
|
|
||||||
When `ALLOWED_PATHS_FILE` is set, any path not listed returns `403 Endpoint not allowed`.
|
|
||||||
@@ -1,208 +0,0 @@
|
|||||||
import logging
|
|
||||||
import os
|
|
||||||
import time
|
|
||||||
from urllib.parse import urljoin
|
|
||||||
|
|
||||||
import requests
|
|
||||||
from flask import Flask, Response, g, request
|
|
||||||
|
|
||||||
app = Flask(__name__)
|
|
||||||
|
|
||||||
logging.basicConfig(
|
|
||||||
level=os.environ.get("LOG_LEVEL", "INFO"),
|
|
||||||
format="%(asctime)s %(levelname)s %(message)s",
|
|
||||||
)
|
|
||||||
logger = logging.getLogger("mtls-bridge")
|
|
||||||
logging.getLogger("werkzeug").setLevel(logging.WARNING)
|
|
||||||
|
|
||||||
# Config via env
|
|
||||||
TARGET_URL = (os.environ.get("TARGET_URL") or "").strip()
|
|
||||||
CLIENT_CERT = os.environ.get("CLIENT_CERT", "/certs/client.crt")
|
|
||||||
CLIENT_KEY = os.environ.get("CLIENT_KEY", "/certs/client.key")
|
|
||||||
UPSTREAM_CA_CERT = os.environ.get("UPSTREAM_CA_CERT", os.environ.get("CA_CERT", "")).strip()
|
|
||||||
TIMEOUT = int(os.environ.get("TIMEOUT", "5"))
|
|
||||||
HEALTH_ENDPOINT = os.environ.get("HEALTH_ENDPOINT", "/_mtls_bridge/health")
|
|
||||||
ALLOWED_PATHS_FILE = (os.environ.get("ALLOWED_PATHS_FILE") or "").strip()
|
|
||||||
|
|
||||||
|
|
||||||
def normalize_path(path: str) -> str:
|
|
||||||
if not path or path == "/":
|
|
||||||
return "/"
|
|
||||||
return f"/{path.lstrip('/')}"
|
|
||||||
|
|
||||||
|
|
||||||
def load_allowed_paths() -> set[str]:
|
|
||||||
if not ALLOWED_PATHS_FILE:
|
|
||||||
return set()
|
|
||||||
|
|
||||||
if not os.path.exists(ALLOWED_PATHS_FILE):
|
|
||||||
logger.warning("ALLOWED_PATHS_FILE does not exist: %s (allow-list disabled)", ALLOWED_PATHS_FILE)
|
|
||||||
return set()
|
|
||||||
|
|
||||||
allowed_paths = set()
|
|
||||||
with open(ALLOWED_PATHS_FILE, encoding="utf-8") as f:
|
|
||||||
for line in f:
|
|
||||||
entry = line.strip()
|
|
||||||
if not entry or entry.startswith("#"):
|
|
||||||
continue
|
|
||||||
allowed_paths.add(normalize_path(entry))
|
|
||||||
|
|
||||||
logger.info("loaded %s allowed path(s) from %s", len(allowed_paths), ALLOWED_PATHS_FILE)
|
|
||||||
return allowed_paths
|
|
||||||
|
|
||||||
|
|
||||||
def get_verify_setting():
|
|
||||||
if not UPSTREAM_CA_CERT:
|
|
||||||
return True
|
|
||||||
|
|
||||||
lowered = UPSTREAM_CA_CERT.lower()
|
|
||||||
if lowered in {"false", "0", "no"}:
|
|
||||||
logger.warning("TLS verification for upstream is disabled via UPSTREAM_CA_CERT=%s", UPSTREAM_CA_CERT)
|
|
||||||
return False
|
|
||||||
|
|
||||||
if not os.path.exists(UPSTREAM_CA_CERT):
|
|
||||||
logger.warning(
|
|
||||||
"Configured UPSTREAM_CA_CERT path does not exist: %s (falling back to system CA bundle)",
|
|
||||||
UPSTREAM_CA_CERT,
|
|
||||||
)
|
|
||||||
return True
|
|
||||||
|
|
||||||
return UPSTREAM_CA_CERT
|
|
||||||
|
|
||||||
|
|
||||||
VERIFY_SETTING = get_verify_setting()
|
|
||||||
ALLOWED_PATHS = load_allowed_paths()
|
|
||||||
|
|
||||||
if TARGET_URL and TARGET_URL.lower().startswith("http://"):
|
|
||||||
logger.warning("TARGET_URL uses http:// (plaintext): %s", TARGET_URL)
|
|
||||||
|
|
||||||
logger.info(
|
|
||||||
"mtls-bridge starting target_url=%s timeout=%ss cert=%s key=%s verify=%s health_endpoint=%s allow_list_file=%s allow_list_entries=%s log_level=%s",
|
|
||||||
TARGET_URL,
|
|
||||||
TIMEOUT,
|
|
||||||
CLIENT_CERT,
|
|
||||||
CLIENT_KEY,
|
|
||||||
VERIFY_SETTING,
|
|
||||||
HEALTH_ENDPOINT,
|
|
||||||
ALLOWED_PATHS_FILE,
|
|
||||||
len(ALLOWED_PATHS),
|
|
||||||
os.environ.get("LOG_LEVEL", "INFO"),
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def build_upstream_url(path: str) -> str:
|
|
||||||
"""Map incoming path directly onto TARGET_URL origin/base path."""
|
|
||||||
if not TARGET_URL:
|
|
||||||
raise ValueError("TARGET_URL is not set")
|
|
||||||
|
|
||||||
normalized_target = TARGET_URL.rstrip("/") + "/"
|
|
||||||
normalized_path = path.lstrip("/")
|
|
||||||
upstream_url = urljoin(normalized_target, normalized_path)
|
|
||||||
|
|
||||||
if request.query_string:
|
|
||||||
upstream_url = f"{upstream_url}?{request.query_string.decode('utf-8', 'ignore')}"
|
|
||||||
|
|
||||||
return upstream_url
|
|
||||||
|
|
||||||
|
|
||||||
def is_path_allowed(request_path: str) -> bool:
|
|
||||||
if not ALLOWED_PATHS:
|
|
||||||
return True
|
|
||||||
return request_path in ALLOWED_PATHS
|
|
||||||
|
|
||||||
|
|
||||||
@app.route(HEALTH_ENDPOINT, methods=["GET"])
|
|
||||||
def health():
|
|
||||||
logger.debug("healthcheck request from %s", request.remote_addr)
|
|
||||||
return "OK", 200
|
|
||||||
|
|
||||||
|
|
||||||
@app.before_request
|
|
||||||
def before_request():
|
|
||||||
g.request_start = time.time()
|
|
||||||
|
|
||||||
|
|
||||||
@app.after_request
|
|
||||||
def after_request(response):
|
|
||||||
elapsed_ms = int((time.time() - g.request_start) * 1000)
|
|
||||||
if request.path != HEALTH_ENDPOINT:
|
|
||||||
logger.info(
|
|
||||||
"request complete method=%s path=%s status=%s elapsed_ms=%s",
|
|
||||||
request.method,
|
|
||||||
request.path,
|
|
||||||
response.status_code,
|
|
||||||
elapsed_ms,
|
|
||||||
)
|
|
||||||
return response
|
|
||||||
|
|
||||||
|
|
||||||
@app.route(
|
|
||||||
"/",
|
|
||||||
defaults={"path": ""},
|
|
||||||
methods=["GET", "POST", "PUT", "DELETE", "PATCH", "OPTIONS", "HEAD"],
|
|
||||||
provide_automatic_options=False,
|
|
||||||
)
|
|
||||||
@app.route(
|
|
||||||
"/<path:path>",
|
|
||||||
methods=["GET", "POST", "PUT", "DELETE", "PATCH", "OPTIONS", "HEAD"],
|
|
||||||
provide_automatic_options=False,
|
|
||||||
)
|
|
||||||
def proxy(path):
|
|
||||||
request_path = normalize_path(path)
|
|
||||||
request_size = len(request.get_data(cache=True))
|
|
||||||
logger.info(
|
|
||||||
"incoming request method=%s path=%s query=%s remote=%s bytes=%s",
|
|
||||||
request.method,
|
|
||||||
request_path,
|
|
||||||
request.query_string.decode("utf-8", "ignore"),
|
|
||||||
request.remote_addr,
|
|
||||||
request_size,
|
|
||||||
)
|
|
||||||
|
|
||||||
if not is_path_allowed(request_path):
|
|
||||||
logger.warning("request blocked by allow-list path=%s", request_path)
|
|
||||||
return Response("Endpoint not allowed", status=403)
|
|
||||||
|
|
||||||
try:
|
|
||||||
upstream_url = build_upstream_url(path)
|
|
||||||
|
|
||||||
headers = {k: v for k, v in request.headers if k.lower() != "host"}
|
|
||||||
headers["X-Forwarded-By"] = "mtls-bridge"
|
|
||||||
|
|
||||||
start_time = time.time()
|
|
||||||
resp = requests.request(
|
|
||||||
method=request.method,
|
|
||||||
url=upstream_url,
|
|
||||||
headers=headers,
|
|
||||||
data=request.get_data(cache=True),
|
|
||||||
cookies=request.cookies,
|
|
||||||
cert=(CLIENT_CERT, CLIENT_KEY),
|
|
||||||
verify=VERIFY_SETTING,
|
|
||||||
timeout=TIMEOUT,
|
|
||||||
allow_redirects=False,
|
|
||||||
)
|
|
||||||
|
|
||||||
elapsed_ms = int((time.time() - start_time) * 1000)
|
|
||||||
logger.info(
|
|
||||||
"upstream response status=%s url=%s elapsed_ms=%s response_bytes=%s",
|
|
||||||
resp.status_code,
|
|
||||||
upstream_url,
|
|
||||||
elapsed_ms,
|
|
||||||
len(resp.content),
|
|
||||||
)
|
|
||||||
|
|
||||||
excluded_headers = {"content-encoding", "content-length", "transfer-encoding", "connection"}
|
|
||||||
response_headers = [(k, v) for k, v in resp.headers.items() if k.lower() not in excluded_headers]
|
|
||||||
|
|
||||||
return Response(resp.content, resp.status_code, response_headers)
|
|
||||||
|
|
||||||
except ValueError as exc:
|
|
||||||
logger.error("proxy request failed: %s", exc)
|
|
||||||
return Response(str(exc), status=500)
|
|
||||||
except Exception as exc: # noqa: BLE001
|
|
||||||
logger.exception("proxy request failed")
|
|
||||||
return Response(str(exc), status=500)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
app.run(host="0.0.0.0", port=8080)
|
|
||||||
@@ -1,49 +0,0 @@
|
|||||||
services:
|
|
||||||
mtls-bridge:
|
|
||||||
profiles: ["monitoring", "all", "mtls-bridge"]
|
|
||||||
build:
|
|
||||||
context: ${PROJECT_ROOT}/monitoring/mtls-bridge
|
|
||||||
container_name: mtls-bridge
|
|
||||||
hostname: mtls-bridge.lan.ddnsgeek.com
|
|
||||||
restart: unless-stopped
|
|
||||||
environment:
|
|
||||||
- TARGET_URL=${MTLS_BRIDGE_TARGET_URL}
|
|
||||||
- CLIENT_CERT=${MTLS_BRIDGE_CLIENT_CERT}
|
|
||||||
- CLIENT_KEY=${MTLS_BRIDGE_CLIENT_KEY}
|
|
||||||
- TIMEOUT=${MTLS_BRIDGE_TIMEOUT}
|
|
||||||
- LOG_LEVEL=${MTLS_BRIDGE_LOG_LEVEL:-INFO}
|
|
||||||
- UPSTREAM_CA_CERT=${MTLS_BRIDGE_UPSTREAM_CA_CERT:-}
|
|
||||||
- ALLOWED_PATHS_FILE=${MTLS_BRIDGE_ALLOWED_PATHS_FILE:-}
|
|
||||||
volumes:
|
|
||||||
- ${PROJECT_ROOT}/core/traefik/certs:/certs:ro
|
|
||||||
labels:
|
|
||||||
- "traefik.http.routers.mtls-bridge.rule=Host(`mtls-bridge.lan.ddnsgeek.com`)"
|
|
||||||
- "traefik.enable=true"
|
|
||||||
- "traefik.http.routers.mtls-bridge.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.mtls-bridge.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.mtls-bridge.middlewares=mtls-bridge-auth,mtls-bridge-cors"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-auth.basicauth.users=${MTLS_BRIDGE_BASIC_AUTH_USERS}"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.rule=Host(`mtls-bridge.lan.ddnsgeek.com`) && Method(`OPTIONS`)"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.middlewares=mtls-bridge-cors"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.priority=100"
|
|
||||||
- "traefik.http.routers.mtls-bridge-preflight.service=mtls-bridge"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-cors.headers.accesscontrolalloworiginlist=${MTLS_BRIDGE_CORS_ALLOW_ORIGIN}"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-cors.headers.accesscontrolallowmethods=GET,POST,PUT,PATCH,DELETE,OPTIONS"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-cors.headers.accesscontrolallowheaders=authorization,content-type,x-grafana-action,x-grafana-device-id"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-cors.headers.accesscontrolallowcredentials=true"
|
|
||||||
- "traefik.http.middlewares.mtls-bridge-cors.headers.addvaryheader=true"
|
|
||||||
- "io.portainer.accesscontrol.public"
|
|
||||||
# - "traefik.http.routers.searxng.middlewares=crowdsec@file,secHeaders@file,error-pages-middleware"
|
|
||||||
- "traefik.http.services.mtls-bridge.loadbalancer.server.port=8080"
|
|
||||||
- "traefik.docker.network=core_traefik"
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8080/_mtls_bridge/health', timeout=3).read()"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 5s
|
|
||||||
retries: 3
|
|
||||||
start_period: 10s
|
|
||||||
networks:
|
|
||||||
- monitor
|
|
||||||
- traefik
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
flask
|
|
||||||
requests
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
services:
|
|
||||||
node-exporter:
|
|
||||||
profiles: ["monitoring","all","node-exporter", "prometheus"]
|
|
||||||
image: prom/node-exporter:latest
|
|
||||||
container_name: node-exporter
|
|
||||||
pid: host
|
|
||||||
volumes:
|
|
||||||
- /proc:/host/proc:ro
|
|
||||||
- /sys:/host/sys:ro
|
|
||||||
- /:/rootfs:ro
|
|
||||||
command:
|
|
||||||
- "--path.procfs=/host/proc"
|
|
||||||
- "--path.sysfs=/host/sys"
|
|
||||||
- "--path.rootfs=/rootfs"
|
|
||||||
restart: unless-stopped
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
- monitor
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD", "wget", "--spider", "-q", "http://localhost:9100/metrics"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 10s
|
|
||||||
retries: 3
|
|
||||||
@@ -1,7 +0,0 @@
|
|||||||
FROM nodered/node-red:latest
|
|
||||||
|
|
||||||
USER root
|
|
||||||
RUN apk add --no-cache docker-cli docker-cli-compose
|
|
||||||
#RUN addgroup -g 131 -S docker && addgroup node-red docker
|
|
||||||
|
|
||||||
USER node-red
|
|
||||||
@@ -1,160 +0,0 @@
|
|||||||
# Node-RED update logging for Grafana
|
|
||||||
|
|
||||||
This guide adds structured update-event logging to your existing Node-RED + Telegraf + Prometheus + Grafana stack without introducing Loki.
|
|
||||||
|
|
||||||
## Goal
|
|
||||||
|
|
||||||
Track and surface (in Grafana) the latest update attempts from Node-RED, including:
|
|
||||||
|
|
||||||
- when an update attempt started,
|
|
||||||
- target container/project,
|
|
||||||
- success/failure,
|
|
||||||
- optional failure reason,
|
|
||||||
- elapsed duration.
|
|
||||||
|
|
||||||
## 1) Add a reusable logger function in Node-RED
|
|
||||||
|
|
||||||
Create a **Function** node named `Build update log event` and use:
|
|
||||||
|
|
||||||
```javascript
|
|
||||||
const nowIso = new Date().toISOString();
|
|
||||||
const startedAt = msg.update_started_at || Date.now();
|
|
||||||
const durationMs = Math.max(0, Date.now() - startedAt);
|
|
||||||
|
|
||||||
const payload = msg.payload || {};
|
|
||||||
const labels = payload.labels || {};
|
|
||||||
|
|
||||||
const status = (msg.update_status || payload.status || "unknown").toString().toLowerCase();
|
|
||||||
const success = status === "success" ? 1 : 0;
|
|
||||||
const failed = status === "failed" ? 1 : 0;
|
|
||||||
|
|
||||||
msg.payload = {
|
|
||||||
ts: nowIso,
|
|
||||||
flow: "docker-updates",
|
|
||||||
event: msg.update_event || "attempt",
|
|
||||||
container: msg.container || labels.container || "unknown",
|
|
||||||
project: labels.com_docker_compose_project || msg.project || "unknown",
|
|
||||||
host: msg.host || "unknown",
|
|
||||||
status,
|
|
||||||
success,
|
|
||||||
failed,
|
|
||||||
duration_ms: durationMs,
|
|
||||||
code: Number.isFinite(Number(payload.code)) ? Number(payload.code) : 0,
|
|
||||||
error: (msg.update_error || payload.error || "").toString().slice(0, 300)
|
|
||||||
};
|
|
||||||
|
|
||||||
// one JSON line per event for file output
|
|
||||||
msg.payload = JSON.stringify(msg.payload);
|
|
||||||
return msg;
|
|
||||||
```
|
|
||||||
|
|
||||||
### Wiring recommendation
|
|
||||||
|
|
||||||
Use the same logger function in these branches:
|
|
||||||
|
|
||||||
- before a pull/update command (`update_status=started`, `update_event=attempt`),
|
|
||||||
- success path (`update_status=success`, `update_event=completed`),
|
|
||||||
- failure path (`update_status=failed`, `update_event=completed`, and include `msg.update_error`).
|
|
||||||
|
|
||||||
Then route each branch into a **File** node configured as:
|
|
||||||
|
|
||||||
- Filename: `/data/update-events.ndjson`
|
|
||||||
- Action: append to file
|
|
||||||
- Add newline: enabled
|
|
||||||
|
|
||||||
## 2) Make update state explicit in existing update flow
|
|
||||||
|
|
||||||
In your current update flow (already present in `flows.json`), add/change **Change** nodes around your shell/docker nodes:
|
|
||||||
|
|
||||||
- At update start:
|
|
||||||
- `msg.update_started_at = $millis()`
|
|
||||||
- `msg.update_status = "started"`
|
|
||||||
- `msg.update_event = "attempt"`
|
|
||||||
- At success:
|
|
||||||
- `msg.update_status = "success"`
|
|
||||||
- `msg.update_event = "completed"`
|
|
||||||
- At failure:
|
|
||||||
- `msg.update_status = "failed"`
|
|
||||||
- `msg.update_event = "completed"`
|
|
||||||
- `msg.update_error = msg.payload.stderr` (or equivalent error field)
|
|
||||||
|
|
||||||
## 3) Let Telegraf ingest Node-RED event logs
|
|
||||||
|
|
||||||
Append this to `monitoring/telegraf/telegraf.conf`:
|
|
||||||
|
|
||||||
```toml
|
|
||||||
[[inputs.tail]]
|
|
||||||
files = ["/var/log/node-red/update-events.ndjson"]
|
|
||||||
from_beginning = false
|
|
||||||
name_override = "node_red_update_event"
|
|
||||||
data_format = "json_v2"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2]]
|
|
||||||
measurement_name = "node_red_update_event"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "flow"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "event"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "container"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "project"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "host"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "status"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "success"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "failed"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "duration_ms"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "code"
|
|
||||||
type = "int"
|
|
||||||
```
|
|
||||||
|
|
||||||
And mount the Node-RED data directory into Telegraf (read-only) in `monitoring/prometheus/docker-compose.yml` under `telegraf.volumes`:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- ${PROJECT_ROOT}/monitoring/node-red/data:/var/log/node-red:ro
|
|
||||||
```
|
|
||||||
|
|
||||||
## 4) Prometheus scrape (already in place)
|
|
||||||
|
|
||||||
No Prometheus scrape change is required as long as it already scrapes Telegraf (`telegraf:9273`).
|
|
||||||
|
|
||||||
## 5) Grafana queries to start with
|
|
||||||
|
|
||||||
Use your Prometheus data source and try:
|
|
||||||
|
|
||||||
- Latest success/failure by container:
|
|
||||||
- `last_over_time(node_red_update_event_success[24h])`
|
|
||||||
- `last_over_time(node_red_update_event_failed[24h])`
|
|
||||||
- Failed updates in the last 24h:
|
|
||||||
- `sum by (container, project) (increase(node_red_update_event_failed[24h]))`
|
|
||||||
- Average update duration in last 24h:
|
|
||||||
- `avg by (container, project) (avg_over_time(node_red_update_event_duration_ms[24h]))`
|
|
||||||
|
|
||||||
Recommended panels:
|
|
||||||
|
|
||||||
- **Table**: container, project, status (last value), duration_ms (last value)
|
|
||||||
- **Time series**: failed count over time
|
|
||||||
- **Stat**: total failed updates in last 24h
|
|
||||||
|
|
||||||
## 6) Validation checklist
|
|
||||||
|
|
||||||
1. Trigger a known update path (including one failure if possible).
|
|
||||||
2. Check Node-RED log file:
|
|
||||||
- `tail -n 20 monitoring/node-red/data/update-events.ndjson`
|
|
||||||
3. Check Telegraf metrics endpoint for `node_red_update_event_` metrics.
|
|
||||||
4. Confirm Grafana panel values match the latest Node-RED run.
|
|
||||||
|
|
||||||
## Optional next step
|
|
||||||
|
|
||||||
If you want searchable raw log text and richer log UX, add Loki + Promtail later. Keep this structured metrics path for high-signal alerting even after adding logs.
|
|
||||||
@@ -1,67 +0,0 @@
|
|||||||
services:
|
|
||||||
node-red:
|
|
||||||
profiles: ["monitoring","all","node-red"]
|
|
||||||
# image: nodered/node-red:latest
|
|
||||||
build:
|
|
||||||
context: ${PROJECT_ROOT}/monitoring/node-red
|
|
||||||
container_name: node-red
|
|
||||||
restart: unless-stopped
|
|
||||||
depends_on:
|
|
||||||
- docker-socket-proxy
|
|
||||||
environment:
|
|
||||||
DOCKER_HOST: ${DOCKER_SOCKET_PROXY_HOST}
|
|
||||||
TZ: ${TZ}
|
|
||||||
PROJECT_ROOT: ${NODE_COMPOSE_ROOT}
|
|
||||||
cap_drop:
|
|
||||||
- ALL
|
|
||||||
security_opt:
|
|
||||||
- no-new-privileges:true
|
|
||||||
# ports:
|
|
||||||
# - "1880:1880"
|
|
||||||
volumes:
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/node-red/data:/data
|
|
||||||
- ${DOCKER_VOLUMES}/node-red-data:/data
|
|
||||||
- ${PROJECT_ROOT}:/compose/docker:ro
|
|
||||||
- /home/nixos/raspi:/compose/raspi:ro
|
|
||||||
# - ${PROJECT_ROOT}:/usr/src/node-red:ro
|
|
||||||
|
|
||||||
# - ${PROJECT_ROOT}/default-environment.env:/usr/src/node-red/default-environment.env:ro
|
|
||||||
# - ${PROJECT_ROOT}/default-network.yml:/usr/src/node-red/default-network.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/core/docker-compose.yml:/usr/src/node-red/core/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/prometheus/docker-compose.yml:/usr/src/node-red/monitoring/prometheus/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/gotify/docker-compose.yml:/usr/src/node-red/monitoring/gotify/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/grafana/docker-compose.yml:/usr/src/node-red/monitoring/grafana/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/portainer/docker-compose.yml:/usr/src/node-red/monitoring/portainer/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/uptime-kuma/docker-compose.yml:/usr/src/node-red/monitoring/uptime-kuma/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/gitea/docker-compose.yml:/usr/src/node-red/apps/gitea/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/gramps/docker-compose.yml:/usr/src/node-red/apps/gramps/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/nextcloud/docker-compose.yml:/usr/src/node-red/apps/nextcloud/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/passbolt/docker-compose.yml:/usr/src/node-red/apps/passbolt/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/searxng/docker-compose.yml:/usr/src/node-red/apps/searxng/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/shift-recorder/docker-compose.yml:/usr/src/node-red/apps/shift-recorder/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/apps/stockfill/docker-compose.yml:/usr/src/node-red/apps/stockfill/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/node-red/docker-compose.yml:/usr/src/node-red/monitoring/node-red/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/core/test/docker-compose.yml:/usr/src/node-red/core/test/docker-compose.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/secrets/stack-secrets.env:/usr/src/node-red/secrets/stack-secrets.env:ro
|
|
||||||
|
|
||||||
# - /run/current-system/sw/bin/docker:/usr/bin/docker:ro
|
|
||||||
# depends_on:
|
|
||||||
# - mosquitto
|
|
||||||
# - influxdb
|
|
||||||
networks:
|
|
||||||
- monitor
|
|
||||||
- traefik
|
|
||||||
labels:
|
|
||||||
- "traefik.enable=true"
|
|
||||||
- "traefik.http.routers.node-red.rule=Host(`node-red.lan.ddnsgeek.com`)"
|
|
||||||
# - "traefik.http.routers.node-red.service=api@internal"
|
|
||||||
- "traefik.http.routers.node-red.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.node-red.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.node-red.tls.options=mtls-private-admin@file"
|
|
||||||
- "traefik.http.routers.node-red.middlewares=authelia"
|
|
||||||
- "io.portainer.accesscontrol.public"
|
|
||||||
- "traefik.docker.network=core_traefik"
|
|
||||||
- "traefik.http.services.node-red.loadbalancer.server.port=1880"
|
|
||||||
#volumes:
|
|
||||||
# node-red-data:
|
|
||||||
# external: true
|
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
services:
|
|
||||||
pihole-exporter:
|
|
||||||
profiles: ["monitoring","all","pihole-exporter", "prometheus"]
|
|
||||||
image: ekofr/pihole-exporter:latest
|
|
||||||
container_name: pihole-exporter
|
|
||||||
# env_file:
|
|
||||||
# - ${PROJECT_ROOT}/secrets/stack-secrets.env
|
|
||||||
environment:
|
|
||||||
PIHOLE_HOSTNAME: ${PIHOLE_HOSTNAME}
|
|
||||||
PIHOLE_PASSWORD: ${PIHOLE_PASSWORD}
|
|
||||||
PORT: ${PIHOLE_EXPORTER_PORT}
|
|
||||||
ports:
|
|
||||||
- "${PIHOLE_EXPORTER_PORT}:${PIHOLE_EXPORTER_PORT}"
|
|
||||||
restart: unless-stopped
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
- monitor
|
|
||||||
@@ -1,56 +0,0 @@
|
|||||||
services:
|
|
||||||
prometheus:
|
|
||||||
profiles: ["monitoring","all","prometheus"]
|
|
||||||
image: prom/prometheus:latest
|
|
||||||
# env_file:
|
|
||||||
# - ${PROJECT_ROOT}/secrets/stack-secrets.env
|
|
||||||
container_name: prometheus
|
|
||||||
depends_on:
|
|
||||||
# - alertmanager
|
|
||||||
- telegraf
|
|
||||||
- influxdb
|
|
||||||
- node-exporter
|
|
||||||
- docker-update-exporter
|
|
||||||
- pihole-exporter
|
|
||||||
command:
|
|
||||||
- "--config.file=/etc/prometheus/prometheus.yml"
|
|
||||||
- "--storage.tsdb.path=/prometheus"
|
|
||||||
- "--storage.tsdb.retention.time=15d"
|
|
||||||
# build:
|
|
||||||
# context: ${PROJECT_ROOT}/monitoring/prometheus
|
|
||||||
volumes:
|
|
||||||
- ${PROJECT_ROOT}/monitoring/prometheus/prometheus.yml:/etc/prometheus/prometheus.yml:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/prometheus/data:/prometheus
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/prometheus/rules:/etc/prometheus/rules:ro
|
|
||||||
- ${DOCKER_VOLUMES}/prometheus/data:/prometheus
|
|
||||||
- ${DOCKER_VOLUMES}/prometheus/rules:/etc/prometheus/rules:ro
|
|
||||||
- ${PROJECT_ROOT}/secrets/prometheus_kuma_basic_auth_password.txt:/run/secrets/prometheus_kuma_basic_auth_password:ro
|
|
||||||
|
|
||||||
restart: unless-stopped
|
|
||||||
labels:
|
|
||||||
- "traefik.http.routers.prometheus.rule=Host(`prometheus.lan.ddnsgeek.com`)"
|
|
||||||
- "traefik.enable=true"
|
|
||||||
- "traefik.http.routers.prometheus.entrypoints=websecure"
|
|
||||||
- "traefik.http.routers.prometheus.tls.certresolver=myresolver"
|
|
||||||
- "traefik.http.routers.prometheus.tls.options=mtls-private-admin@file"
|
|
||||||
- "io.portainer.accesscontrol.public"
|
|
||||||
- "traefik.http.services.prometheus.loadbalancer.server.port=9090"
|
|
||||||
- "traefik.http.routers.prometheus.middlewares=authelia"
|
|
||||||
- "traefik.docker.network=core_traefik"
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
# - traefik_reverse_proxy
|
|
||||||
- traefik
|
|
||||||
- monitor
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD", "wget", "--spider", "-q", "http://localhost:9090/-/healthy"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 10s
|
|
||||||
retries: 3
|
|
||||||
start_period: 30s
|
|
||||||
|
|
||||||
#volumes:
|
|
||||||
# prometheus-data:
|
|
||||||
# external: true
|
|
||||||
# prometheus-rules:
|
|
||||||
# external: true
|
|
||||||
@@ -1,166 +0,0 @@
|
|||||||
global:
|
|
||||||
scrape_interval: 15s
|
|
||||||
evaluation_interval: 15s
|
|
||||||
|
|
||||||
#alerting:
|
|
||||||
# alertmanagers:
|
|
||||||
# - static_configs:
|
|
||||||
# - targets:
|
|
||||||
# - alertmanager:9093
|
|
||||||
|
|
||||||
scrape_configs:
|
|
||||||
|
|
||||||
# Prometheus itself
|
|
||||||
- job_name: "prometheus"
|
|
||||||
static_configs:
|
|
||||||
- targets: ["prometheus:9090"]
|
|
||||||
labels:
|
|
||||||
role: prometheus
|
|
||||||
# =========================
|
|
||||||
# Node Exporters (ALL hosts)
|
|
||||||
# =========================
|
|
||||||
- job_name: "node"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- node-exporter:9100
|
|
||||||
labels:
|
|
||||||
role: docker
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- raspberrypi.tail13f623.ts.net:9100
|
|
||||||
labels:
|
|
||||||
role: raspberrypi
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- pve.sweet.home:9100
|
|
||||||
labels:
|
|
||||||
role: proxmox
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- pbs.sweet.home:9100
|
|
||||||
labels:
|
|
||||||
role: backup
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- pihole:9100
|
|
||||||
labels:
|
|
||||||
role: pihole
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- server:9100
|
|
||||||
labels:
|
|
||||||
role: server
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- nix-cache:9100
|
|
||||||
labels:
|
|
||||||
role: cache
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# Telegraf (Docker metrics)
|
|
||||||
# =========================
|
|
||||||
- job_name: "telegraf"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- telegraf:9273
|
|
||||||
- raspberrypi.tail13f623.ts.net:9273
|
|
||||||
labels:
|
|
||||||
role: docker
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# Traefik (all instances)
|
|
||||||
# =========================
|
|
||||||
- job_name: "traefik"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- traefik.lan.ddnsgeek.com:8080
|
|
||||||
labels:
|
|
||||||
role: docker
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- raspberrypi.tail13f623.ts.net:8080
|
|
||||||
labels:
|
|
||||||
role: raspberrypi
|
|
||||||
|
|
||||||
metric_relabel_configs:
|
|
||||||
- source_labels: [service]
|
|
||||||
regex: '(.+)@.+'
|
|
||||||
target_label: service
|
|
||||||
replacement: '$1'
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# Uptime Kuma (separate due to auth)
|
|
||||||
# =========================
|
|
||||||
- job_name: "kuma"
|
|
||||||
metrics_path: /metrics
|
|
||||||
scrape_interval: 30s
|
|
||||||
|
|
||||||
basic_auth:
|
|
||||||
username: wayne.bennett@live.com
|
|
||||||
password_file: /run/secrets/prometheus_kuma_basic_auth_password
|
|
||||||
# password: '4vjCco?[%{=+,t`):C'
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- monitor-kuma:3001
|
|
||||||
labels:
|
|
||||||
role: docker
|
|
||||||
|
|
||||||
- targets:
|
|
||||||
- kuma.lan.ddnsgeek.com
|
|
||||||
labels:
|
|
||||||
role: raspberrypi
|
|
||||||
tls_config:
|
|
||||||
# ca_file: /prometheus/clients-ca.crt
|
|
||||||
cert_file: /prometheus/office-pc.crt
|
|
||||||
key_file: /prometheus/office-pc.key
|
|
||||||
# server_name: kuma.lan.ddnsgeek.com
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# Proxmox Storage Exporters
|
|
||||||
# =========================
|
|
||||||
- job_name: "proxmox-storage"
|
|
||||||
metrics_path: /metrics
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- pve.sweet.home:9101
|
|
||||||
labels:
|
|
||||||
role: proxmox
|
|
||||||
storage: lvm
|
|
||||||
|
|
||||||
# - targets:
|
|
||||||
# - pbs.sweet.home:9102
|
|
||||||
# labels:
|
|
||||||
# role: backup
|
|
||||||
# storage: datastore
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# Docker Updates Exporter
|
|
||||||
# =========================
|
|
||||||
|
|
||||||
|
|
||||||
- job_name: "container-updates"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- docker-update-exporter:9105
|
|
||||||
labels:
|
|
||||||
role: docker
|
|
||||||
- targets:
|
|
||||||
- raspberrypi.tail13f623.ts.net:9105
|
|
||||||
labels:
|
|
||||||
role: raspberrypi
|
|
||||||
|
|
||||||
# =========================
|
|
||||||
# pihole Exporter
|
|
||||||
# =========================
|
|
||||||
|
|
||||||
|
|
||||||
- job_name: "pihole"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- pihole-exporter:9617
|
|
||||||
labels:
|
|
||||||
role: pihole
|
|
||||||
|
|
||||||
|
|
||||||
#rule_files:
|
|
||||||
# - /etc/prometheus/rules/*.yml
|
|
||||||
@@ -1,192 +0,0 @@
|
|||||||
global:
|
|
||||||
scrape_interval: 15s
|
|
||||||
evaluation_interval: 15s
|
|
||||||
|
|
||||||
alerting:
|
|
||||||
alertmanagers:
|
|
||||||
- static_configs:
|
|
||||||
- targets:
|
|
||||||
- alertmanager:9093
|
|
||||||
|
|
||||||
scrape_configs:
|
|
||||||
- job_name: "prometheus"
|
|
||||||
static_configs:
|
|
||||||
- targets: ["prometheus:9090"]
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "docker-node"
|
|
||||||
static_configs:
|
|
||||||
- targets: ["node-exporter:9100"]
|
|
||||||
labels:
|
|
||||||
type: "virtual"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
# - job_name: "cadvisor"
|
|
||||||
# static_configs:
|
|
||||||
# - targets: ["cadvisor:8080"]
|
|
||||||
# labels:
|
|
||||||
# type: "container"
|
|
||||||
# relabel_configs:
|
|
||||||
# - source_labels: [__address__]
|
|
||||||
# regex: '([^:]+):.*'
|
|
||||||
# target_label: instance
|
|
||||||
- job_name: "raspberrypi-node"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "raspberrypi.tail13f623.ts.net:9100"
|
|
||||||
labels:
|
|
||||||
type: "physical"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "proxmox-node"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "pve.sweet.home:9100"
|
|
||||||
labels:
|
|
||||||
type: "physical"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "proxmox-backup-server"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "pbs.sweet.home:9100"
|
|
||||||
labels:
|
|
||||||
type: "virtual"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "pihole"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "pihole:9100"
|
|
||||||
labels:
|
|
||||||
type: "virtual"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "server"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "server:9100"
|
|
||||||
labels:
|
|
||||||
type: "virtual"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: "nix-cache"
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- "nix-cache:9100"
|
|
||||||
labels:
|
|
||||||
type: "virtual"
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: docker
|
|
||||||
static_configs:
|
|
||||||
- targets: ['telegraf:9273']
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
|
|
||||||
- job_name: 'traefik'
|
|
||||||
static_configs:
|
|
||||||
- targets: ['traefik.lan.ddnsgeek.com:8080'] # replace with your Traefik host:port
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
metric_relabel_configs:
|
|
||||||
- source_labels: [service]
|
|
||||||
regex: '(.+)@.+'
|
|
||||||
target_label: service
|
|
||||||
replacement: '$1'
|
|
||||||
|
|
||||||
- job_name: 'raspi-traefik'
|
|
||||||
static_configs:
|
|
||||||
- targets: ['raspberrypi.tail13f623.ts.net:8080'] # replace with your Traefik host:port
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
metric_relabel_configs:
|
|
||||||
- source_labels: [service]
|
|
||||||
regex: '(.+)@.+'
|
|
||||||
target_label: service
|
|
||||||
replacement: '$1'
|
|
||||||
|
|
||||||
- job_name: 'raspi-kuma'
|
|
||||||
metrics_path: /metrics
|
|
||||||
scrape_interval: 30s
|
|
||||||
|
|
||||||
basic_auth:
|
|
||||||
username: wayne.bennett@live.com
|
|
||||||
password: '4vjCco?[%{=+,t`):C'
|
|
||||||
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- kuma.lan.ddnsgeek.com
|
|
||||||
|
|
||||||
- job_name: 'docker-kuma'
|
|
||||||
metrics_path: /metrics
|
|
||||||
scrape_interval: 30s
|
|
||||||
|
|
||||||
basic_auth:
|
|
||||||
username: wayne.bennett@live.com
|
|
||||||
password: '4vjCco?[%{=+,t`):C'
|
|
||||||
|
|
||||||
static_configs:
|
|
||||||
- targets:
|
|
||||||
- uptime-kuma:3001
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
|
|
||||||
- job_name: 'proxmox LVM storage'
|
|
||||||
static_configs:
|
|
||||||
- targets: ['pve.sweet.home:9101']
|
|
||||||
metrics_path: /metrics
|
|
||||||
scheme: http
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
- job_name: 'proxmox backup storage'
|
|
||||||
static_configs:
|
|
||||||
- targets: ['pbs.sweet.home:9102']
|
|
||||||
metrics_path: /metrics
|
|
||||||
scheme: http
|
|
||||||
relabel_configs:
|
|
||||||
- source_labels: [__address__]
|
|
||||||
regex: '([^:]+):.*'
|
|
||||||
target_label: instance
|
|
||||||
|
|
||||||
|
|
||||||
rule_files:
|
|
||||||
- /etc/prometheus/rules/*.yml
|
|
||||||
|
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
services:
|
|
||||||
telegraf:
|
|
||||||
profiles: ["monitoring","all","telegraf", "prometheus"]
|
|
||||||
image: telegraf:latest
|
|
||||||
container_name: telegraf
|
|
||||||
restart: unless-stopped
|
|
||||||
depends_on:
|
|
||||||
- docker-socket-proxy
|
|
||||||
# cap_drop:
|
|
||||||
# - ALL
|
|
||||||
security_opt:
|
|
||||||
- no-new-privileges:true
|
|
||||||
volumes:
|
|
||||||
- ${PROJECT_ROOT}/monitoring/telegraf/telegraf.conf:/etc/telegraf/telegraf.conf:ro
|
|
||||||
# - ${PROJECT_ROOT}/monitoring/node-red/data:/var/log/node-red:ro
|
|
||||||
|
|
||||||
- ${DOCKER_VOLUMES}/node-red-data:/var/log/node-red:ro
|
|
||||||
networks:
|
|
||||||
# - edge
|
|
||||||
- monitor
|
|
||||||
healthcheck:
|
|
||||||
test: ["CMD-SHELL", "curl -f http://localhost:9273/metrics || exit 1"]
|
|
||||||
interval: 30s
|
|
||||||
timeout: 5s
|
|
||||||
retries: 3
|
|
||||||
start_period: 10s
|
|
||||||
@@ -1,45 +0,0 @@
|
|||||||
[agent]
|
|
||||||
interval = "10s"
|
|
||||||
|
|
||||||
[[inputs.docker]]
|
|
||||||
endpoint = "tcp://docker-socket-proxy:2375"
|
|
||||||
gather_services = false
|
|
||||||
|
|
||||||
[[outputs.prometheus_client]]
|
|
||||||
listen = ":9273"
|
|
||||||
|
|
||||||
# Node-RED update-event logs (structured NDJSON) -> Prometheus metrics for Grafana
|
|
||||||
[[inputs.tail]]
|
|
||||||
files = ["/var/log/node-red/update-events.ndjson"]
|
|
||||||
from_beginning = false
|
|
||||||
name_override = "node_red_update_event"
|
|
||||||
data_format = "json_v2"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2]]
|
|
||||||
measurement_name = "node_red_update_event"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "flow"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "event"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "container"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "project"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "host"
|
|
||||||
[[inputs.tail.json_v2.tag]]
|
|
||||||
path = "status"
|
|
||||||
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "success"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "failed"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "duration_ms"
|
|
||||||
type = "int"
|
|
||||||
[[inputs.tail.json_v2.field]]
|
|
||||||
path = "code"
|
|
||||||
type = "int"
|
|
||||||
@@ -29,7 +29,7 @@ services:
|
|||||||
networks:
|
networks:
|
||||||
# - traefik_reverse_proxy
|
# - traefik_reverse_proxy
|
||||||
- traefik
|
- traefik
|
||||||
- monitor
|
# - monitor
|
||||||
|
|
||||||
#networks:
|
#networks:
|
||||||
# traefik_reverse_proxy:
|
# traefik_reverse_proxy:
|
||||||
|
|||||||
Reference in New Issue
Block a user