Files
AZ-NIX-CLIENTS/roles/thin-client/monitoring/default.nix
T
m3ta-chiron 98dc2917e8 feat: +thin-client role + AZ-TC-01..03 pilot
Auf Basis des Grilling-Sessions mit 20 design decisions umgesetztes
v1-Skeleton der Thin-Client-Rolle plus 3 Pilot-Hosts.

Architektur:
- roles/thin-client/ als geschlossene Rolle (default.nix compose + 7
  Subdirectories: hardware, session, identity, network, peripherals,
  apps, monitoring, deployment)
- hosts/AZ-TC-NN/default.nix als ~20-Zeilen-Wrapper pro Fleet-Host
- flake.nix instanziiert AZ-TC-01..03 via Fleet-Helper
- secrets.nix mit per-host agenix-Secret-Stubs

Submodule:
- hardware: dell-optiplex-micro + generic-x86_64-uefi Fallback
- identity: AD (sssd/krb5/keytab via agenix), lokale Notfalluser
  (sascha.koenig + jannik.mueller ohne m3ta-home), sudo-Policy
- session: KDE Plasma 6 + Wayland + SDDM, Branding (Wallpaper + Footer),
  PipeWire Audio
- network: NetworkManager + wpa_supplicant + 802.1X EAP-TLS, NetBird
  + SSH via NetBird, systemd-resolved (Corp + NetBird DNS), hardened
  firewall, OpenSSH
- peripherals: CUPS mit Pull-Print-Queue, pam_mount für DFS-Shares
- apps: Chromium (ManagedBookmarks, Bitwarden force-install, no local
  passwords), Office-Web .desktop-Shortcuts, Remmina (mehrere TS,
  Kerberos SSO), RustDesk Client + Daemon, OBS Studio, Autostart
- monitoring: node_exporter → Pushgateway, Alloy (stub für Loki),
  Snipe-IT Asset-Checkin (stub)
- deployment: Disko BTRFS-Layout, auto-upgrade daily + reboot window,
  snapper snapshots

Build-Validierung: 'nix flake check' bestanden für AZ-TC-01/02/03.

Siehe roles/thin-client/README.md für den Provisionierungs-Workflow
und die Liste der noch auszufüllenden Platzhalter (TODO-Kommentare
in den jeweiligen Modulen).
2026-07-29 08:34:01 +02:00

114 lines
3.9 KiB
Nix

# roles/thin-client/monitoring/default.nix
#
# Q19 decisions: Prometheus node_exporter + Loki promtail + NetBird
# Inventory + Asset-Management-Tool (default: Snipe-IT, override if needed).
{config, lib, pkgs, ...}: let
inherit (lib) mkIf mkOption types;
cfg = config.az.tc;
in {
options.az.tc.monitoring = {
prometheusPushGateway = mkOption {
type = types.str;
default = "pushgateway.az-group.local:9091"; # TODO: real
description = ''
Prometheus Pushgateway URL. node_exporter runs locally and pushes
metrics here (thin clients are usually behind NetBird NAT, so
pull-scraping isn't reliable).
'';
};
lokiUrl = mkOption {
type = types.str;
default = "http://loki.az-group.local:3100"; # TODO: real
description = "Loki URL for promtail log shipping.";
};
assetTool = mkOption {
type = types.enum ["snipe-it" "it-glue" "glpi" "none"];
default = "snipe-it";
description = ''
Asset management tool to integrate with. snipe-it is the
open-source default. Override if you use a different one.
'';
};
};
config = mkIf cfg.enable {
services.prometheus.exporters.node = {
enable = true;
enabledCollectors = ["systemd" "processes" "tcpstat" "wifi"];
listenAddress = "127.0.0.1";
port = 9100;
};
# Push metrics to the central Pushgateway every 60s
systemd.services."push-node-metrics" = {
description = "Push node_exporter metrics to Pushgateway";
wantedBy = ["multi-user.target"];
after = ["network-online.target" "prometheus-node-exporter.service"];
wants = ["network-online.target"];
serviceConfig = {
Type = "simple";
ExecStart = pkgs.writeShellScript "push-node-metrics" ''
while true; do
textfile_collector=$(ls /var/lib/prometheus-node-exporter/textfiles 2>/dev/null)
${pkgs.curl}/bin/curl -s --data-binary @- \
http://127.0.0.1:9100/metrics \
| ${pkgs.curl}/bin/curl -s -X POST \
--data-binary @- \
"http://${config.az.tc.monitoring.prometheusPushGateway}/metrics/job/${config.networking.hostName}"
sleep 60
done
'';
Restart = "always";
RestartSec = "10s";
User = "root";
};
};
# Log shipping via Grafana Alloy (promtail is deprecated/EOL).
# Q19 decision: Loki + promtail (now: Alloy).
#
# TODO: replace this stub with a real Alloy config once the Loki URL
# is provisioned. See
# https://grafana.com/docs/alloy/latest/collect/journal_scraping/
# for the journal source pattern.
services.alloy = {
enable = true;
extraFlags = [
"--server.http.listen.address=127.0.0.1"
"--server.http.listen.port=12345"
];
};
# Stub River config — just keeps the service alive. Real journal
# shipping TODO once Loki is reachable.
environment.etc."alloy/config.alloy".text = ''
// Auto-generated stub — replace with real journal scrape + loki.write
// block once Loki URL is provisioned.
logging {
level = "info"
format = "logfmt"
}
'';
# Asset tool integration — snipe-it (default) sends an HTTP POST on
# boot with hostname + serial + asset tag. Other tools can be added
# in separate systemd services.
systemd.services.snipe-it-checkin = mkIf (config.az.tc.monitoring.assetTool == "snipe-it") {
description = "Report host presence to Snipe-IT asset management";
wantedBy = ["multi-user.target"];
after = ["network-online.target"];
wants = ["network-online.target"];
serviceConfig = {
Type = "oneshot";
ExecStart = pkgs.writeShellScript "snipe-it-checkin" ''
# TODO: replace with real Snipe-IT API call once configured
echo "Snipe-IT check-in would happen here for ${config.networking.hostName}"
'';
User = "root";
};
};
};
}