98dc2917e8
Auf Basis des Grilling-Sessions mit 20 design decisions umgesetztes v1-Skeleton der Thin-Client-Rolle plus 3 Pilot-Hosts. Architektur: - roles/thin-client/ als geschlossene Rolle (default.nix compose + 7 Subdirectories: hardware, session, identity, network, peripherals, apps, monitoring, deployment) - hosts/AZ-TC-NN/default.nix als ~20-Zeilen-Wrapper pro Fleet-Host - flake.nix instanziiert AZ-TC-01..03 via Fleet-Helper - secrets.nix mit per-host agenix-Secret-Stubs Submodule: - hardware: dell-optiplex-micro + generic-x86_64-uefi Fallback - identity: AD (sssd/krb5/keytab via agenix), lokale Notfalluser (sascha.koenig + jannik.mueller ohne m3ta-home), sudo-Policy - session: KDE Plasma 6 + Wayland + SDDM, Branding (Wallpaper + Footer), PipeWire Audio - network: NetworkManager + wpa_supplicant + 802.1X EAP-TLS, NetBird + SSH via NetBird, systemd-resolved (Corp + NetBird DNS), hardened firewall, OpenSSH - peripherals: CUPS mit Pull-Print-Queue, pam_mount für DFS-Shares - apps: Chromium (ManagedBookmarks, Bitwarden force-install, no local passwords), Office-Web .desktop-Shortcuts, Remmina (mehrere TS, Kerberos SSO), RustDesk Client + Daemon, OBS Studio, Autostart - monitoring: node_exporter → Pushgateway, Alloy (stub für Loki), Snipe-IT Asset-Checkin (stub) - deployment: Disko BTRFS-Layout, auto-upgrade daily + reboot window, snapper snapshots Build-Validierung: 'nix flake check' bestanden für AZ-TC-01/02/03. Siehe roles/thin-client/README.md für den Provisionierungs-Workflow und die Liste der noch auszufüllenden Platzhalter (TODO-Kommentare in den jeweiligen Modulen).
114 lines
3.9 KiB
Nix
114 lines
3.9 KiB
Nix
# roles/thin-client/monitoring/default.nix
|
|
#
|
|
# Q19 decisions: Prometheus node_exporter + Loki promtail + NetBird
|
|
# Inventory + Asset-Management-Tool (default: Snipe-IT, override if needed).
|
|
{config, lib, pkgs, ...}: let
|
|
inherit (lib) mkIf mkOption types;
|
|
cfg = config.az.tc;
|
|
in {
|
|
options.az.tc.monitoring = {
|
|
prometheusPushGateway = mkOption {
|
|
type = types.str;
|
|
default = "pushgateway.az-group.local:9091"; # TODO: real
|
|
description = ''
|
|
Prometheus Pushgateway URL. node_exporter runs locally and pushes
|
|
metrics here (thin clients are usually behind NetBird NAT, so
|
|
pull-scraping isn't reliable).
|
|
'';
|
|
};
|
|
|
|
lokiUrl = mkOption {
|
|
type = types.str;
|
|
default = "http://loki.az-group.local:3100"; # TODO: real
|
|
description = "Loki URL for promtail log shipping.";
|
|
};
|
|
|
|
assetTool = mkOption {
|
|
type = types.enum ["snipe-it" "it-glue" "glpi" "none"];
|
|
default = "snipe-it";
|
|
description = ''
|
|
Asset management tool to integrate with. snipe-it is the
|
|
open-source default. Override if you use a different one.
|
|
'';
|
|
};
|
|
};
|
|
|
|
config = mkIf cfg.enable {
|
|
services.prometheus.exporters.node = {
|
|
enable = true;
|
|
enabledCollectors = ["systemd" "processes" "tcpstat" "wifi"];
|
|
listenAddress = "127.0.0.1";
|
|
port = 9100;
|
|
};
|
|
|
|
# Push metrics to the central Pushgateway every 60s
|
|
systemd.services."push-node-metrics" = {
|
|
description = "Push node_exporter metrics to Pushgateway";
|
|
wantedBy = ["multi-user.target"];
|
|
after = ["network-online.target" "prometheus-node-exporter.service"];
|
|
wants = ["network-online.target"];
|
|
serviceConfig = {
|
|
Type = "simple";
|
|
ExecStart = pkgs.writeShellScript "push-node-metrics" ''
|
|
while true; do
|
|
textfile_collector=$(ls /var/lib/prometheus-node-exporter/textfiles 2>/dev/null)
|
|
${pkgs.curl}/bin/curl -s --data-binary @- \
|
|
http://127.0.0.1:9100/metrics \
|
|
| ${pkgs.curl}/bin/curl -s -X POST \
|
|
--data-binary @- \
|
|
"http://${config.az.tc.monitoring.prometheusPushGateway}/metrics/job/${config.networking.hostName}"
|
|
sleep 60
|
|
done
|
|
'';
|
|
Restart = "always";
|
|
RestartSec = "10s";
|
|
User = "root";
|
|
};
|
|
};
|
|
|
|
# Log shipping via Grafana Alloy (promtail is deprecated/EOL).
|
|
# Q19 decision: Loki + promtail (now: Alloy).
|
|
#
|
|
# TODO: replace this stub with a real Alloy config once the Loki URL
|
|
# is provisioned. See
|
|
# https://grafana.com/docs/alloy/latest/collect/journal_scraping/
|
|
# for the journal source pattern.
|
|
services.alloy = {
|
|
enable = true;
|
|
extraFlags = [
|
|
"--server.http.listen.address=127.0.0.1"
|
|
"--server.http.listen.port=12345"
|
|
];
|
|
};
|
|
|
|
# Stub River config — just keeps the service alive. Real journal
|
|
# shipping TODO once Loki is reachable.
|
|
environment.etc."alloy/config.alloy".text = ''
|
|
// Auto-generated stub — replace with real journal scrape + loki.write
|
|
// block once Loki URL is provisioned.
|
|
logging {
|
|
level = "info"
|
|
format = "logfmt"
|
|
}
|
|
'';
|
|
|
|
# Asset tool integration — snipe-it (default) sends an HTTP POST on
|
|
# boot with hostname + serial + asset tag. Other tools can be added
|
|
# in separate systemd services.
|
|
systemd.services.snipe-it-checkin = mkIf (config.az.tc.monitoring.assetTool == "snipe-it") {
|
|
description = "Report host presence to Snipe-IT asset management";
|
|
wantedBy = ["multi-user.target"];
|
|
after = ["network-online.target"];
|
|
wants = ["network-online.target"];
|
|
serviceConfig = {
|
|
Type = "oneshot";
|
|
ExecStart = pkgs.writeShellScript "snipe-it-checkin" ''
|
|
# TODO: replace with real Snipe-IT API call once configured
|
|
echo "Snipe-IT check-in would happen here for ${config.networking.hostName}"
|
|
'';
|
|
User = "root";
|
|
};
|
|
};
|
|
};
|
|
}
|