# roles/thin-client/monitoring/default.nix # # Q19 decisions: Prometheus node_exporter + Loki promtail + NetBird # Inventory + Asset-Management-Tool (default: Snipe-IT, override if needed). {config, lib, pkgs, ...}: let inherit (lib) mkIf mkOption types; cfg = config.az.tc; in { options.az.tc.monitoring = { prometheusPushGateway = mkOption { type = types.str; default = "pushgateway.az-group.local:9091"; # TODO: real description = '' Prometheus Pushgateway URL. node_exporter runs locally and pushes metrics here (thin clients are usually behind NetBird NAT, so pull-scraping isn't reliable). ''; }; lokiUrl = mkOption { type = types.str; default = "http://loki.az-group.local:3100"; # TODO: real description = "Loki URL for promtail log shipping."; }; assetTool = mkOption { type = types.enum ["snipe-it" "it-glue" "glpi" "none"]; default = "snipe-it"; description = '' Asset management tool to integrate with. snipe-it is the open-source default. Override if you use a different one. ''; }; }; config = mkIf cfg.enable { services.prometheus.exporters.node = { enable = true; enabledCollectors = ["systemd" "processes" "tcpstat" "wifi"]; listenAddress = "127.0.0.1"; port = 9100; }; # Push metrics to the central Pushgateway every 60s systemd.services."push-node-metrics" = { description = "Push node_exporter metrics to Pushgateway"; wantedBy = ["multi-user.target"]; after = ["network-online.target" "prometheus-node-exporter.service"]; wants = ["network-online.target"]; serviceConfig = { Type = "simple"; ExecStart = pkgs.writeShellScript "push-node-metrics" '' while true; do textfile_collector=$(ls /var/lib/prometheus-node-exporter/textfiles 2>/dev/null) ${pkgs.curl}/bin/curl -s --data-binary @- \ http://127.0.0.1:9100/metrics \ | ${pkgs.curl}/bin/curl -s -X POST \ --data-binary @- \ "http://${config.az.tc.monitoring.prometheusPushGateway}/metrics/job/${config.networking.hostName}" sleep 60 done ''; Restart = "always"; RestartSec = "10s"; User = "root"; }; }; # Log shipping via Grafana Alloy (promtail is deprecated/EOL). # Q19 decision: Loki + promtail (now: Alloy). # # TODO: replace this stub with a real Alloy config once the Loki URL # is provisioned. See # https://grafana.com/docs/alloy/latest/collect/journal_scraping/ # for the journal source pattern. services.alloy = { enable = true; extraFlags = [ "--server.http.listen.address=127.0.0.1" "--server.http.listen.port=12345" ]; }; # Stub River config — just keeps the service alive. Real journal # shipping TODO once Loki is reachable. environment.etc."alloy/config.alloy".text = '' // Auto-generated stub — replace with real journal scrape + loki.write // block once Loki URL is provisioned. logging { level = "info" format = "logfmt" } ''; # Asset tool integration — snipe-it (default) sends an HTTP POST on # boot with hostname + serial + asset tag. Other tools can be added # in separate systemd services. systemd.services.snipe-it-checkin = mkIf (config.az.tc.monitoring.assetTool == "snipe-it") { description = "Report host presence to Snipe-IT asset management"; wantedBy = ["multi-user.target"]; after = ["network-online.target"]; wants = ["network-online.target"]; serviceConfig = { Type = "oneshot"; ExecStart = pkgs.writeShellScript "snipe-it-checkin" '' # TODO: replace with real Snipe-IT API call once configured echo "Snipe-IT check-in would happen here for ${config.networking.hostName}" ''; User = "root"; }; }; }; }