diff --git a/alloy/README.md b/alloy/README.md index 29b17e3..63a8fc0 100644 --- a/alloy/README.md +++ b/alloy/README.md @@ -18,7 +18,7 @@ One-file [Grafana Alloy](https://grafana.com/docs/alloy/latest/) setup that ship | System logs (files) | `loki.source.file` | `/var/log/syslog`, `/var/log/messages`, `/var/log/*.log` | | Remote config | `remotecfg` | polls Grafana Fleet Management every 60s | -Filtering follows the upstream Grafana Cloud integration configs: node metrics drop only `node_scrape_collector_*` meta-metrics (everything else ships); cadvisor uses the documented allowlist; logs are unfiltered. +Filtering follows the upstream Grafana Cloud integration configs verbatim — `keep`-lists copied from each integration's **Metrics** section ([Linux Node](https://grafana.com/docs/grafana-cloud/monitor-infrastructure/integrations/integration-reference/integration-linux-node/#metrics), [Docker](https://grafana.com/docs/grafana-cloud/monitor-infrastructure/integrations/integration-reference/integration-docker/#metrics)). Logs are unfiltered. > **Log duplication caveat.** On systems where rsyslog mirrors journald to `/var/log/syslog` (e.g. Debian/Ubuntu defaults), enabling both pipelines double-ships the same lines. If that's the case for your hosts, drop one source — typically the file-based one is redundant on systemd-only stacks. diff --git a/alloy/docker-compose.yml b/alloy/docker-compose.yml index 411b175..f7503f6 100644 --- a/alloy/docker-compose.yml +++ b/alloy/docker-compose.yml @@ -114,15 +114,17 @@ configs: forward_to = [prometheus.relabel.integrations_node_exporter.receiver] } - // Linux-Node integration ships everything node_exporter emits and only drops - // the per-collector scrape meta-metrics (matches the upstream cloud-config). + // Keep only the metrics enumerated under "Metrics" in the upstream + // Grafana Cloud Linux Node integration page (157 metrics, verbatim). + // The recording-rule output `instance:node_num_cpu:sum` is computed + // server-side by Grafana Cloud's ruler — not shipped from here. prometheus.relabel "integrations_node_exporter" { forward_to = [prometheus.remote_write.metrics_service.receiver] rule { source_labels = ["__name__"] - regex = "node_scrape_collector_.+" - action = "drop" + regex = "node_arp_entries|node_boot_time_seconds|node_context_switches_total|node_cpu_seconds_total|node_disk_io_time_seconds_total|node_disk_io_time_weighted_seconds_total|node_disk_read_bytes_total|node_disk_read_time_seconds_total|node_disk_reads_completed_total|node_disk_write_time_seconds_total|node_disk_writes_completed_total|node_disk_written_bytes_total|node_filefd_allocated|node_filefd_maximum|node_filesystem_avail_bytes|node_filesystem_device_error|node_filesystem_files|node_filesystem_files_free|node_filesystem_readonly|node_filesystem_size_bytes|node_intr_total|node_load1|node_load15|node_load5|node_md_disks|node_md_disks_required|node_memory_Active_anon_bytes|node_memory_Active_bytes|node_memory_Active_file_bytes|node_memory_AnonHugePages_bytes|node_memory_AnonPages_bytes|node_memory_Bounce_bytes|node_memory_Buffers_bytes|node_memory_Cached_bytes|node_memory_CommitLimit_bytes|node_memory_Committed_AS_bytes|node_memory_DirectMap1G_bytes|node_memory_DirectMap2M_bytes|node_memory_DirectMap4k_bytes|node_memory_Dirty_bytes|node_memory_HugePages_Free|node_memory_HugePages_Rsvd|node_memory_HugePages_Surp|node_memory_HugePages_Total|node_memory_Hugepagesize_bytes|node_memory_Inactive_anon_bytes|node_memory_Inactive_bytes|node_memory_Inactive_file_bytes|node_memory_Mapped_bytes|node_memory_MemAvailable_bytes|node_memory_MemFree_bytes|node_memory_MemTotal_bytes|node_memory_SReclaimable_bytes|node_memory_SUnreclaim_bytes|node_memory_ShmemHugePages_bytes|node_memory_ShmemPmdMapped_bytes|node_memory_Shmem_bytes|node_memory_Slab_bytes|node_memory_SwapTotal_bytes|node_memory_VmallocChunk_bytes|node_memory_VmallocTotal_bytes|node_memory_VmallocUsed_bytes|node_memory_WritebackTmp_bytes|node_memory_Writeback_bytes|node_netstat_Icmp6_InErrors|node_netstat_Icmp6_InMsgs|node_netstat_Icmp6_OutMsgs|node_netstat_Icmp_InErrors|node_netstat_Icmp_InMsgs|node_netstat_Icmp_OutMsgs|node_netstat_IpExt_InOctets|node_netstat_IpExt_OutOctets|node_netstat_TcpExt_ListenDrops|node_netstat_TcpExt_ListenOverflows|node_netstat_TcpExt_TCPSynRetrans|node_netstat_Tcp_InErrs|node_netstat_Tcp_InSegs|node_netstat_Tcp_OutRsts|node_netstat_Tcp_OutSegs|node_netstat_Tcp_RetransSegs|node_netstat_Udp6_InDatagrams|node_netstat_Udp6_InErrors|node_netstat_Udp6_NoPorts|node_netstat_Udp6_OutDatagrams|node_netstat_Udp6_RcvbufErrors|node_netstat_Udp6_SndbufErrors|node_netstat_UdpLite_InErrors|node_netstat_Udp_InDatagrams|node_netstat_Udp_InErrors|node_netstat_Udp_NoPorts|node_netstat_Udp_OutDatagrams|node_netstat_Udp_RcvbufErrors|node_netstat_Udp_SndbufErrors|node_network_carrier|node_network_info|node_network_mtu_bytes|node_network_receive_bytes_total|node_network_receive_compressed_total|node_network_receive_drop_total|node_network_receive_errs_total|node_network_receive_fifo_total|node_network_receive_multicast_total|node_network_receive_packets_total|node_network_speed_bytes|node_network_transmit_bytes_total|node_network_transmit_compressed_total|node_network_transmit_drop_total|node_network_transmit_errs_total|node_network_transmit_fifo_total|node_network_transmit_multicast_total|node_network_transmit_packets_total|node_network_transmit_queue_length|node_network_up|node_nf_conntrack_entries|node_nf_conntrack_entries_limit|node_os_info|node_procs_running|node_sockstat_FRAG6_inuse|node_sockstat_FRAG_inuse|node_sockstat_RAW6_inuse|node_sockstat_RAW_inuse|node_sockstat_TCP6_inuse|node_sockstat_TCP_alloc|node_sockstat_TCP_inuse|node_sockstat_TCP_mem|node_sockstat_TCP_mem_bytes|node_sockstat_TCP_orphan|node_sockstat_TCP_tw|node_sockstat_UDP6_inuse|node_sockstat_UDPLITE6_inuse|node_sockstat_UDPLITE_inuse|node_sockstat_UDP_inuse|node_sockstat_UDP_mem|node_sockstat_UDP_mem_bytes|node_sockstat_sockets_used|node_softnet_dropped_total|node_softnet_processed_total|node_softnet_times_squeezed_total|node_systemd_service_restart_total|node_systemd_unit_state|node_textfile_scrape_error|node_time_zone_offset_seconds|node_timex_estimated_error_seconds|node_timex_maxerror_seconds|node_timex_offset_seconds|node_timex_sync_status|node_uname_info|node_vmstat_oom_kill|node_vmstat_pgfault|node_vmstat_pgmajfault|node_vmstat_pgpgin|node_vmstat_pgpgout|node_vmstat_pswpin|node_vmstat_pswpout|process_max_fds|process_open_fds|up" + action = "keep" } } diff --git a/alloy/docs/upstream-sources-of-truth.md b/alloy/docs/upstream-sources-of-truth.md index 0f77436..906987c 100644 --- a/alloy/docs/upstream-sources-of-truth.md +++ b/alloy/docs/upstream-sources-of-truth.md @@ -45,11 +45,11 @@ Where official mixin dashboards exist they're tier-2 corroboration: | Block in `docker-compose.yml` | Upstream source | |---|---| | `prometheus.exporter.unix` (collectors, mounts, fs/net excludes) | Linux Node integration page → "Configure Alloy" | -| `prometheus.relabel "integrations_node_exporter"` (drop `node_scrape_collector_*`) | Linux Node integration page → drop rule snippet | +| `prometheus.relabel "integrations_node_exporter"` (`keep` allowlist of 157 metrics) | Linux Node integration page → [Metrics](https://grafana.com/docs/grafana-cloud/monitor-infrastructure/integrations/integration-reference/integration-linux-node/#metrics) section, verbatim | | `loki.source.journal "default"` + relabel rules (`unit`, `boot_id`, `transport`, `level`) | Linux Node integration page → log scraping | | `loki.source.file` for `/var/log/{syslog,messages,*.log}` | Linux Node integration page → file log scraping | | `prometheus.exporter.cadvisor` (`docker_only = true`) | Docker integration page | -| `prometheus.relabel "integrations_cadvisor"` keep allowlist | Docker integration page → metric list | +| `prometheus.relabel "integrations_cadvisor"` (`keep` allowlist of 16 metrics) | Docker integration page → [Metrics](https://grafana.com/docs/grafana-cloud/monitor-infrastructure/integrations/integration-reference/integration-docker/#metrics) section, verbatim | | `discovery.docker` + `loki.source.docker` (job/instance/container/stream) | Docker integration page → log scraping | ## What "follow upstream" means in practice @@ -60,10 +60,12 @@ Where official mixin dashboards exist they're tier-2 corroboration: ## Audit (2026-04-26) -Verified with tier 1 + tier 2 sources only. The Grafana Cloud integration's full dashboard set (the 7 Linux-Node dashboards + 2 Docker dashboards) is **not** publicly hosted; auditing them requires tier 4 access (an authenticated Grafana Cloud API call against your own stack). That step was not performed and is the only known gap. +Both keep-lists in `docker-compose.yml` are copied verbatim from the **Metrics** section of each integration page (tier 1): -- **Linux-Node** — tier 1 specifies `drop "node_scrape_collector_.+"`, tier 3 (`prometheus/node_exporter` mixin) confirms that the metrics that mixin's dashboards reference all pass through the drop rule. Config matches. **No action.** -- **Docker** — tier 1 lists 14 metrics + `up` / `machine_scrape_error` (16 total). Tier 2 (`grafana/jsonnet-libs` `docker.json`) confirms the same 14 panel-referenced metrics. Config matches. **No action.** +- **Linux-Node** — 157 raw metrics (`node_*`, `process_max_fds`, `process_open_fds`, `up`). The list also contains `instance:node_num_cpu:sum`, which is a recording-rule output computed server-side by Grafana Cloud's ruler — it's intentionally **not** in the keep-list because the agent doesn't produce it. +- **Docker** — 16 metrics (`container_*`, `machine_memory_bytes`, `machine_scrape_error`, `up`). + +The Grafana Cloud integration's full dashboard set (the 7 Linux-Node + 2 Docker dashboards) is not publicly hosted. Tier-4 verification (against the live stack via authenticated API) was **not** performed and is the only known gap. ### Re-running the audit