Introduction
Nagios Core is the industry-standard infrastructure monitoring system — tracks host availability, service health, and performance metrics with alerting. Ansible automates the entire stack: Nagios server installation, NRPE agent deployment on all hosts, host/service configuration from inventory, custom check plugins, and contact/escalation management.
Deploy Nagios Server
---
- name: Deploy Nagios Core server
hosts: nagios_server
become: true
vars:
nagios_admin_password: "{{ vault_nagios_admin_password }}"
nagios_admin_email: ops@example.com
tasks:
- name: Install prerequisites
ansible.builtin.apt:
name:
- apache2
- php
- libapache2-mod-php
- build-essential
- libgd-dev
- unzip
- nagios4
- nagios-plugins-basic
- nagios-plugins-standard
- nagios-nrpe-plugin
state: present
when: ansible_os_family == 'Debian'
- name: Set Nagios admin password
community.general.htpasswd:
path: /etc/nagios4/htpasswd.users
name: nagiosadmin
password: "{{ nagios_admin_password }}"
mode: '0640'
owner: root
group: nagios
- name: Deploy main nagios.cfg
ansible.builtin.template:
src: nagios.cfg.j2
dest: /etc/nagios4/nagios.cfg
mode: '0664'
notify: restart nagios
- name: Create host config directory
ansible.builtin.file:
path: /etc/nagios4/conf.d/hosts
state: directory
mode: '0755'
- name: Deploy host configurations
ansible.builtin.template:
src: host.cfg.j2
dest: "/etc/nagios4/conf.d/hosts/{{ hostvars[item].inventory_hostname }}.cfg"
mode: '0644'
loop: "{{ groups['all'] | difference(groups['nagios_server']) }}"
notify: restart nagios
- name: Deploy contacts
ansible.builtin.template:
src: contacts.cfg.j2
dest: /etc/nagios4/conf.d/contacts.cfg
mode: '0644'
notify: restart nagios
- name: Verify configuration
ansible.builtin.command: nagios4 -v /etc/nagios4/nagios.cfg
register: nagios_verify
changed_when: false
- name: Start Nagios
ansible.builtin.service:
name: nagios4
state: started
enabled: true
handlers:
- name: restart nagios
ansible.builtin.service:
name: nagios4
state: restarted
Host Configuration Template
# templates/host.cfg.j2
define host {
use linux-server
host_name {{ hostvars[item].inventory_hostname }}
alias {{ hostvars[item].nagios_alias | default(hostvars[item].inventory_hostname) }}
address {{ hostvars[item].ansible_host | default(hostvars[item].inventory_hostname) }}
max_check_attempts 5
check_period 24x7
notification_interval 30
notification_period 24x7
contact_groups admins
}
# Standard checks
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description PING
check_command check_ping!200.0,20%!600.0,60%
}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description SSH
check_command check_ssh
}
{% if hostvars[item].nagios_check_nrpe | default(true) %}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description CPU Load
check_command check_nrpe!check_load
}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description Disk Usage
check_command check_nrpe!check_disk
}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description Memory
check_command check_nrpe!check_mem
}
{% endif %}
{% if 'webservers' in hostvars[item].group_names %}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description HTTP
check_command check_http
}
{% endif %}
{% if 'databases' in hostvars[item].group_names %}
define service {
use generic-service
host_name {{ hostvars[item].inventory_hostname }}
service_description PostgreSQL
check_command check_nrpe!check_pgsql
}
{% endif %}
Deploy NRPE Agents
---
- name: Deploy NRPE agent on all monitored hosts
hosts: all:!nagios_server
become: true
vars:
nagios_server_ip: "{{ hostvars[groups['nagios_server'][0]].ansible_host }}"
nrpe_disk_warning: 20
nrpe_disk_critical: 10
nrpe_load_warning: "5.0,4.0,3.0"
nrpe_load_critical: "10.0,6.0,4.0"
tasks:
- name: Install NRPE and plugins
ansible.builtin.package:
name:
- nagios-nrpe-server
- nagios-plugins-basic
- nagios-plugins-standard
state: present
- name: Configure NRPE
ansible.builtin.template:
src: nrpe.cfg.j2
dest: /etc/nagios/nrpe.cfg
mode: '0644'
notify: restart nrpe
- name: Deploy custom check scripts
ansible.builtin.copy:
src: "{{ item }}"
dest: /usr/local/nagios/plugins/
mode: '0755'
loop:
- check_mem.sh
- check_updates.sh
notify: restart nrpe
- name: Allow NRPE through firewall
ansible.posix.firewalld:
port: 5666/tcp
permanent: true
state: enabled
immediate: true
- name: Start NRPE
ansible.builtin.service:
name: nagios-nrpe-server
state: started
enabled: true
handlers:
- name: restart nrpe
ansible.builtin.service:
name: nagios-nrpe-server
state: restarted
NRPE Config Template
# templates/nrpe.cfg.j2
server_port=5666
nrpe_user=nagios
nrpe_group=nagios
allowed_hosts=127.0.0.1,{{ nagios_server_ip }}
dont_blame_nrpe=0
command_timeout=60
# Standard checks
command[check_load]=/usr/lib/nagios/plugins/check_load -w {{ nrpe_load_warning }} -c {{ nrpe_load_critical }}
command[check_disk]=/usr/lib/nagios/plugins/check_disk -w {{ nrpe_disk_warning }}% -c {{ nrpe_disk_critical }}% -p /
command[check_mem]=/usr/local/nagios/plugins/check_mem.sh -w 80 -c 90
command[check_swap]=/usr/lib/nagios/plugins/check_swap -w 50% -c 20%
command[check_procs]=/usr/lib/nagios/plugins/check_procs -w 250 -c 400
command[check_zombie_procs]=/usr/lib/nagios/plugins/check_procs -w 5 -c 10 -s Z
command[check_updates]=/usr/local/nagios/plugins/check_updates.sh
{% if 'webservers' in group_names %}
command[check_nginx]=/usr/lib/nagios/plugins/check_procs -c 1: -C nginx
{% endif %}
{% if 'databases' in group_names %}
command[check_pgsql]=/usr/lib/nagios/plugins/check_pgsql -H localhost
{% endif %}
Contacts and Alerting
# templates/contacts.cfg.j2
define contact {
contact_name admin
use generic-contact
alias Admin
email {{ nagios_admin_email }}
service_notification_period 24x7
host_notification_period 24x7
service_notification_options w,u,c,r
host_notification_options d,u,r
service_notification_commands notify-service-by-email
host_notification_commands notify-host-by-email
}
define contactgroup {
contactgroup_name admins
alias Admins
members admin
}
Monitoring Check
- name: Check Nagios is running
ansible.builtin.uri:
url: "http://{{ inventory_hostname }}/nagios4/"
user: nagiosadmin
password: "{{ vault_nagios_admin_password }}"
force_basic_auth: true
status_code: 200
- name: Verify NRPE connectivity
ansible.builtin.command: "/usr/lib/nagios/plugins/check_nrpe -H {{ item }}"
loop: "{{ groups['all'] | difference(groups['nagios_server']) }}"
delegate_to: "{{ groups['nagios_server'][0] }}"
changed_when: false
Troubleshooting
Config Verification
- name: Validate Nagios config before restart
ansible.builtin.command: nagios4 -v /etc/nagios4/nagios.cfg
register: config_check
changed_when: false
failed_when: "'Error' in config_check.stdout"
NRPE Connection Refused
- name: Test NRPE port
ansible.builtin.wait_for:
host: "{{ item }}"
port: 5666
timeout: 5
loop: "{{ groups['all'] | difference(groups['nagios_server']) }}"
Related Articles
Conclusion
Ansible generates Nagios host and service configurations directly from inventory — adding a server to a group automatically creates the monitoring config. Deploy NRPE agents with group-specific checks (web servers get HTTP checks, databases get PostgreSQL checks), manage contacts and escalations as code, and validate config before every restart. Monitoring infrastructure becomes as reproducible as the infrastructure it monitors.