Introduction
Ceph is a unified distributed storage system providing block (RBD), file (CephFS), and object (RGW/S3) storage from a single cluster. Ansible automates Ceph deployment using cephadm — bootstrap the cluster, add monitors and OSDs, create pools, configure CephFS, and manage the cluster lifecycle.
Architecture
┌─────────────────────────────────────────────┐
│ Ceph Cluster │
│ │
│ ┌─────┐ ┌─────┐ ┌─────┐ ┌─────┐ │
│ │ MON │ │ MON │ │ MON │ │ MGR │ │
│ └──┬──┘ └──┬──┘ └──┬──┘ └──┬──┘ │
│ │ │ │ │ │
│ ┌──┴──┐ ┌──┴──┐ ┌──┴──┐ ┌──┴──┐ │
│ │ OSD │ │ OSD │ │ OSD │ │ OSD │ │
│ │/dev/ │ │/dev/ │ │/dev/ │ │/dev/ │ │
│ │sdb │ │sdb │ │sdb │ │sdb │ │
│ └─────┘ └─────┘ └─────┘ └─────┘ │
│ │
│ ┌─────────┐ ┌─────────┐ ┌─────────┐ │
│ │ RBD │ │ CephFS │ │ RGW │ │
│ │ (block) │ │ (file) │ │ (S3) │ │
│ └─────────┘ └─────────┘ └─────────┘ │
└─────────────────────────────────────────────┘
Bootstrap Ceph Cluster
---
- name: Bootstrap Ceph cluster with cephadm
hosts: ceph_mon[0]
become: true
vars:
ceph_cluster_network: 10.0.1.0/24
ceph_public_network: 10.0.0.0/24
ceph_release: reef
tasks:
- name: Install cephadm
ansible.builtin.package:
name: cephadm
state: present
- name: Bootstrap Ceph cluster
ansible.builtin.command: >
cephadm bootstrap
--mon-ip {{ ansible_default_ipv4.address }}
--cluster-network {{ ceph_cluster_network }}
--allow-fqdn-hostname
--skip-monitoring-stack
args:
creates: /etc/ceph/ceph.conf
register: bootstrap_result
- name: Display dashboard URL
ansible.builtin.debug:
var: bootstrap_result.stdout_lines
when: bootstrap_result.changed
- name: Copy ceph.conf to all nodes
ansible.builtin.fetch:
src: /etc/ceph/ceph.conf
dest: /tmp/ceph.conf
flat: true
- name: Copy admin keyring
ansible.builtin.fetch:
src: /etc/ceph/ceph.client.admin.keyring
dest: /tmp/ceph.client.admin.keyring
flat: true
Add Hosts to Cluster
---
- name: Add hosts to Ceph cluster
hosts: ceph_all:!ceph_mon[0]
become: true
tasks:
- name: Install cephadm
ansible.builtin.package:
name: cephadm
state: present
- name: Copy SSH key from bootstrap node
ansible.builtin.command: >
ssh-copy-id -f -i /etc/ceph/ceph.pub {{ inventory_hostname }}
delegate_to: "{{ groups['ceph_mon'][0] }}"
- name: Add host to cluster
ansible.builtin.command: >
ceph orch host add {{ inventory_hostname }}
{{ ansible_default_ipv4.address }}
--labels={{ ceph_labels | default('_admin') }}
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
Deploy OSDs
- name: Deploy OSDs on all available devices
ansible.builtin.command: ceph orch apply osd --all-available-devices
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
# Or target specific devices
- name: Deploy OSD on specific device
ansible.builtin.command: >
ceph orch daemon add osd {{ inventory_hostname }}:{{ item }}
loop: "{{ ceph_osd_devices }}"
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
Create Pools
- name: Create Ceph pools
ansible.builtin.command: >
ceph osd pool create {{ item.name }} {{ item.pg_num | default(128) }}
{{ item.pg_num | default(128) }} {{ item.type | default('replicated') }}
loop:
- { name: rbd-pool, pg_num: 128 }
- { name: cephfs-data, pg_num: 128 }
- { name: cephfs-metadata, pg_num: 64 }
- { name: rgw-pool, pg_num: 128 }
delegate_to: "{{ groups['ceph_mon'][0] }}"
register: pool_result
changed_when: "'already exists' not in pool_result.stderr"
failed_when: false
- name: Set pool replication size
ansible.builtin.command: >
ceph osd pool set {{ item }} size 3 --yes-i-really-mean-it
loop: [rbd-pool, cephfs-data, cephfs-metadata]
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
- name: Enable RBD application on pool
ansible.builtin.command: ceph osd pool application enable rbd-pool rbd
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
CephFS Filesystem
- name: Create CephFS
ansible.builtin.command: ceph fs new cephfs cephfs-metadata cephfs-data
delegate_to: "{{ groups['ceph_mon'][0] }}"
register: fs_result
changed_when: "'already exists' not in fs_result.stderr"
failed_when: false
- name: Deploy MDS daemons
ansible.builtin.command: ceph orch apply mds cephfs --placement="3"
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
Mount CephFS on Clients
- name: Mount CephFS on client
hosts: ceph_clients
become: true
tasks:
- name: Install Ceph client
ansible.builtin.package:
name: ceph-common
state: present
- name: Copy ceph.conf
ansible.builtin.copy:
src: /tmp/ceph.conf
dest: /etc/ceph/ceph.conf
mode: '0644'
- name: Mount CephFS
ansible.posix.mount:
path: /mnt/cephfs
src: "{{ groups['ceph_mon'] | join(',') }}:/"
fstype: ceph
opts: "name=admin,secretfile=/etc/ceph/admin.secret,_netdev"
state: mounted
RGW (S3 Gateway)
- name: Deploy Rados Gateway
ansible.builtin.command: >
ceph orch apply rgw mystore --placement="2"
--port={{ rgw_port | default(7480) }}
delegate_to: "{{ groups['ceph_mon'][0] }}"
changed_when: true
- name: Create S3 user
ansible.builtin.command: >
radosgw-admin user create
--uid={{ rgw_user }}
--display-name="{{ rgw_display_name }}"
--access-key={{ rgw_access_key }}
--secret={{ vault_rgw_secret_key }}
delegate_to: "{{ groups['ceph_mon'][0] }}"
register: user_result
changed_when: "'already exists' not in user_result.stderr"
failed_when: false
no_log: true
Health Monitoring
- name: Check cluster health
ansible.builtin.command: ceph health
register: ceph_health
changed_when: false
delegate_to: "{{ groups['ceph_mon'][0] }}"
- name: Get cluster status
ansible.builtin.command: ceph status
register: ceph_status
changed_when: false
delegate_to: "{{ groups['ceph_mon'][0] }}"
- name: Check OSD status
ansible.builtin.command: ceph osd tree
register: osd_tree
changed_when: false
delegate_to: "{{ groups['ceph_mon'][0] }}"
- name: Alert on unhealthy cluster
ansible.builtin.fail:
msg: "Ceph cluster is {{ ceph_health.stdout }}"
when: "'HEALTH_OK' not in ceph_health.stdout"
ignore_errors: true
Troubleshooting
OSD Not Starting
- name: Check OSD logs
ansible.builtin.command: cephadm logs --name osd.{{ osd_id }}
register: osd_logs
changed_when: false
Slow Requests
- name: Check for slow ops
ansible.builtin.command: ceph daemon osd.0 dump_ops_in_flight
register: slow_ops
changed_when: false
Related Articles
Conclusion
Ceph provides unified block, file, and object storage from a single cluster — Ansible automates the entire lifecycle with cephadm: bootstrap, add hosts and OSDs, create pools, deploy CephFS and RGW, and monitor health. Use Ceph for persistent volumes in Kubernetes, shared filesystems, S3-compatible object storage, and disaster recovery. Ansible makes scaling from 3 to 300 nodes the same playbook with different inventory.