Introduction

Ceph is a unified distributed storage system providing block (RBD), file (CephFS), and object (RGW/S3) storage from a single cluster. Ansible automates Ceph deployment using cephadm — bootstrap the cluster, add monitors and OSDs, create pools, configure CephFS, and manage the cluster lifecycle.

Architecture

┌─────────────────────────────────────────────┐
│               Ceph Cluster                   │
│                                              │
│  ┌─────┐  ┌─────┐  ┌─────┐  ┌─────┐        │
│  │ MON  │  │ MON  │  │ MON  │  │ MGR  │      │
│  └──┬──┘  └──┬──┘  └──┬──┘  └──┬──┘        │
│     │        │        │        │             │
│  ┌──┴──┐  ┌──┴──┐  ┌──┴──┐  ┌──┴──┐        │
│  │ OSD  │  │ OSD  │  │ OSD  │  │ OSD  │      │
│  │/dev/ │  │/dev/ │  │/dev/ │  │/dev/ │      │
│  │sdb   │  │sdb   │  │sdb   │  │sdb   │      │
│  └─────┘  └─────┘  └─────┘  └─────┘        │
│                                              │
│  ┌─────────┐  ┌─────────┐  ┌─────────┐     │
│  │  RBD     │  │ CephFS   │  │  RGW    │     │
│  │ (block)  │  │ (file)   │  │  (S3)   │     │
│  └─────────┘  └─────────┘  └─────────┘     │
└─────────────────────────────────────────────┘

Bootstrap Ceph Cluster

---
- name: Bootstrap Ceph cluster with cephadm
  hosts: ceph_mon[0]
  become: true
  vars:
    ceph_cluster_network: 10.0.1.0/24
    ceph_public_network: 10.0.0.0/24
    ceph_release: reef
  tasks:
    - name: Install cephadm
      ansible.builtin.package:
        name: cephadm
        state: present

    - name: Bootstrap Ceph cluster
      ansible.builtin.command: >
        cephadm bootstrap
        --mon-ip {{ ansible_default_ipv4.address }}
        --cluster-network {{ ceph_cluster_network }}
        --allow-fqdn-hostname
        --skip-monitoring-stack
      args:
        creates: /etc/ceph/ceph.conf
      register: bootstrap_result

    - name: Display dashboard URL
      ansible.builtin.debug:
        var: bootstrap_result.stdout_lines
      when: bootstrap_result.changed

    - name: Copy ceph.conf to all nodes
      ansible.builtin.fetch:
        src: /etc/ceph/ceph.conf
        dest: /tmp/ceph.conf
        flat: true

    - name: Copy admin keyring
      ansible.builtin.fetch:
        src: /etc/ceph/ceph.client.admin.keyring
        dest: /tmp/ceph.client.admin.keyring
        flat: true

Add Hosts to Cluster

---
- name: Add hosts to Ceph cluster
  hosts: ceph_all:!ceph_mon[0]
  become: true
  tasks:
    - name: Install cephadm
      ansible.builtin.package:
        name: cephadm
        state: present

    - name: Copy SSH key from bootstrap node
      ansible.builtin.command: >
        ssh-copy-id -f -i /etc/ceph/ceph.pub {{ inventory_hostname }}
      delegate_to: "{{ groups['ceph_mon'][0] }}"

    - name: Add host to cluster
      ansible.builtin.command: >
        ceph orch host add {{ inventory_hostname }}
        {{ ansible_default_ipv4.address }}
        --labels={{ ceph_labels | default('_admin') }}
      delegate_to: "{{ groups['ceph_mon'][0] }}"
      changed_when: true

Deploy OSDs

- name: Deploy OSDs on all available devices
  ansible.builtin.command: ceph orch apply osd --all-available-devices
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

# Or target specific devices
- name: Deploy OSD on specific device
  ansible.builtin.command: >
    ceph orch daemon add osd {{ inventory_hostname }}:{{ item }}
  loop: "{{ ceph_osd_devices }}"
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

Create Pools

- name: Create Ceph pools
  ansible.builtin.command: >
    ceph osd pool create {{ item.name }} {{ item.pg_num | default(128) }}
    {{ item.pg_num | default(128) }} {{ item.type | default('replicated') }}
  loop:
    - { name: rbd-pool, pg_num: 128 }
    - { name: cephfs-data, pg_num: 128 }
    - { name: cephfs-metadata, pg_num: 64 }
    - { name: rgw-pool, pg_num: 128 }
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  register: pool_result
  changed_when: "'already exists' not in pool_result.stderr"
  failed_when: false

- name: Set pool replication size
  ansible.builtin.command: >
    ceph osd pool set {{ item }} size 3 --yes-i-really-mean-it
  loop: [rbd-pool, cephfs-data, cephfs-metadata]
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

- name: Enable RBD application on pool
  ansible.builtin.command: ceph osd pool application enable rbd-pool rbd
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

CephFS Filesystem

- name: Create CephFS
  ansible.builtin.command: ceph fs new cephfs cephfs-metadata cephfs-data
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  register: fs_result
  changed_when: "'already exists' not in fs_result.stderr"
  failed_when: false

- name: Deploy MDS daemons
  ansible.builtin.command: ceph orch apply mds cephfs --placement="3"
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

Mount CephFS on Clients

- name: Mount CephFS on client
  hosts: ceph_clients
  become: true
  tasks:
    - name: Install Ceph client
      ansible.builtin.package:
        name: ceph-common
        state: present

    - name: Copy ceph.conf
      ansible.builtin.copy:
        src: /tmp/ceph.conf
        dest: /etc/ceph/ceph.conf
        mode: '0644'

    - name: Mount CephFS
      ansible.posix.mount:
        path: /mnt/cephfs
        src: "{{ groups['ceph_mon'] | join(',') }}:/"
        fstype: ceph
        opts: "name=admin,secretfile=/etc/ceph/admin.secret,_netdev"
        state: mounted

RGW (S3 Gateway)

- name: Deploy Rados Gateway
  ansible.builtin.command: >
    ceph orch apply rgw mystore --placement="2"
    --port={{ rgw_port | default(7480) }}
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  changed_when: true

- name: Create S3 user
  ansible.builtin.command: >
    radosgw-admin user create
    --uid={{ rgw_user }}
    --display-name="{{ rgw_display_name }}"
    --access-key={{ rgw_access_key }}
    --secret={{ vault_rgw_secret_key }}
  delegate_to: "{{ groups['ceph_mon'][0] }}"
  register: user_result
  changed_when: "'already exists' not in user_result.stderr"
  failed_when: false
  no_log: true

Health Monitoring

- name: Check cluster health
  ansible.builtin.command: ceph health
  register: ceph_health
  changed_when: false
  delegate_to: "{{ groups['ceph_mon'][0] }}"

- name: Get cluster status
  ansible.builtin.command: ceph status
  register: ceph_status
  changed_when: false
  delegate_to: "{{ groups['ceph_mon'][0] }}"

- name: Check OSD status
  ansible.builtin.command: ceph osd tree
  register: osd_tree
  changed_when: false
  delegate_to: "{{ groups['ceph_mon'][0] }}"

- name: Alert on unhealthy cluster
  ansible.builtin.fail:
    msg: "Ceph cluster is {{ ceph_health.stdout }}"
  when: "'HEALTH_OK' not in ceph_health.stdout"
  ignore_errors: true

Troubleshooting

OSD Not Starting

- name: Check OSD logs
  ansible.builtin.command: cephadm logs --name osd.{{ osd_id }}
  register: osd_logs
  changed_when: false

Slow Requests

- name: Check for slow ops
  ansible.builtin.command: ceph daemon osd.0 dump_ops_in_flight
  register: slow_ops
  changed_when: false

Conclusion

Ceph provides unified block, file, and object storage from a single cluster — Ansible automates the entire lifecycle with cephadm: bootstrap, add hosts and OSDs, create pools, deploy CephFS and RGW, and monitor health. Use Ceph for persistent volumes in Kubernetes, shared filesystems, S3-compatible object storage, and disaster recovery. Ansible makes scaling from 3 to 300 nodes the same playbook with different inventory.