pool: support CRUSH MSR rules for EC pools

Added failureDomains and osdsPerDomain fields to the
ErasureCodedSpec, which maps to the crush-num-failure-domains and
crush-osds-per-failure-domain Ceph EC profile parameters. This enables
creating EC pools that distribute chunks across fewer, larger hosts
without needing one host per data chunk.

Signed-off-by: Prabhala Tara Aasrita <taraprabhala@Prabhalas-MacBook-Pro.local>
(cherry picked from commit 715f38de91)
This commit is contained in:
Prabhala Tara Aasrita
2026-07-28 19:46:04 +00:00
committed by Mergify
parent 3b678364c8
commit ebf90170bc
6 changed files with 336 additions and 7 deletions
+28
View File
@@ -7199,6 +7199,34 @@ k8s.io/apimachinery/pkg/api/resource.Quantity
Value must be a multiple of 4096 (4Ki).</p>
</td>
</tr>
<tr>
<td>
<code>crushNumFailureDomains</code><br/>
<em>
int32
</em>
</td>
<td>
<em>(Optional)</em>
<p>Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.</p>
</td>
</tr>
<tr>
<td>
<code>crushOSDsPerFailureDomain</code><br/>
<em>
int32
</em>
</td>
<td>
<em>(Optional)</em>
<p>Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.</p>
</td>
</tr>
</tbody>
</table>
<h3 id="ceph.rook.io/v1.ExternalSpec">ExternalSpec
@@ -468,6 +468,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -494,6 +510,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -7790,6 +7809,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -7816,6 +7851,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -8000,6 +8038,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -8026,6 +8080,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -12954,6 +13011,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -12980,6 +13053,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -14611,6 +14687,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -14637,6 +14729,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -15704,6 +15799,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -15730,6 +15841,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -15909,6 +16023,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -15935,6 +16065,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
+133
View File
@@ -470,6 +470,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -496,6 +512,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -7785,6 +7804,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -7811,6 +7846,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -7995,6 +8033,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -8021,6 +8075,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -12943,6 +13000,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -12969,6 +13042,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -14600,6 +14676,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -14626,6 +14718,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -15690,6 +15785,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -15716,6 +15827,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
@@ -15895,6 +16009,22 @@ spec:
This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
minimum: 0
type: integer
crushNumFailureDomains:
description: |-
Number of failure domains to use for erasure coded chunk placement.
When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
that distributes chunks across this many failure domains.
format: int32
minimum: 1
type: integer
crushOSDsPerFailureDomain:
description: |-
Number of OSDs allowed per failure domain for erasure coded chunk placement.
When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
that allows up to this many chunks on OSDs within each failure domain.
format: int32
minimum: 1
type: integer
dataChunks:
description: |-
Number of data chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
@@ -15921,6 +16051,9 @@ spec:
- codingChunks
- dataChunks
type: object
x-kubernetes-validations:
- message: crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together
rule: has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)
failureDomain:
description: 'The failure domain: osd/host/(region or zone if available) - technically also any type in the crush map'
type: string
+15
View File
@@ -1434,6 +1434,7 @@ type QuotaSpec struct {
}
// ErasureCodedSpec represents the spec for erasure code in a pool
// +kubebuilder:validation:XValidation:message="crushNumFailureDomains and crushOSDsPerFailureDomain must be specified together",rule="has(self.crushNumFailureDomains) == has(self.crushOSDsPerFailureDomain)"
type ErasureCodedSpec struct {
// Number of coding chunks per object in an erasure coded storage pool (required for erasure-coded pool type).
// This is the number of OSDs that can be lost simultaneously before data cannot be recovered.
@@ -1457,6 +1458,20 @@ type ErasureCodedSpec struct {
// +kubebuilder:validation:Enum={"4Ki","16Ki","64Ki","256Ki","1Mi"}
// +optional
StripeUnit *resource.Quantity `json:"stripeUnit,omitempty"`
// Number of failure domains to use for erasure coded chunk placement.
// When specified along with crushOSDsPerFailureDomain, a CRUSH MSR rule will be created
// that distributes chunks across this many failure domains.
// +kubebuilder:validation:Minimum=1
// +optional
CrushNumFailureDomains int32 `json:"crushNumFailureDomains,omitempty"`
// Number of OSDs allowed per failure domain for erasure coded chunk placement.
// When specified along with crushNumFailureDomains, a CRUSH MSR rule will be created
// that allows up to this many chunks on OSDs within each failure domain.
// +kubebuilder:validation:Minimum=1
// +optional
CrushOSDsPerFailureDomain int32 `json:"crushOSDsPerFailureDomain,omitempty"`
}
// +genclient
@@ -92,6 +92,12 @@ func CreateErasureCodeProfile(context *clusterd.Context, clusterInfo *ClusterInf
if pool.DeviceClass != "" {
profilePairs = append(profilePairs, fmt.Sprintf("crush-device-class=%s", pool.DeviceClass))
}
if pool.ErasureCoded.CrushNumFailureDomains > 0 {
profilePairs = append(profilePairs, fmt.Sprintf("crush-num-failure-domains=%d", pool.ErasureCoded.CrushNumFailureDomains))
}
if pool.ErasureCoded.CrushOSDsPerFailureDomain > 0 {
profilePairs = append(profilePairs, fmt.Sprintf("crush-osds-per-failure-domain=%d", pool.ErasureCoded.CrushOSDsPerFailureDomain))
}
if pool.ErasureCoded.StripeUnit != nil && !pool.ErasureCoded.StripeUnit.IsZero() {
stripeBytes, ok := pool.ErasureCoded.StripeUnit.AsInt64()
if !ok {
@@ -29,27 +29,33 @@ import (
)
func TestCreateProfile(t *testing.T) {
testCreateProfile(t, "", "myroot", "")
testCreateProfile(t, "", "myroot", "", 0, 0)
}
func TestCreateProfileWithFailureDomain(t *testing.T) {
testCreateProfile(t, "osd", "", "")
testCreateProfile(t, "osd", "", "", 0, 0)
}
func TestCreateProfileWithDeviceClass(t *testing.T) {
testCreateProfile(t, "osd", "", "hdd")
testCreateProfile(t, "osd", "", "hdd", 0, 0)
}
func testCreateProfile(t *testing.T, failureDomain, crushRoot, deviceClass string) {
func TestCreateProfileWithMSR(t *testing.T) {
testCreateProfile(t, "host", "", "", 5, 3)
}
func testCreateProfile(t *testing.T, failureDomain, crushRoot, deviceClass string, failureDomains, osdsPerDomain int32) {
stripeUnit := resource.MustParse("4Ki")
spec := cephv1.PoolSpec{
FailureDomain: failureDomain,
CrushRoot: crushRoot,
DeviceClass: deviceClass,
ErasureCoded: cephv1.ErasureCodedSpec{
DataChunks: 2,
CodingChunks: 3,
StripeUnit: &stripeUnit,
DataChunks: 2,
CodingChunks: 3,
StripeUnit: &stripeUnit,
CrushNumFailureDomains: failureDomains,
CrushOSDsPerFailureDomain: osdsPerDomain,
},
}
@@ -82,6 +88,14 @@ func testCreateProfile(t *testing.T, failureDomain, crushRoot, deviceClass strin
assert.Equal(t, fmt.Sprintf("crush-device-class=%s", deviceClass), args[nextArg])
nextArg++
}
if failureDomains > 0 {
assert.Equal(t, fmt.Sprintf("crush-num-failure-domains=%d", failureDomains), args[nextArg])
nextArg++
}
if osdsPerDomain > 0 {
assert.Equal(t, fmt.Sprintf("crush-osds-per-failure-domain=%d", osdsPerDomain), args[nextArg])
nextArg++
}
if spec.ErasureCoded.StripeUnit != nil && !spec.ErasureCoded.StripeUnit.IsZero() {
stripeBytes, ok := spec.ErasureCoded.StripeUnit.AsInt64()
assert.True(t, ok)