+static inline
+struct lov_comp_md_entry_v1 *comp_entry_v1(struct lov_comp_md_v1 *comp, int i)
+{
+ LASSERTF((le32_to_cpu(comp->lcm_magic) & ~LOV_MAGIC_DEFINED) ==
+ LOV_USER_MAGIC_COMP_V1 ||
+ (le32_to_cpu(comp->lcm_magic) & ~LOV_MAGIC_DEFINED) ==
+ LOV_USER_MAGIC_SEL, "Wrong magic %x\n",
+ le32_to_cpu(comp->lcm_magic));
+ LASSERTF(i >= 0 && i < le16_to_cpu(comp->lcm_entry_count),
+ "bad index %d, max = %d\n",
+ i, le16_to_cpu(comp->lcm_entry_count));
+
+ return &comp->lcm_entries[i];
+}
+
+#define for_each_comp_entry_v1(comp, entry) \
+ for (entry = comp_entry_v1(comp, 0); \
+ entry <= comp_entry_v1(comp, \
+ le16_to_cpu(comp->lcm_entry_count) - 1); \
+ entry++)
+
+int lod_erase_dom_stripe(struct lov_comp_md_v1 *comp_v1,
+ struct lov_comp_md_entry_v1 *dom_ent)
+{
+ struct lov_comp_md_entry_v1 *ent;
+ __u16 entries;
+ __u32 dom_off, dom_size, comp_size, off;
+ void *src, *dst;
+ unsigned int size, shift;
+
+ entries = le16_to_cpu(comp_v1->lcm_entry_count) - 1;
+ LASSERT(entries > 0);
+ comp_v1->lcm_entry_count = cpu_to_le16(entries);
+
+ comp_size = le32_to_cpu(comp_v1->lcm_size);
+ dom_off = le32_to_cpu(dom_ent->lcme_offset);
+ dom_size = le32_to_cpu(dom_ent->lcme_size);
+
+ /* all entries offsets are shifted by entry size at least */
+ shift = sizeof(*dom_ent);
+ for_each_comp_entry_v1(comp_v1, ent) {
+ off = le32_to_cpu(ent->lcme_offset);
+ if (off == dom_off) {
+ /* Entry deletion creates two holes in layout data:
+ * - hole in entries array
+ * - hole in layout data at dom_off with dom_size
+ *
+ * First memmove is one entry shift from next entry
+ * start with size up to dom_off in blob
+ */
+ dst = (void *)ent;
+ src = (void *)(ent + 1);
+ size = (unsigned long)((void *)comp_v1 + dom_off - src);
+ memmove(dst, src, size);
+ /* take 'off' from just moved entry */
+ off = le32_to_cpu(ent->lcme_offset);
+ /* second memmove is blob tail after 'off' up to
+ * component end
+ */
+ dst = (void *)comp_v1 + dom_off - sizeof(*ent);
+ src = (void *)comp_v1 + off;
+ size = (unsigned long)(comp_size - off);
+ memmove(dst, src, size);
+ /* all entries offsets after DoM entry are shifted by
+ * dom_size additionally
+ */
+ shift += dom_size;
+ }
+ ent->lcme_offset = cpu_to_le32(off - shift);
+ }
+ comp_v1->lcm_size = cpu_to_le32(comp_size - shift);
+
+ /* notify a caller to re-check entry */
+ return -ERESTART;
+}
+
+void lod_dom_stripesize_recalc(struct lod_device *d)
+{
+ __u64 threshold_mb = d->lod_dom_threshold_free_mb;
+ __u32 max_size = d->lod_dom_stripesize_max_kb;
+ __u32 def_size = d->lod_dom_stripesize_cur_kb;
+
+ /* use maximum allowed value if free space is above threshold */
+ if (d->lod_lsfs_free_mb >= threshold_mb) {
+ def_size = max_size;
+ } else if (!d->lod_lsfs_free_mb || max_size <= LOD_DOM_MIN_SIZE_KB) {
+ def_size = 0;
+ } else {
+ /* recalc threshold like it would be with def_size as max */
+ threshold_mb = mult_frac(threshold_mb, def_size, max_size);
+ if (d->lod_lsfs_free_mb < threshold_mb)
+ def_size = rounddown(def_size / 2, LOD_DOM_MIN_SIZE_KB);
+ else if (d->lod_lsfs_free_mb > threshold_mb * 2)
+ def_size = max_t(unsigned int, def_size * 2,
+ LOD_DOM_MIN_SIZE_KB);
+ }
+
+ if (d->lod_dom_stripesize_cur_kb != def_size) {
+ CDEBUG(D_LAYOUT, "Change default DOM stripe size %d->%d\n",
+ d->lod_dom_stripesize_cur_kb, def_size);
+ d->lod_dom_stripesize_cur_kb = def_size;
+ }
+}
+
+static __u32 lod_dom_stripesize_limit(const struct lu_env *env,
+ struct lod_device *d)
+{
+ int rc;
+
+ /* set bfree as fraction of total space */
+ if (OBD_FAIL_CHECK(OBD_FAIL_MDS_STATFS_SPOOF)) {
+ spin_lock(&d->lod_lsfs_lock);
+ d->lod_lsfs_free_mb = mult_frac(d->lod_lsfs_total_mb,
+ min_t(int, cfs_fail_val, 100), 100);
+ GOTO(recalc, rc = 0);
+ }
+
+ if (d->lod_lsfs_age < ktime_get_seconds() - LOD_DOM_SFS_MAX_AGE) {
+ struct obd_statfs sfs;
+
+ spin_lock(&d->lod_lsfs_lock);
+ if (d->lod_lsfs_age > ktime_get_seconds() - LOD_DOM_SFS_MAX_AGE)
+ GOTO(unlock, rc = 0);
+
+ d->lod_lsfs_age = ktime_get_seconds();
+ spin_unlock(&d->lod_lsfs_lock);
+ rc = dt_statfs(env, d->lod_child, &sfs);
+ if (rc) {
+ CDEBUG(D_LAYOUT,
+ "%s: failed to get OSD statfs: rc = %d\n",
+ lod2obd(d)->obd_name, rc);
+ GOTO(out, rc);
+ }
+ /* udpate local OSD cached statfs data */
+ spin_lock(&d->lod_lsfs_lock);
+ d->lod_lsfs_total_mb = (sfs.os_blocks * sfs.os_bsize) >> 20;
+ d->lod_lsfs_free_mb = (sfs.os_bfree * sfs.os_bsize) >> 20;
+recalc:
+ lod_dom_stripesize_recalc(d);
+unlock:
+ spin_unlock(&d->lod_lsfs_lock);
+ }
+out:
+ return d->lod_dom_stripesize_cur_kb << 10;
+}
+
+int lod_dom_stripesize_choose(const struct lu_env *env, struct lod_device *d,
+ struct lov_comp_md_v1 *comp_v1,
+ struct lov_comp_md_entry_v1 *dom_ent,
+ __u32 stripe_size)
+{
+ struct lov_comp_md_entry_v1 *ent;
+ struct lu_extent *dom_ext, *ext;
+ struct lov_user_md_v1 *lum;
+ __u32 max_stripe_size;
+ __u16 mid, dom_mid;
+ int rc = 0;
+ bool dom_next_entry = false;
+
+ dom_ext = &dom_ent->lcme_extent;
+ dom_mid = mirror_id_of(le32_to_cpu(dom_ent->lcme_id));
+ max_stripe_size = lod_dom_stripesize_limit(env, d);
+
+ /* Check stripe size againts current per-MDT limit */
+ if (stripe_size <= max_stripe_size)
+ return 0;
+
+ lum = (void *)comp_v1 + le32_to_cpu(dom_ent->lcme_offset);
+ CDEBUG(D_LAYOUT, "overwrite DoM component size %u with MDT limit %u\n",
+ stripe_size, max_stripe_size);
+ lum->lmm_stripe_size = cpu_to_le32(max_stripe_size);
+
+ /* In common case the DoM stripe is first entry in a mirror and
+ * can be deleted only if it is not single entry in layout or
+ * mirror, otherwise error should be returned.
+ */
+ for_each_comp_entry_v1(comp_v1, ent) {
+ if (ent == dom_ent)
+ continue;
+
+ mid = mirror_id_of(le32_to_cpu(ent->lcme_id));
+ if (mid != dom_mid)
+ continue;
+
+ ext = &ent->lcme_extent;
+ if (ext->e_start != dom_ext->e_end)
+ continue;
+
+ /* Found next component after the DoM one with the same
+ * mirror_id and adjust its start with DoM component end.
+ *
+ * NOTE: we are considering here that there can be only one
+ * DoM component in a file, all replicas are located on OSTs
+ * always and don't need adjustment since use own layouts.
+ */
+ ext->e_start = cpu_to_le64(max_stripe_size);
+ dom_next_entry = true;
+ break;
+ }
+
+ if (max_stripe_size == 0) {
+ /* DoM component size is zero due to server setting, remove
+ * it from the layout but only if next component exists in
+ * the same mirror. That must be checked prior calling the
+ * lod_erase_dom_stripe().
+ */
+ if (!dom_next_entry)
+ return -EFBIG;
+
+ rc = lod_erase_dom_stripe(comp_v1, dom_ent);
+ } else {
+ /* Update DoM extent end finally */
+ dom_ext->e_end = cpu_to_le64(max_stripe_size);
+ }
+
+ return rc;
+}
+