+static inline
+struct lov_comp_md_entry_v1 *comp_entry_v1(struct lov_comp_md_v1 *comp, int i)
+{
+ LASSERTF((le32_to_cpu(comp->lcm_magic) & ~LOV_MAGIC_DEFINED) ==
+ LOV_USER_MAGIC_COMP_V1, "Wrong magic %x\n",
+ le32_to_cpu(comp->lcm_magic));
+ LASSERTF(i >= 0 && i < le16_to_cpu(comp->lcm_entry_count),
+ "bad index %d, max = %d\n",
+ i, le16_to_cpu(comp->lcm_entry_count));
+
+ return &comp->lcm_entries[i];
+}
+
+#define for_each_comp_entry_v1(comp, entry) \
+ for (entry = comp_entry_v1(comp, 0); \
+ entry <= comp_entry_v1(comp, \
+ le16_to_cpu(comp->lcm_entry_count) - 1); \
+ entry++)
+
+int lod_erase_dom_stripe(struct lov_comp_md_v1 *comp_v1)
+{
+ struct lov_comp_md_entry_v1 *ent, *dom_ent;
+ __u16 entries;
+ __u32 dom_off, dom_size, comp_size;
+ void *blob_src, *blob_dst;
+ unsigned int blob_size, blob_shift;
+
+ entries = le16_to_cpu(comp_v1->lcm_entry_count) - 1;
+ /* if file has only DoM stripe return just error */
+ if (entries == 0)
+ return -EFBIG;
+
+ comp_size = le32_to_cpu(comp_v1->lcm_size);
+ dom_ent = &comp_v1->lcm_entries[0];
+ dom_off = le32_to_cpu(dom_ent->lcme_offset);
+ dom_size = le32_to_cpu(dom_ent->lcme_size);
+
+ /* shift entries array first */
+ comp_v1->lcm_entry_count = cpu_to_le16(entries);
+ memmove(dom_ent, dom_ent + 1,
+ entries * sizeof(struct lov_comp_md_entry_v1));
+
+ /* now move blob of layouts */
+ blob_dst = (void *)comp_v1 + dom_off - sizeof(*dom_ent);
+ blob_src = (void *)comp_v1 + dom_off + dom_size;
+ blob_size = (unsigned long)((void *)comp_v1 + comp_size - blob_src);
+ blob_shift = sizeof(*dom_ent) + dom_size;
+
+ memmove(blob_dst, blob_src, blob_size);
+
+ for_each_comp_entry_v1(comp_v1, ent) {
+ __u32 off;
+
+ off = le32_to_cpu(ent->lcme_offset);
+ ent->lcme_offset = cpu_to_le32(off - blob_shift);
+ }
+
+ comp_v1->lcm_size = cpu_to_le32(comp_size - blob_shift);
+
+ /* notify a caller to re-check entry */
+ return -ERESTART;
+}
+
+int lod_fix_dom_stripe(struct lod_device *d, struct lov_comp_md_v1 *comp_v1)
+{
+ struct lov_comp_md_entry_v1 *ent, *dom_ent;
+ struct lu_extent *dom_ext, *ext;
+ struct lov_user_md_v1 *lum;
+ __u32 stripe_size;
+ __u16 mid, dom_mid;
+ int rc = 0;
+
+ dom_ent = &comp_v1->lcm_entries[0];
+ dom_ext = &dom_ent->lcme_extent;
+ dom_mid = mirror_id_of(le32_to_cpu(dom_ent->lcme_id));
+ stripe_size = d->lod_dom_max_stripesize;
+
+ lum = (void *)comp_v1 + le32_to_cpu(dom_ent->lcme_offset);
+ CDEBUG(D_LAYOUT, "DoM component size %u was bigger than MDT limit %u, "
+ "new size is %u\n", le32_to_cpu(lum->lmm_stripe_size),
+ d->lod_dom_max_stripesize, stripe_size);
+ lum->lmm_stripe_size = cpu_to_le32(stripe_size);
+
+ for_each_comp_entry_v1(comp_v1, ent) {
+ if (ent == dom_ent)
+ continue;
+
+ mid = mirror_id_of(le32_to_cpu(ent->lcme_id));
+ if (mid != dom_mid)
+ continue;
+
+ ext = &ent->lcme_extent;
+ if (ext->e_start != dom_ext->e_end)
+ continue;
+
+ /* Found next component after the DoM one with the same
+ * mirror_id and adjust its start with DoM component end.
+ *
+ * NOTE: we are considering here that there can be only one
+ * DoM component in a file, all replicas are located on OSTs
+ * always and don't need adjustment since use own layouts.
+ */
+ ext->e_start = cpu_to_le64(stripe_size);
+ break;
+ }
+
+ if (stripe_size == 0) {
+ /* DoM component size is zero due to server setting,
+ * remove it from the layout */
+ rc = lod_erase_dom_stripe(comp_v1);
+ } else {
+ /* Update DoM extent end finally */
+ dom_ext->e_end = cpu_to_le64(stripe_size);
+ }
+
+ return rc;
+}
+
+/**
+ * Verify LOV striping.
+ *
+ * \param[in] d LOD device
+ * \param[in] buf buffer with LOV EA to verify
+ * \param[in] is_from_disk 0 - from user, allow some fields to be 0
+ * 1 - from disk, do not allow
+ * \param[in] start extent start for composite layout
+ *
+ * \retval 0 if the striping is valid
+ * \retval -EINVAL if striping is invalid
+ */
+int lod_verify_striping(struct lod_device *d, struct lod_object *lo,
+ const struct lu_buf *buf, bool is_from_disk)
+{
+ struct lov_desc *desc = &d->lod_desc;
+ struct lov_user_md_v1 *lum;
+ struct lov_comp_md_v1 *comp_v1;
+ struct lov_comp_md_entry_v1 *ent;
+ struct lu_extent *ext;
+ struct lu_buf tmp;
+ __u64 prev_end = 0;
+ __u32 stripe_size = 0;
+ __u16 prev_mid = -1, mirror_id = -1;
+ __u32 mirror_count;
+ __u32 magic;
+ int rc = 0;
+ ENTRY;
+
+ if (buf->lb_len < sizeof(lum->lmm_magic)) {
+ CDEBUG(D_LAYOUT, "invalid buf len %zu\n", buf->lb_len);
+ RETURN(-EINVAL);
+ }
+
+ lum = buf->lb_buf;
+
+ magic = le32_to_cpu(lum->lmm_magic) & ~LOV_MAGIC_DEFINED;
+ /* treat foreign LOV EA/object case first
+ * XXX is it expected to try setting again a foreign?
+ * XXX should we care about different current vs new layouts ?
+ */
+ if (unlikely(magic == LOV_USER_MAGIC_FOREIGN)) {
+ struct lov_foreign_md *lfm = buf->lb_buf;
+
+ if (buf->lb_len < offsetof(typeof(*lfm), lfm_value)) {
+ CDEBUG(D_LAYOUT,
+ "buf len %zu < min lov_foreign_md size (%zu)\n",
+ buf->lb_len, offsetof(typeof(*lfm),
+ lfm_value));
+ RETURN(-EINVAL);
+ }
+
+ if (foreign_size_le(lfm) > buf->lb_len) {
+ CDEBUG(D_LAYOUT,
+ "buf len %zu < this lov_foreign_md size (%zu)\n",
+ buf->lb_len, foreign_size_le(lfm));
+ RETURN(-EINVAL);
+ }
+ /* Don't do anything with foreign layouts */
+ RETURN(0);
+ }
+
+ /* normal LOV/layout cases */
+
+ if (buf->lb_len < sizeof(*lum)) {
+ CDEBUG(D_LAYOUT, "buf len %zu too small for lov_user_md\n",
+ buf->lb_len);
+ RETURN(-EINVAL);
+ }
+
+ if (magic != LOV_USER_MAGIC_V1 &&
+ magic != LOV_USER_MAGIC_V3 &&
+ magic != LOV_USER_MAGIC_SPECIFIC &&
+ magic != LOV_USER_MAGIC_COMP_V1) {
+ CDEBUG(D_LAYOUT, "bad userland LOV MAGIC: %#x\n",
+ le32_to_cpu(lum->lmm_magic));
+ RETURN(-EINVAL);
+ }
+
+ if (magic != LOV_USER_MAGIC_COMP_V1)
+ RETURN(lod_verify_v1v3(d, buf, is_from_disk));
+
+ /* magic == LOV_USER_MAGIC_COMP_V1 */
+ comp_v1 = buf->lb_buf;
+ if (buf->lb_len < le32_to_cpu(comp_v1->lcm_size)) {
+ CDEBUG(D_LAYOUT, "buf len %zu is less than %u\n",
+ buf->lb_len, le32_to_cpu(comp_v1->lcm_size));
+ RETURN(-EINVAL);
+ }
+
+recheck:
+ mirror_count = 0;
+ if (le16_to_cpu(comp_v1->lcm_entry_count) == 0) {
+ CDEBUG(D_LAYOUT, "entry count is zero\n");
+ RETURN(-EINVAL);
+ }
+
+ if (S_ISREG(lod2lu_obj(lo)->lo_header->loh_attr) &&
+ lo->ldo_comp_cnt > 0) {
+ /* could be called from lustre.lov.add */
+ __u32 cnt = lo->ldo_comp_cnt;
+
+ ext = &lo->ldo_comp_entries[cnt - 1].llc_extent;
+ prev_end = ext->e_end;
+
+ ++mirror_count;
+ }
+
+ for_each_comp_entry_v1(comp_v1, ent) {
+ ext = &ent->lcme_extent;
+
+ if (le64_to_cpu(ext->e_start) > le64_to_cpu(ext->e_end)) {
+ CDEBUG(D_LAYOUT, "invalid extent "DEXT"\n",
+ le64_to_cpu(ext->e_start),
+ le64_to_cpu(ext->e_end));
+ RETURN(-EINVAL);
+ }
+
+ if (is_from_disk) {
+ /* lcme_id contains valid value */
+ if (le32_to_cpu(ent->lcme_id) == 0 ||
+ le32_to_cpu(ent->lcme_id) > LCME_ID_MAX) {
+ CDEBUG(D_LAYOUT, "invalid id %u\n",
+ le32_to_cpu(ent->lcme_id));
+ RETURN(-EINVAL);
+ }
+
+ if (le16_to_cpu(comp_v1->lcm_mirror_count) > 0) {
+ mirror_id = mirror_id_of(
+ le32_to_cpu(ent->lcme_id));
+
+ /* first component must start with 0 */
+ if (mirror_id != prev_mid &&
+ le64_to_cpu(ext->e_start) != 0) {
+ CDEBUG(D_LAYOUT,
+ "invalid start:%llu, expect:0\n",
+ le64_to_cpu(ext->e_start));
+ RETURN(-EINVAL);
+ }
+
+ prev_mid = mirror_id;
+ }
+ }
+
+ if (le64_to_cpu(ext->e_start) == 0) {
+ ++mirror_count;
+ prev_end = 0;
+ }
+
+ /* the next must be adjacent with the previous one */
+ if (le64_to_cpu(ext->e_start) != prev_end) {
+ CDEBUG(D_LAYOUT,
+ "invalid start actual:%llu, expect:%llu\n",
+ le64_to_cpu(ext->e_start), prev_end);
+ RETURN(-EINVAL);
+ }
+
+ tmp.lb_buf = (char *)comp_v1 + le32_to_cpu(ent->lcme_offset);
+ tmp.lb_len = le32_to_cpu(ent->lcme_size);
+
+ /* Check DoM entry is always the first one */
+ lum = tmp.lb_buf;
+ if (lov_pattern(le32_to_cpu(lum->lmm_pattern)) ==
+ LOV_PATTERN_MDT) {
+ /* DoM component can be only the first stripe */
+ if (le64_to_cpu(ext->e_start) > 0) {
+ CDEBUG(D_LAYOUT, "invalid DoM component "
+ "with %llu extent start\n",
+ le64_to_cpu(ext->e_start));
+ RETURN(-EINVAL);
+ }
+ stripe_size = le32_to_cpu(lum->lmm_stripe_size);
+ /* There is just one stripe on MDT and it must
+ * cover whole component size. */
+ if (stripe_size != le64_to_cpu(ext->e_end)) {
+ CDEBUG(D_LAYOUT, "invalid DoM layout "
+ "stripe size %u != %llu "
+ "(component size)\n",
+ stripe_size, prev_end);
+ RETURN(-EINVAL);
+ }
+ /* Check stripe size againts per-MDT limit */
+ if (stripe_size > d->lod_dom_max_stripesize) {
+ CDEBUG(D_LAYOUT, "DoM component size "
+ "%u is bigger than MDT limit %u, check "
+ "dom_max_stripesize parameter\n",
+ stripe_size, d->lod_dom_max_stripesize);
+ rc = lod_fix_dom_stripe(d, comp_v1);
+ if (rc == -ERESTART) {
+ /* DoM entry was removed, re-check
+ * new layout from start */
+ goto recheck;
+ } else if (rc) {
+ RETURN(rc);
+ }
+ }
+ }
+
+ prev_end = le64_to_cpu(ext->e_end);
+
+ rc = lod_verify_v1v3(d, &tmp, is_from_disk);
+ if (rc)
+ RETURN(rc);
+
+ if (prev_end == LUSTRE_EOF)
+ continue;
+
+ /* extent end must be aligned with the stripe_size */
+ stripe_size = le32_to_cpu(lum->lmm_stripe_size);
+ if (stripe_size == 0)
+ stripe_size = desc->ld_default_stripe_size;
+ if (prev_end % stripe_size) {
+ CDEBUG(D_LAYOUT, "stripe size isn't aligned, "
+ "stripe_sz: %u, [%llu, %llu)\n",
+ stripe_size, ext->e_start, prev_end);
+ RETURN(-EINVAL);
+ }
+ }
+
+ /* make sure that the mirror_count is telling the truth */
+ if (mirror_count != le16_to_cpu(comp_v1->lcm_mirror_count) + 1)
+ RETURN(-EINVAL);
+
+ RETURN(0);
+}
+
+/**
+ * set the default stripe size, if unset.
+ *
+ * \param[in,out] val number of bytes per OST stripe
+ *
+ * The minimum stripe size is 64KB to ensure that a single stripe is an
+ * even multiple of a client PAGE_SIZE (IA64, PPC, etc). Otherwise, it
+ * is difficult to split dirty pages across OSCs during writes.
+ */