1 /* -*- mode: c; c-basic-offset: 8; indent-tabs-mode: nil; -*-
2 * vim:expandtab:shiftwidth=8:tabstop=8:
4 * Copyright (C) 2002, 2003 Cluster File Systems, Inc.
6 * This file is part of Lustre, http://www.lustre.org.
8 * Lustre is free software; you can redistribute it and/or
9 * modify it under the terms of version 2 of the GNU General Public
10 * License as published by the Free Software Foundation.
12 * Lustre is distributed in the hope that it will be useful,
13 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
15 * GNU General Public License for more details.
17 * You should have received a copy of the GNU General Public License
18 * along with Lustre; if not, write to the Free Software
19 * Foundation, Inc., 675 Mass Ave, Cambridge, MA 02139, USA.
23 # define EXPORT_SYMTAB
25 #define DEBUG_SUBSYSTEM S_LMV
27 #include <linux/slab.h>
28 #include <linux/module.h>
29 #include <linux/init.h>
30 #include <linux/slab.h>
31 #include <linux/pagemap.h>
32 #include <asm/div64.h>
33 #include <linux/seq_file.h>
35 #include <liblustre.h>
38 #include <linux/obd_support.h>
39 #include <linux/lustre_lib.h>
40 #include <linux/lustre_net.h>
41 #include <linux/lustre_idl.h>
42 #include <linux/lustre_dlm.h>
43 #include <linux/lustre_mds.h>
44 #include <linux/obd_class.h>
45 #include <linux/obd_ost.h>
46 #include <linux/lprocfs_status.h>
47 #include <linux/lustre_fsfilt.h>
48 #include <linux/obd_lmv.h>
49 #include "lmv_internal.h"
52 static inline void lmv_drop_intent_lock(struct lookup_intent *it)
54 if (it->d.lustre.it_lock_mode != 0)
55 ldlm_lock_decref((void *)&it->d.lustre.it_lock_handle,
56 it->d.lustre.it_lock_mode);
59 int lmv_handle_remote_inode(struct obd_export *exp, struct ll_uctxt *uctxt,
60 void *lmm, int lmmsize,
61 struct lookup_intent *it, int flags,
62 struct ptlrpc_request **reqp,
63 ldlm_blocking_callback cb_blocking)
65 struct obd_device *obd = exp->exp_obd;
66 struct lmv_obd *lmv = &obd->u.lmv;
67 struct mds_body *body = NULL;
71 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
72 LASSERT(body != NULL);
74 if (body->valid & OBD_MD_MDS) {
75 /* oh, MDS reports that this is remote inode case
76 * i.e. we have to ask for real attrs on another MDS */
77 struct ptlrpc_request *req;
79 struct lustre_handle plock;
82 if (it->it_op == IT_LOOKUP) {
83 /* unfortunately, we have to lie to MDC/MDS to
84 * retrieve attributes llite needs */
85 it->it_op = IT_GETATTR;
88 /* we got LOOKUP lock, but we really need attrs */
89 pmode = it->d.lustre.it_lock_mode;
91 memcpy(&plock, &it->d.lustre.it_lock_handle,
93 it->d.lustre.it_lock_mode = 0;
97 it->d.lustre.it_disposition &= ~DISP_ENQ_COMPLETE;
98 rc = md_intent_lock(lmv->tgts[nfid.mds].ltd_exp, uctxt, &nfid,
99 NULL, 0, lmm, lmmsize, NULL, it, flags,
102 /* llite needs LOOKUP lock to track dentry revocation in
103 * order to maintain dcache consistency. thus drop UPDATE
104 * lock here and put LOOKUP in request */
106 lmv_drop_intent_lock(it);
107 memcpy(&it->d.lustre.it_lock_handle, &plock,
109 it->d.lustre.it_lock_mode = pmode;
112 ldlm_lock_decref(&plock, pmode);
114 ptlrpc_req_finished(*reqp);
120 int lmv_intent_open(struct obd_export *exp, struct ll_uctxt *uctxt,
121 struct ll_fid *pfid, const char *name, int len,
122 void *lmm, int lmmsize, struct ll_fid *cfid,
123 struct lookup_intent *it, int flags,
124 struct ptlrpc_request **reqp,
125 ldlm_blocking_callback cb_blocking)
127 struct obd_device *obd = exp->exp_obd;
128 struct lmv_obd *lmv = &obd->u.lmv;
129 struct mds_body *body = NULL;
130 struct ll_fid rpfid = *pfid;
136 /* IT_OPEN is intended to open (and create, possible) an object. Parent
137 * (pfid) may be splitted dir */
141 obj = lmv_grab_obj(obd, &rpfid);
143 /* directory is already splitted, so we have to forward
144 * request to the right MDS */
145 mds = raw_name2idx(obj->objcount, (char *)name, len);
146 CDEBUG(D_OTHER, "forward to MDS #%u\n", mds);
148 rpfid = obj->objs[mds].fid;
152 rc = md_intent_lock(lmv->tgts[mds].ltd_exp, uctxt, &rpfid, name,
153 len, lmm, lmmsize, cfid, it, flags, reqp,
155 if (rc == -ERESTART) {
156 /* directory got splitted. time to update local object
157 * and repeat the request with proper MDS */
158 LASSERT(fid_equal(pfid, &rpfid));
159 rc = lmv_get_mea_and_update_object(exp, &rpfid);
161 ptlrpc_req_finished(*reqp);
168 /* okay, MDS has returned success. Probably name has been resolved in
170 rc = lmv_handle_remote_inode(exp, uctxt, lmm, lmmsize, it,
171 flags, reqp, cb_blocking);
177 /* caller may use attrs MDS returns on IT_OPEN lock request so, we have
178 * to update them for splitted dir */
179 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
180 LASSERT(body != NULL);
183 obj = lmv_grab_obj(obd, cfid);
184 if (!obj && (mea = body_of_splitted_dir(*reqp, 1))) {
185 /* wow! this is splitted dir, we'd like to handle it */
186 obj = lmv_create_obj(exp, &body->fid1, mea);
188 RETURN(PTR_ERR(obj));
192 /* this is splitted dir and we'd want to get attrs */
193 CDEBUG(D_OTHER, "attrs from slaves for %lu/%lu/%lu\n",
194 (unsigned long)cfid->mds, (unsigned long)cfid->id,
195 (unsigned long)cfid->generation);
196 rc = lmv_revalidate_slaves(exp, reqp, cfid,
198 } else if (S_ISDIR(body->mode)) {
199 /*CWARN("hmmm, %lu/%lu/%lu has not lmv obj?!\n",
200 (unsigned long) cfid->mds,
201 (unsigned long) cfid->id,
202 (unsigned long) cfid->generation);*/
211 int lmv_intent_getattr(struct obd_export *exp, struct ll_uctxt *uctxt,
212 struct ll_fid *pfid, const char *name, int len,
213 void *lmm, int lmmsize, struct ll_fid *cfid,
214 struct lookup_intent *it, int flags,
215 struct ptlrpc_request **reqp,
216 ldlm_blocking_callback cb_blocking)
218 struct obd_device *obd = exp->exp_obd;
219 struct lmv_obd *lmv = &obd->u.lmv;
220 struct mds_body *body = NULL;
221 struct ll_fid rpfid = *pfid;
222 struct lmv_obj *obj, *obj2;
228 /* caller wants to revalidate attrs of obj we have to revalidate
229 * slaves if requested object is splitted directory */
230 CDEBUG(D_OTHER, "revalidate attrs for %lu/%lu/%lu\n",
231 (unsigned long)cfid->mds, (unsigned long)cfid->id,
232 (unsigned long)cfid->generation);
234 obj = lmv_grab_obj(obd, cfid);
236 /* in fact, we need not this with current intent_lock(),
237 * but it may change some day */
238 rpfid = obj->objs[mds].fid;
241 rc = md_intent_lock(lmv->tgts[mds].ltd_exp, uctxt, &rpfid, name,
242 len, lmm, lmmsize, cfid, it, flags, reqp,
244 if (obj && rc >= 0) {
245 /* this is splitted dir. In order to optimize things a
246 * bit, we consider obj valid updating missing parts.
248 * FIXME: do we need to return any lock here? It would
249 * be fine if we don't. this means that nobody should
250 * use UPDATE lock to notify about object * removal */
252 "revalidate slaves for %lu/%lu/%lu, rc %d\n",
253 (unsigned long)cfid->mds, (unsigned long)cfid->id,
254 (unsigned long)cfid->generation, rc);
256 rc = lmv_revalidate_slaves(exp, reqp, cfid, it, rc,
263 CDEBUG(D_OTHER, "INTENT getattr for %*s on %lu/%lu/%lu\n",
264 len, name, (unsigned long)pfid->mds, (unsigned long)pfid->id,
265 (unsigned long)pfid->generation);
268 obj = lmv_grab_obj(obd, pfid);
270 /* directory is already splitted. calculate mds */
271 mds = raw_name2idx(obj->objcount, (char *) name, len);
272 rpfid = obj->objs[mds].fid;
275 CDEBUG(D_OTHER, "forward to MDS #%u (slave %lu/%lu/%lu)\n",
276 mds, (unsigned long)rpfid.mds, (unsigned long)rpfid.id,
277 (unsigned long)rpfid.generation);
280 rc = md_intent_lock(lmv->tgts[mds].ltd_exp, uctxt, &rpfid, name,
281 len, lmm, lmmsize, NULL, it, flags, reqp,
289 /* okay, MDS has returned success. probably name has been
290 * resolved in remote inode */
291 rc = lmv_handle_remote_inode(exp, uctxt, lmm, lmmsize, it,
292 flags, reqp, cb_blocking);
296 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
297 LASSERT(body != NULL);
300 obj2 = lmv_grab_obj(obd, cfid);
302 if (!obj2 && (mea = body_of_splitted_dir(*reqp, 1))) {
303 /* wow! this is splitted dir, we'd like to handle it. */
304 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
305 LASSERT(body != NULL);
307 obj2 = lmv_create_obj(exp, &body->fid1, mea);
309 RETURN(PTR_ERR(obj2));
313 /* this is splitted dir and we'd want to get attrs */
314 CDEBUG(D_OTHER, "attrs from slaves for %lu/%lu/%lu, rc %d\n",
315 (unsigned long)cfid->mds, (unsigned long)cfid->id,
316 (unsigned long)cfid->generation, rc);
318 rc = lmv_revalidate_slaves(exp, reqp, cfid, it, 1, cb_blocking);
324 void lmv_update_body_from_obj(struct mds_body *body, struct lmv_inode *obj)
327 body->size += obj->size;
328 /* body->atime = obj->atime;
329 body->ctime = obj->ctime;
330 body->mtime = obj->mtime;
331 body->nlink = obj->nlink;*/
334 int lmv_lookup_slaves(struct obd_export *exp, struct ptlrpc_request **reqp)
336 struct obd_device *obd = exp->exp_obd;
337 struct lmv_obd *lmv = &obd->u.lmv;
338 struct mds_body *body = NULL;
339 struct lustre_handle *lockh;
340 struct ldlm_lock *lock;
341 struct mds_body *body2;
342 struct ll_uctxt uctxt;
350 /* master is locked. we'd like to take locks on slaves and update
351 * attributes to be returned from the slaves it's important that lookup
352 * is called in two cases:
354 * - for first time (dcache has no such a resolving yet).
355 * - ->d_revalidate() returned false.
357 * last case possible only if all the objs (master and all slaves aren't
360 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
361 LASSERT(body != NULL);
363 obj = lmv_grab_obj(obd, &body->fid1);
364 LASSERT(obj != NULL);
366 CDEBUG(D_OTHER, "lookup slaves for %lu/%lu/%lu\n",
367 (unsigned long)body->fid1.mds,
368 (unsigned long)body->fid1.id,
369 (unsigned long)body->fid1.generation);
376 for (i = 0; i < obj->objcount; i++) {
377 struct ll_fid fid = obj->objs[i].fid;
378 struct ptlrpc_request *req = NULL;
379 struct lookup_intent it;
381 if (fid_equal(&fid, &obj->fid))
382 /* skip master obj */
385 CDEBUG(D_OTHER, "lookup slave %lu/%lu/%lu\n",
386 (unsigned long)fid.mds, (unsigned long)fid.id,
387 (unsigned long)fid.generation);
390 memset(&it, 0, sizeof(it));
391 it.it_op = IT_GETATTR;
392 rc = md_intent_lock(lmv->tgts[fid.mds].ltd_exp, &uctxt, &fid,
393 NULL, 0, NULL, 0, &fid, &it, 0, &req,
394 lmv_dirobj_blocking_ast);
396 lockh = (struct lustre_handle *)&it.d.lustre.it_lock_handle;
398 /* nice, this slave is valid */
399 LASSERT(req == NULL);
400 CDEBUG(D_OTHER, "cached\n");
405 /* error during revalidation */
408 /* rc == 0, this means we have no such a lock and can't think
409 * obj is still valid. lookup it again */
410 LASSERT(req == NULL);
413 memset(&it, 0, sizeof(it));
414 it.it_op = IT_GETATTR;
415 rc = md_intent_lock(lmv->tgts[fid.mds].ltd_exp, &uctxt, &fid,
416 NULL, 0, NULL, 0, NULL, &it, 0, &req,
417 lmv_dirobj_blocking_ast);
419 lockh = (struct lustre_handle *) &it.d.lustre.it_lock_handle;
423 /* error during lookup */
426 lock = ldlm_handle2lock(lockh);
429 lock->l_ast_data = lmv_get_obj(obj);
431 body2 = lustre_msg_buf(req->rq_repmsg, 1, sizeof(*body2));
434 obj->objs[i].size = body2->size;
436 CDEBUG(D_OTHER, "fresh: %lu\n",
437 (unsigned long)obj->objs[i].size);
442 ptlrpc_req_finished(req);
444 lmv_update_body_from_obj(body, obj->objs + i);
446 if (it.d.lustre.it_lock_mode)
447 ldlm_lock_decref(lockh, it.d.lustre.it_lock_mode);
455 int lmv_intent_lookup(struct obd_export *exp, struct ll_uctxt *uctxt,
456 struct ll_fid *pfid, const char *name, int len,
457 void *lmm, int lmmsize, struct ll_fid *cfid,
458 struct lookup_intent *it, int flags,
459 struct ptlrpc_request **reqp,
460 ldlm_blocking_callback cb_blocking)
462 struct obd_device *obd = exp->exp_obd;
463 struct lmv_obd *lmv = &obd->u.lmv;
464 struct mds_body *body = NULL;
465 struct ll_fid rpfid = *pfid;
471 /* IT_LOOKUP is intended to produce name -> fid resolving (let's call
472 * this lookup below) or to confirm requested resolving is still valid
473 * (let's call this revalidation) cfid != NULL specifies revalidation */
476 /* this is revalidation: we have to check is LOOKUP lock still
477 * valid for given fid. very important part is that we have to
478 * choose right mds because namespace is per mds */
480 obj = lmv_grab_obj(obd, pfid);
482 mds = raw_name2idx(obj->objcount, (char *) name, len);
483 rpfid = obj->objs[mds].fid;
488 CDEBUG(D_OTHER, "revalidate lookup for %lu/%lu/%lu to %d MDS\n",
489 (unsigned long)cfid->mds, (unsigned long)cfid->id,
490 (unsigned long)cfid->generation, mds);
492 rc = md_intent_lock(lmv->tgts[mds].ltd_exp, uctxt, pfid, name,
493 len, lmm, lmmsize, cfid, it, flags,
500 /* this is lookup. during lookup we have to update all the attributes,
501 * because returned values will be put in struct inode */
503 obj = lmv_grab_obj(obd, pfid);
505 /* directory is already splitted. calculate mds */
506 mds = raw_name2idx(obj->objcount, (char *)name, len);
507 rpfid = obj->objs[mds].fid;
511 rc = md_intent_lock(lmv->tgts[mds].ltd_exp, uctxt, &rpfid, name,
512 len, lmm, lmmsize, NULL, it, flags, reqp,
515 /* very interesting. it seems object is still valid but for some
516 * reason llite calls lookup, not revalidate */
517 CWARN("lookup for %lu/%lu/%lu and data should be uptodate\n",
518 (unsigned long)rpfid.mds, (unsigned long)rpfid.id,
519 (unsigned long)rpfid.generation);
521 LASSERT(*reqp == NULL);
525 if (rc == 0 && *reqp == NULL) {
526 /* once again, we're asked for lookup, not revalidate */
527 CWARN("lookup for %lu/%lu/%lu and data should be uptodate\n",
528 (unsigned long)rpfid.mds, (unsigned long)rpfid.id,
529 (unsigned long)rpfid.generation);
533 if (rc == -ERESTART) {
534 /* directory got splitted since last update. this shouldn't be
535 * becasue splitting causes lock revocation, so revalidate had
536 * to fail and lookup on dir had to return mea */
537 CWARN("we haven't knew about directory splitting!\n");
538 LASSERT(obj == NULL);
540 obj = lmv_create_obj(exp, &rpfid, NULL);
542 RETURN(PTR_ERR(obj));
550 /* okay, MDS has returned success. probably name has been resolved in
552 rc = lmv_handle_remote_inode(exp, uctxt, lmm, lmmsize, it, flags,
555 if (rc == 0 && (mea = body_of_splitted_dir(*reqp, 1))) {
556 /* wow! this is splitted dir, we'd like to handle it */
557 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
558 LASSERT(body != NULL);
560 obj = lmv_grab_obj(obd, &body->fid1);
562 obj = lmv_create_obj(exp, &body->fid1, mea);
564 RETURN(PTR_ERR(obj));
573 int lmv_intent_lock(struct obd_export *exp, struct ll_uctxt *uctxt,
574 struct ll_fid *pfid, const char *name, int len,
575 void *lmm, int lmmsize, struct ll_fid *cfid,
576 struct lookup_intent *it, int flags,
577 struct ptlrpc_request **reqp,
578 ldlm_blocking_callback cb_blocking)
580 struct obd_device *obd = exp->exp_obd;
587 CDEBUG(D_OTHER, "INTENT LOCK '%s' for '%*s' on %lu/%lu -> %u\n",
588 LL_IT2STR(it), len, name, (unsigned long) pfid->id,
589 (unsigned long) pfid->generation, pfid->mds);
591 rc = lmv_check_connect(obd);
595 if (it->it_op == IT_LOOKUP)
596 rc = lmv_intent_lookup(exp, uctxt, pfid, name, len, lmm,
597 lmmsize, cfid, it, flags, reqp,
599 else if (it->it_op & IT_OPEN)
600 rc = lmv_intent_open(exp, uctxt, pfid, name, len, lmm,
601 lmmsize, cfid, it, flags, reqp,
603 else if (it->it_op == IT_GETATTR || it->it_op == IT_CHDIR)
604 rc = lmv_intent_getattr(exp, uctxt, pfid, name, len, lmm,
605 lmmsize, cfid, it, flags, reqp,
612 int lmv_revalidate_slaves(struct obd_export *exp, struct ptlrpc_request **reqp,
613 struct ll_fid *mfid, struct lookup_intent *oit,
614 int master_valid, ldlm_blocking_callback cb_blocking)
616 struct obd_device *obd = exp->exp_obd;
617 struct ptlrpc_request *mreq = *reqp;
618 struct lmv_obd *lmv = &obd->u.lmv;
619 struct lustre_handle master_lockh;
620 struct ldlm_lock *lock;
621 unsigned long size = 0;
622 struct mds_body *body;
623 struct ll_uctxt uctxt;
625 int master_lock_mode;
629 /* we have to loop over the subobjects, check validity and update them
630 * from MDSs if needed. it's very useful that we need not to update all
631 * the fields. say, common fields (that are equal on all the subojects
632 * need not to be update, another fields (i_size, for example) are
633 * cached all the time */
634 obj = lmv_grab_obj(obd, mfid);
635 LASSERT(obj != NULL);
639 master_lock_mode = 0;
643 for (i = 0; i < obj->objcount; i++) {
644 struct ll_fid fid = obj->objs[i].fid;
645 struct lustre_handle *lockh = NULL;
646 struct ptlrpc_request *req = NULL;
647 ldlm_blocking_callback cb;
648 struct lookup_intent it;
651 CDEBUG(D_OTHER, "revalidate subobj %lu/%lu/%lu\n",
652 (unsigned long)fid.mds, (unsigned long)fid.id,
653 (unsigned long) fid.generation);
655 memset(&it, 0, sizeof(it));
656 it.it_op = IT_GETATTR;
657 cb = lmv_dirobj_blocking_ast;
659 if (fid_equal(&fid, &obj->fid)) {
661 /* lmv_intent_getattr() already checked
662 * validness and took the lock */
664 /* it even got the reply refresh attrs
666 body = lustre_msg_buf(mreq->rq_repmsg,
668 LASSERT(body != NULL);
671 /* take already cached attrs into account */
673 "master is locked and cached\n");
681 rc = md_intent_lock(lmv->tgts[fid.mds].ltd_exp, &uctxt, &fid,
682 NULL, 0, NULL, 0, &fid, &it, 0, &req, cb);
683 lockh = (struct lustre_handle *) &it.d.lustre.it_lock_handle;
685 /* nice, this slave is valid */
686 LASSERT(req == NULL);
687 CDEBUG(D_OTHER, "cached\n");
692 /* error during revalidation */
695 /* rc == 0, this means we have no such a lock and can't think
696 * obj is still valid. lookup it again */
697 LASSERT(req == NULL);
700 memset(&it, 0, sizeof(it));
701 it.it_op = IT_GETATTR;
702 rc = md_intent_lock(lmv->tgts[fid.mds].ltd_exp, &uctxt, &fid,
703 NULL, 0, NULL, 0, NULL, &it, 0, &req, cb);
704 lockh = (struct lustre_handle *) &it.d.lustre.it_lock_handle;
708 /* error during lookup */
712 LASSERT(master_valid == 0);
713 /* save lock on master to be returned to the caller */
714 CDEBUG(D_OTHER, "no lock on master yet\n");
715 memcpy(&master_lockh, lockh, sizeof(master_lockh));
716 master_lock_mode = it.d.lustre.it_lock_mode;
717 it.d.lustre.it_lock_mode = 0;
719 /* this is slave. we want to control it */
720 lock = ldlm_handle2lock(lockh);
722 lock->l_ast_data = lmv_get_obj(obj);
727 /* this is first reply, we'll use it to return
728 * updated data back to the caller */
730 ptlrpc_request_addref(req);
735 body = lustre_msg_buf(req->rq_repmsg, 1, sizeof(*body));
739 obj->objs[i].size = body->size;
741 CDEBUG(D_OTHER, "fresh: %lu\n",
742 (unsigned long)obj->objs[i].size);
745 ptlrpc_req_finished(req);
747 size += obj->objs[i].size;
749 if (it.d.lustre.it_lock_mode)
750 ldlm_lock_decref(lockh, it.d.lustre.it_lock_mode);
754 /* some attrs got refreshed, we have reply and it's time to put
755 * fresh attrs to it */
756 CDEBUG(D_OTHER, "return refreshed attrs: size = %lu\n",
757 (unsigned long)size);
759 body = lustre_msg_buf((*reqp)->rq_repmsg, 1, sizeof(*body));
762 /* FIXME: what about other attributes? */
766 /* very important to maintain lli->mds the same because
767 * of revalidation. mreq == NULL means that caller has
768 * no reply and the only attr we can return is size */
769 body->valid = OBD_MD_FLSIZE;
770 body->mds = obj->fid.mds;
772 if (master_valid == 0) {
773 memcpy(&oit->d.lustre.it_lock_handle,
774 &master_lockh, sizeof(master_lockh));
775 oit->d.lustre.it_lock_mode = master_lock_mode;
779 /* it seems all the attrs are fresh and we did no request */
780 CDEBUG(D_OTHER, "all the attrs were fresh\n");
781 if (master_valid == 0)
782 oit->d.lustre.it_lock_mode = master_lock_mode;