replay_barrier() {
local facet=$1
- do_facet $facet sync
+ do_facet $facet "sync; sync; sync"
df $MOUNT
# make sure there will be no seq change
replay_barrier_nodf() {
local facet=$1 echo running=${running}
- do_facet $facet sync
+ do_facet $facet "sync; sync; sync"
local svc=${facet}_svc
echo Replay barrier on ${!svc}
do_facet $facet $LCTL --device %${!svc} notransno
local facet=$1
stop $facet
change_active $facet
+ wait_for_facet $facet
mount_facet $facet -o abort_recovery
clients_up || echo "first df failed: $?"
clients_up || error "post-failover df: $?"
exit 1
}
+host_nids_address() {
+ local nodes=$1
+ local kind=$2
+
+ if [ -n "$kind" ]; then
+ nids=$(do_nodes $nodes "$LCTL list_nids | grep $kind | cut -f 1 -d '@'")
+ else
+ nids=$(do_nodes $nodes "$LCTL list_nids all | cut -f 1 -d '@'")
+ fi
+ echo $nids
+}
+
h2name_or_ip() {
if [ "$1" = "client" -o "$1" = "'*'" ]; then echo \'*\'; else
echo $1"@$2"
fi
}
-at_max_get() {
+at_get() {
local facet=$1
+ local at=$2
- # suppose that all ost-s has the same at_max set
- if [ $facet == "ost" ]; then
- do_facet ost1 "lctl get_param -n at_max"
- else
- do_facet $facet "lctl get_param -n at_max"
- fi
+ # suppose that all ost-s have the same $at value set
+ [ $facet != "ost" ] || facet=ost1
+
+ do_facet $facet "lctl get_param -n $at"
+}
+
+at_max_get() {
+ at_get $1 at_max
}
at_max_set() {
shift
local facet
+ local hosts
for facet in $@; do
if [ $facet == "ost" ]; then
- for i in `seq $OSTCOUNT`; do
- do_facet ost$i "lctl set_param at_max=$at_max"
-
- done
+ facet=$(get_facets OST)
elif [ $facet == "mds" ]; then
- for i in `seq $MDSCOUNT`; do
- do_facet mds$i "lctl set_param at_max=$at_max"
- done
- else
- do_facet $facet "lctl set_param at_max=$at_max"
+ facet=$(get_facets MDS)
fi
+ hosts=$(expand_list $hosts $(facets_hosts $facet))
done
+
+ do_nodes $hosts lctl set_param at_max=$at_max
}
##################################
ostuuid_from_index()
{
- $LFS osts $2 | awk '/^'$1'/ { print $2 }'
+ $LFS osts $2 | sed -ne "/^$1: /s/.* \(.*\) .*$/\1/p"
+}
+
+ostname_from_index() {
+ local uuid=$(ostuuid_from_index $1)
+ echo ${uuid/_UUID/}
+}
+
+index_from_ostuuid()
+{
+ $LFS osts $2 | sed -ne "/${1}/s/\(.*\): .* .*$/\1/p"
}
remote_node () {
}
get_clientosc_proc_path() {
- local ost=$1
-
- # exclude -osc-M*
- echo "${1}-osc-[!M]*"
+ echo "${1}-osc-[^M]*"
}
get_lustre_version () {
_wait_import_state () {
local expected=$1
local CONN_PROC=$2
- local maxtime=${3:-max_recovery_time}
+ local maxtime=${3:-$(max_recovery_time)}
local CONN_STATE
local i=0
CONN_STATE=$($LCTL get_param -n $CONN_PROC 2>/dev/null | cut -f2)
while [ "${CONN_STATE}" != "${expected}" ]; do
+ if [ "${expected}" == "DISCONN" ]; then
+ # for disconn we can check after proc entry is removed
+ [ "x${CONN_STATE}" == "x" ] && return 0
+ # with AT enabled, we can have connect request timeout near of
+ # reconnect timeout and test can't see real disconnect
+ [ "${CONN_STATE}" == "CONNECTING" ] && return 0
+ fi
[ $i -ge $maxtime ] && \
error "can't put import for $CONN_PROC into ${expected} state after $i sec, have ${CONN_STATE}" && \
return 1
wait_import_state() {
local state=$1
local params=$2
- local maxtime=${3:-max_recovery_time}
+ local maxtime=${3:-$(max_recovery_time)}
local param
for param in ${params//,/ }; do
_wait_import_state $state $param $maxtime || return
done
}
+
+# One client request could be timed out because server was not ready
+# when request was sent by client.
+# The request timeout calculation details :
+# ptl_send_rpc ()
+# /* We give the server rq_timeout secs to process the req, and
+# add the network latency for our local timeout. */
+# request->rq_deadline = request->rq_sent + request->rq_timeout +
+# ptlrpc_at_get_net_latency(request) ;
+#
+# ptlrpc_connect_import ()
+# request->rq_timeout = INITIAL_CONNECT_TIMEOUT
+#
+# init_imp_at () ->
+# -> at_init(&at->iat_net_latency, 0, 0) -> iat_net_latency=0
+# ptlrpc_at_get_net_latency(request) ->
+# at_get (max (iat_net_latency=0, at_min)) = at_min
+#
+# i.e.:
+# request->rq_timeout + ptlrpc_at_get_net_latency(request) =
+# INITIAL_CONNECT_TIMEOUT + at_min
+#
+# We will use obd_timeout instead of INITIAL_CONNECT_TIMEOUT
+# because we can not get this value in runtime,
+# the value depends on configure options, and it is not stored in /proc.
+# obd_support.h:
+# #define CONNECTION_SWITCH_MIN 5U
+# #ifndef CRAY_XT3
+# #define INITIAL_CONNECT_TIMEOUT max(CONNECTION_SWITCH_MIN,obd_timeout/20)
+# #else
+# #define INITIAL_CONNECT_TIMEOUT max(CONNECTION_SWITCH_MIN,obd_timeout/2)
+
+request_timeout () {
+ local facet=$1
+
+ # request->rq_timeout = INITIAL_CONNECT_TIMEOUT
+ local init_connect_timeout=$TIMEOUT
+ [[ $init_connect_timeout -ge 5 ]] || init_connect_timeout=5
+
+ local at_min=$(at_get $facet at_min)
+
+ echo $(( init_connect_timeout + at_min ))
+}
+
wait_osc_import_state() {
local facet=$1
local ost_facet=$2
local expected=$3
local ost=$(get_osc_import_name $facet $ost_facet)
- local CONN_PROC
- local CONN_STATE
- local i=0
- CONN_PROC="osc.${ost}.ost_server_uuid"
- CONN_STATE=$(do_facet $facet lctl get_param -n $CONN_PROC 2>/dev/null | cut -f2)
- while [ "${CONN_STATE}" != "${expected}" ]; do
- if [ "${expected}" == "DISCONN" ]; then
- # for disconn we can check after proc entry is removed
- [ "x${CONN_STATE}" == "x" ] && return 0
- # with AT we can have connect request timeout ~ reconnect timeout
- # and test can't see real disconnect
- [ "${CONN_STATE}" == "CONNECTING" ] && return 0
- fi
- # disconnect rpc should be wait not more obd_timeout
- [ $i -ge $(($TIMEOUT * 3 / 2)) ] && \
- error "can't put import for ${ost}(${ost_facet}) into ${expected} state" && return 1
- sleep 1
- CONN_STATE=$(do_facet $facet lctl get_param -n $CONN_PROC 2>/dev/null | cut -f2)
- i=$(($i + 1))
- done
+ local param="osc.${ost}.ost_server_uuid"
+
+ # 1. wait the deadline of client 1st request (it could be skipped)
+ # 2. wait the deadline of client 2nd request
+ local maxtime=$(( 2 * $(request_timeout $facet)))
+
+ if ! do_rpc_nodes $(facet_host $facet) \
+ _wait_import_state $expected $param $maxtime; then
+ error "import is not in ${expected} state"
+ return 1
+ fi
- log "${ost_facet} now in ${CONN_STATE} state"
return 0
}
+
get_clientmdc_proc_path() {
echo "${1}-mdc-*"
}
max_recovery_time () {
local init_connect_timeout=$(( TIMEOUT / 20 ))
- [[ $init_connect_timeout > 5 ]] || init_connect_timeout=5
+ [[ $init_connect_timeout -ge 5 ]] || init_connect_timeout=5
local service_time=$(( $(at_max_get client) + $(( 2 * $(( 25 + 1 + init_connect_timeout)) )) ))
local mdtdev=$2
shift 2
local files="$@"
- local mntpt=${MOUNT%/*}/$facet
+ local mntpt=$(facet_mntpt $facet)
echo "removing files from $mdtdev on $facet: $files"
mount -t $FSTYPE $MDS_MOUNT_OPTS $mdtdev $mntpt || return $?
local mdtdev=$2
shift 2
local files="$@"
- local mntpt=${MOUNT%/*}/$facet
+ local mntpt=$(facet_mntpt $facet)
echo "duplicating files on $mdtdev on $facet: $files"
mkdir -p $mntpt || return $?