#!/usr/bin/sh
#
# License:      GNU General Public License (GPL)
# Support:      zfs@lists.illumos.org
# Written by:   Saso Kiselkov
#
#	This script manages ZFS pools
#	It can import a ZFS pool or export it
#
#	usage: $0 {start|stop|status|monitor|validate-all|meta-data}
#
#	The "start" arg imports a ZFS pool.
#	The "stop" arg exports it.
#
#       OCF parameters are as follows
#       OCF_RESKEY_pool - the pool to import/export
#
#######################################################################
# Initialization:

: ${OCF_FUNCTIONS_DIR=${OCF_ROOT}/lib/heartbeat}
. ${OCF_FUNCTIONS_DIR}/ocf-shellfuncs

# Defaults
OCF_RESKEY_pool_default=""
OCF_RESKEY_importargs_default=""
OCF_RESKEY_importforce_default=true
OCF_RESKEY_settle_timeout_default="10"
OCF_RESKEY_monitor_settle_ms_default="500"
OCF_RESKEY_kstat_root_default="/proc/spl/kstat/zfs"

: ${OCF_RESKEY_pool=${OCF_RESKEY_pool_default}}
: ${OCF_RESKEY_importargs=${OCF_RESKEY_importargs_default}}
: ${OCF_RESKEY_importforce=${OCF_RESKEY_importforce_default}}
: ${OCF_RESKEY_settle_timeout=${OCF_RESKEY_settle_timeout_default}}
: ${OCF_RESKEY_monitor_settle_ms=${OCF_RESKEY_monitor_settle_ms_default}}

: ${OCF_RESKEY_kstat_root=${OCF_RESKEY_kstat_root_default}}

USAGE="usage: $0 {start|stop|status|monitor|validate-all|meta-data}";

#######################################################################

meta_data() {
	cat <<END
<?xml version="1.0"?>
<!DOCTYPE resource-agent SYSTEM "ra-api-1.dtd">
<resource-agent name="ZFS" version="1.0">
<version>1.0</version>
<longdesc lang="en">
This script manages ZFS pools
It can import a ZFS pool or export it
</longdesc>
<shortdesc lang="en">Manages ZFS pools</shortdesc>

<parameters>
<parameter name="pool" unique="1" required="1">
<longdesc lang="en">
The name of the ZFS pool to manage, e.g. "tank".
</longdesc>
<shortdesc lang="en">ZFS pool name</shortdesc>
<content type="string" default="${OCF_RESKEY_pool_default}" />
</parameter>
<parameter name="importargs" unique="0" required="0">
<longdesc lang="en">
Arguments to zpool import, e.g. "-d /dev/disk/by-id".
</longdesc>
<shortdesc lang="en">Import arguments</shortdesc>
<content type="string" default="${OCF_RESKEY_importargs_default}" />
</parameter>
<parameter name="importforce" unique="0" required="0">
<longdesc lang="en">
zpool import is given the -f option.
</longdesc>
<shortdesc lang="en">Import is forced</shortdesc>
<content type="boolean" default="${OCF_RESKEY_importforce_default}" />
</parameter>
<parameter name="settle_timeout" unique="0" required="0">
<longdesc lang="en">
Seconds start may spend waiting for the pool state to settle after a
successful import.

/proc/spl/kstat/zfs/&lt;pool&gt;/state is a lock-free snapshot of the root vdev
state, and after an import it does not simply transition once to ONLINE: it
oscillates while the vdevs are reopened. A single reading is ONLINE most of
the time while the pool is still unsettled, so start waits for three
consecutive healthy reads 100ms apart instead of one.

Pacemaker starts the recurring monitor as soon as start returns, and a reading
taken inside that window would fail a healthy pool and trigger a needless
recovery. The wait never fails the start; judging the pool is the monitor's
job. The state is read before any waiting, so a pool that is already settled -
the common case - costs nothing. Set to 0 to disable it.
</longdesc>
<shortdesc lang="en">Seconds start waits for the state to settle</shortdesc>
<content type="integer" default="${OCF_RESKEY_settle_timeout_default}" />
</parameter>
<parameter name="monitor_settle_ms" unique="0" required="0">
<longdesc lang="en">
Milliseconds the recurring monitor may spend re-reading an unexpected pool
state before reporting it.

The same oscillation can be sampled by a monitor rather than by start - after
an import this agent performed, or one it did not, such as a probe racing a
manual import. The monitor re-reads for at most this long and reports whatever
the pool settles on.

This is deliberately far smaller than settle_timeout: the monitor only has to
outlast a burst, and every millisecond here delays the detection of a pool
that is genuinely offline. Set to 0 to report the first reading, as before.
</longdesc>
<shortdesc lang="en">Milliseconds the monitor re-reads</shortdesc>
<content type="integer" default="${OCF_RESKEY_monitor_settle_ms_default}" />
</parameter>
<parameter name="kstat_root" unique="0" required="0">
<longdesc lang="en">
Root of the ZFS kstat tree: the agent reads the pool state from
kstat_root/&lt;pool&gt;/state, which is a lock-free snapshot and therefore safe
to poll from the monitor. The default matches every known ZFS build; override
it only when ZFS keeps its kstat tree elsewhere, or to point the agent at a
fake tree when testing.
</longdesc>
<shortdesc lang="en">Root of the ZFS kstat tree</shortdesc>
<content type="string" default="${OCF_RESKEY_kstat_root_default}" />
</parameter>
</parameters>

<actions>
<action name="start"   timeout="60s" />
<action name="stop"    timeout="60s" />
<action name="monitor" depth="0"  timeout="30s" interval="5s" />
<action name="validate-all"  timeout="30s" />
<action name="meta-data"  timeout="5s" />
</actions>
</resource-agent>
END
	exit $OCF_SUCCESS
}

zpool_kstat_state () {
	echo "${OCF_RESKEY_kstat_root}/$OCF_RESKEY_pool/state"
}

zpool_kstat_state_exists () {
	[ -f "$(zpool_kstat_state)" ]
}

# Reads the pool health, preferring the lock-free kstat and falling back to
# zpool list on ZFS versions that do not export it.
zpool_read_health () {
	if zpool_kstat_state_exists; then
		cat "$(zpool_kstat_state)" 2>/dev/null
	else
		zpool list -H -o health "$OCF_RESKEY_pool" 2>/dev/null
	fi
}

zpool_is_imported () {
	# Check if ZFS kstats exists
	if [ -d  "${OCF_RESKEY_kstat_root}/" ] ; then
		# Check the existence of kstats for the pool. If the stats exists,
		# the pool was imported.
		[ -d  "${OCF_RESKEY_kstat_root}/${OCF_RESKEY_pool}" ]
		rc=$?
	else
		# If ZFS kstats do not exists, fallback to the standard check
		zpool list -H "$OCF_RESKEY_pool" > /dev/null
		rc=$?
	fi
	return $rc
}

# Reads the health until it is stable, and echoes the value it settled on.
#
# The kstat is a lock-free snapshot of the root vdev state, and after an import
# it does not simply transition once to ONLINE: it oscillates while the vdevs
# are reopened. Sampling on one system measured 1210 transitions over 65
# minutes, with OFFLINE bursts of 4ms median and 78ms maximum, concentrated in
# the first ~0.5s. A single reading is therefore ONLINE about 93% of the time
# while the pool is still unsettled, so one sample is not stability.
#
# Requiring $1 consecutive healthy reads $2 milliseconds apart covers a burst
# longer than the gap, and returns as soon as the pool is genuinely stable.
#
#   $1 consecutive healthy reads required
#   $2 milliseconds between reads
#   $3 total budget in milliseconds
#
# Returns 0 when stable within the budget, 1 otherwise. The caller classifies
# the echoed value; this never decides success or failure on its own.
zpool_stable_health () {
	need=$1
	gap_ms=$2
	budget_ms=$3
	streak=0
	spent_ms=0
	last=""

	while :; do
		last=$(zpool_read_health)
		case "$last" in
			ONLINE|DEGRADED)
				streak=$((streak + 1))
				if [ "$streak" -ge "$need" ]; then
					echo "$last"
					return 0
				fi
				;;
			*)
				streak=0
				;;
		esac

		if [ "$spent_ms" -ge "$budget_ms" ]; then
			echo "$last"
			return 1
		fi

		sleep "0.$(printf '%03d' "$gap_ms")"
		spent_ms=$((spent_ms + gap_ms))
	done
}

# Clamps a wait to what the current operation timeout can afford. Never spend
# the whole timeout waiting: the configured value is an upper bound the admin
# asked for, the operation timeout is the real one. CRM_meta_timeout is
# supplied per operation by pacemaker, not configured, and is absent during
# validate-all, so it is checked here.
zpool_clamp_budget_s () {
	budget=$1
	if ocf_is_decimal "$OCF_RESKEY_CRM_meta_timeout"; then
		op_budget=$((OCF_RESKEY_CRM_meta_timeout / 1000 - 5))
		if [ "$op_budget" -lt 1 ]; then
			op_budget=1
		fi
		if [ "$budget" -gt "$op_budget" ]; then
			budget=$op_budget
		fi
	fi
	echo "$budget"
}

# Waits for the pool state to settle after an import.
#
# /proc/spl/kstat/zfs/<pool>/state is a lock-free snapshot of the root vdev
# state, so right after an import it can still read OFFLINE (the root vdev is
# briefly VDEV_STATE_CLOSED while vdevs are reopened) or TRANSITIONING.
# Pacemaker starts the recurring monitor as soon as start returns, which is
# precisely when that window is open, and zpool_monitor maps anything that is
# not ONLINE, DEGRADED or FAULTED to OCF_ERR_GENERIC.
#
# This never fails the start. Judging the pool is the monitor's job, and the
# wait must not become a new way for start to fail.
zpool_settle () {
	# Older ZFS has no state kstat, and a pool only has one once imported,
	# so this is a runtime condition rather than a configuration check.
	zpool_kstat_state_exists || return 0

	[ "$OCF_RESKEY_settle_timeout" -gt 0 ] || return 0

	budget_ms=$(( $(zpool_clamp_budget_s "$OCF_RESKEY_settle_timeout") * 1000 ))

	if settled=$(zpool_stable_health 3 100 "$budget_ms"); then
		ocf_log debug "$OCF_RESKEY_pool: start settled on '$settled'"
	else
		ocf_log warn "$OCF_RESKEY_pool: still '$settled' after ${budget_ms}ms"
	fi
	return 0
}

# Forcibly imports a ZFS pool, mounting all of its auto-mounted filesystems
# (as configured in the 'mountpoint' and 'canmount' properties)
# If the pool is already imported, no operation is taken.
zpool_import () {
	if ! zpool_is_imported; then
		ocf_log debug "${OCF_RESKEY_pool}:starting import"

		# The meanings of the options to import are as follows:
		#   -f : import even if the pool is marked as imported to another
		#        system - the system may have failed and not exported it
		#        cleanly.
		#   -o cachefile=none : the import should be temporary, so do not
		#        cache it persistently (across machine reboots). We want
		#        the CRM to explicitly control imports of this pool.
	if ocf_is_true "${OCF_RESKEY_importforce}"; then
		FORCE=-f
	else
		FORCE=""
	fi
		if zpool import $FORCE $OCF_RESKEY_importargs -o cachefile=none "$OCF_RESKEY_pool" ; then
			ocf_log debug "${OCF_RESKEY_pool}:import successful"
			zpool_settle
			return $OCF_SUCCESS
		else
			ocf_log debug "${OCF_RESKEY_pool}:import failed"
			return $OCF_ERR_GENERIC
		fi
	fi
}

# Forcibly exports a ZFS pool, unmounting all of its filesystems in the process
# If the pool is not imported, no operation is taken.
zpool_export () {
	if zpool_is_imported; then
		ocf_log debug "${OCF_RESKEY_pool}:starting export"

		# -f : force the export, even if we have mounted filesystems
		# Please note that this may fail with a "busy" error if there are
		# other kernel subsystems accessing the pool (e.g. SCSI targets).
		# Always make sure the pool export is last in your failover logic.
		if zpool export -f "$OCF_RESKEY_pool" ; then
			ocf_log debug "${OCF_RESKEY_pool}:export successful"
			return $OCF_SUCCESS
        else
			ocf_log debug "${OCF_RESKEY_pool}:export failed"
			return $OCF_ERR_GENERIC
        fi
	fi
}

# Monitors the health of a ZFS pool resource. Please note that this only
# checks whether the pool is imported and functional, not whether it has
# any degraded devices (use monitoring systems such as Zabbix for that).
zpool_monitor () {
	# If the pool is not imported, then we can't monitor its health
	if ! zpool_is_imported; then
		return $OCF_NOT_RUNNING
	fi

	# Check the pool status
	# Since version 0.7.10 status can be obtained without locks
	# https://github.com/zfsonlinux/zfs/pull/7563
	HEALTH=$(zpool_read_health)

	case "$HEALTH" in
		ONLINE|DEGRADED|FAULTED)
			;;
		*)
			# Not necessarily a broken pool: the kstat is a lock-free
			# snapshot and can be read mid-transition, or come back
			# empty if the read raced an export. Re-read before
			# deciding; one that is really broken stays broken.
			if zpool_kstat_state_exists; then k=present; else k=absent; fi
			ocf_log debug "$OCF_RESKEY_pool: '$HEALTH' (kstat=$k), re-reading"

			if [ "$OCF_RESKEY_monitor_settle_ms" -gt 0 ]; then
				budget=$(( $(zpool_clamp_budget_s 1) * 1000 ))
				if [ "$budget" -gt "$OCF_RESKEY_monitor_settle_ms" ]; then
					budget=$OCF_RESKEY_monitor_settle_ms
				fi
				HEALTH=$(zpool_stable_health 2 100 "$budget")
				ocf_log debug "$OCF_RESKEY_pool: settled on '$HEALTH'"
			fi
			;;
	esac

	case "$HEALTH" in
		ONLINE|DEGRADED) return $OCF_SUCCESS;;
		FAULTED)         return $OCF_NOT_RUNNING;;
		*)
			ocf_exit_reason "$OCF_RESKEY_pool: failing on '$HEALTH'"
			return $OCF_ERR_GENERIC
			;;
	esac
}

# Validates whether we can import a given ZFS pool
zpool_validate () {
	# Check that the 'zpool' command is known
	if ! which zpool > /dev/null; then
		return $OCF_ERR_INSTALLED
	fi

	if ! ocf_is_decimal "$OCF_RESKEY_settle_timeout"; then
		ocf_exit_reason "Invalid settle_timeout='${OCF_RESKEY_settle_timeout}': expected a non-negative integer"
		return $OCF_ERR_CONFIGURED
	fi

	if ! ocf_is_decimal "$OCF_RESKEY_monitor_settle_ms"; then
		ocf_exit_reason \
			"monitor_settle_ms='$OCF_RESKEY_monitor_settle_ms' is not a number"
		return $OCF_ERR_CONFIGURED
	fi

	if [ -z "$OCF_RESKEY_pool" ]; then
		ocf_exit_reason "pool is required"
		return $OCF_ERR_CONFIGURED
	fi

	# If the pool is imported, then it is obviously valid
	if zpool_is_imported; then
		return $OCF_SUCCESS
	fi

	# Check that the pool can be imported
	if zpool import $OCF_RESKEY_importargs | grep 'pool:' | grep "\\<$OCF_RESKEY_pool\\>" > /dev/null;
	then
		return $OCF_SUCCESS
	else
		return $OCF_ERR_CONFIGURED
	fi
}

usage () {
	echo "$USAGE" >&2
	return $1
}

if [ $# -ne 1 ]; then
	usage $OCF_ERR_ARGS
fi

case $1 in
	meta-data)      meta_data;;
	start)          zpool_validate || exit $?
	                zpool_import;;
	stop)           zpool_export;;
	status|monitor) zpool_monitor;;
	validate-all)   zpool_validate;;
	usage)          usage $OCF_SUCCESS;;
	*)              usage $OCF_ERR_UNIMPLEMENTED;;
esac

exit $?
