futex.c

来自「Kernel code of linux kernel」· C语言 代码 · 共 2,098 行 · 第 1/4 页

C
2,098
字号
			if (rt_mutex_trylock(&q.pi_state->pi_mutex))				ret = 0;			else {				/*				 * pi_state is incorrect, some other				 * task did a lock steal and we				 * returned due to timeout or signal				 * without taking the rt_mutex. Too				 * late. We can access the				 * rt_mutex_owner without locking, as				 * the other task is now blocked on				 * the hash bucket lock. Fix the state				 * up.				 */				struct task_struct *owner;				int res;				owner = rt_mutex_owner(&q.pi_state->pi_mutex);				res = fixup_pi_state_owner(uaddr, &q, owner,							   fshared);				/* propagate -EFAULT, if the fixup failed */				if (res)					ret = res;			}		} else {			/*			 * Paranoia check. If we did not take the lock			 * in the trylock above, then we should not be			 * the owner of the rtmutex, neither the real			 * nor the pending one:			 */			if (rt_mutex_owner(&q.pi_state->pi_mutex) == curr)				printk(KERN_ERR "futex_lock_pi: ret = %d "				       "pi-mutex: %p pi-state %p\n", ret,				       q.pi_state->pi_mutex.owner,				       q.pi_state->owner);		}	}	/* Unqueue and drop the lock */	unqueue_me_pi(&q);	futex_unlock_mm(fshared);	if (to)		destroy_hrtimer_on_stack(&to->timer);	return ret != -EINTR ? ret : -ERESTARTNOINTR; out_unlock_release_sem:	queue_unlock(&q, hb); out_release_sem:	futex_unlock_mm(fshared);	if (to)		destroy_hrtimer_on_stack(&to->timer);	return ret; uaddr_faulted:	/*	 * We have to r/w  *(int __user *)uaddr, but we can't modify it	 * non-atomically.  Therefore, if get_user below is not	 * enough, we need to handle the fault ourselves, while	 * still holding the mmap_sem.	 *	 * ... and hb->lock. :-) --ANK	 */	queue_unlock(&q, hb);	if (attempt++) {		ret = futex_handle_fault((unsigned long)uaddr, fshared,					 attempt);		if (ret)			goto out_release_sem;		goto retry_unlocked;	}	futex_unlock_mm(fshared);	ret = get_user(uval, uaddr);	if (!ret && (uval != -EFAULT))		goto retry;	if (to)		destroy_hrtimer_on_stack(&to->timer);	return ret;}/* * Userspace attempted a TID -> 0 atomic transition, and failed. * This is the in-kernel slowpath: we look up the PI state (if any), * and do the rt-mutex unlock. */static int futex_unlock_pi(u32 __user *uaddr, struct rw_semaphore *fshared){	struct futex_hash_bucket *hb;	struct futex_q *this, *next;	u32 uval;	struct plist_head *head;	union futex_key key;	int ret, attempt = 0;retry:	if (get_user(uval, uaddr))		return -EFAULT;	/*	 * We release only a lock we actually own:	 */	if ((uval & FUTEX_TID_MASK) != task_pid_vnr(current))		return -EPERM;	/*	 * First take all the futex related locks:	 */	futex_lock_mm(fshared);	ret = get_futex_key(uaddr, fshared, &key);	if (unlikely(ret != 0))		goto out;	hb = hash_futex(&key);retry_unlocked:	spin_lock(&hb->lock);	/*	 * To avoid races, try to do the TID -> 0 atomic transition	 * again. If it succeeds then we can return without waking	 * anyone else up:	 */	if (!(uval & FUTEX_OWNER_DIED))		uval = cmpxchg_futex_value_locked(uaddr, task_pid_vnr(current), 0);	if (unlikely(uval == -EFAULT))		goto pi_faulted;	/*	 * Rare case: we managed to release the lock atomically,	 * no need to wake anyone else up:	 */	if (unlikely(uval == task_pid_vnr(current)))		goto out_unlock;	/*	 * Ok, other tasks may need to be woken up - check waiters	 * and do the wakeup if necessary:	 */	head = &hb->chain;	plist_for_each_entry_safe(this, next, head, list) {		if (!match_futex (&this->key, &key))			continue;		ret = wake_futex_pi(uaddr, uval, this);		/*		 * The atomic access to the futex value		 * generated a pagefault, so retry the		 * user-access and the wakeup:		 */		if (ret == -EFAULT)			goto pi_faulted;		goto out_unlock;	}	/*	 * No waiters - kernel unlocks the futex:	 */	if (!(uval & FUTEX_OWNER_DIED)) {		ret = unlock_futex_pi(uaddr, uval);		if (ret == -EFAULT)			goto pi_faulted;	}out_unlock:	spin_unlock(&hb->lock);out:	futex_unlock_mm(fshared);	return ret;pi_faulted:	/*	 * We have to r/w  *(int __user *)uaddr, but we can't modify it	 * non-atomically.  Therefore, if get_user below is not	 * enough, we need to handle the fault ourselves, while	 * still holding the mmap_sem.	 *	 * ... and hb->lock. --ANK	 */	spin_unlock(&hb->lock);	if (attempt++) {		ret = futex_handle_fault((unsigned long)uaddr, fshared,					 attempt);		if (ret)			goto out;		uval = 0;		goto retry_unlocked;	}	futex_unlock_mm(fshared);	ret = get_user(uval, uaddr);	if (!ret && (uval != -EFAULT))		goto retry;	return ret;}/* * Support for robust futexes: the kernel cleans up held futexes at * thread exit time. * * Implementation: user-space maintains a per-thread list of locks it * is holding. Upon do_exit(), the kernel carefully walks this list, * and marks all locks that are owned by this thread with the * FUTEX_OWNER_DIED bit, and wakes up a waiter (if any). The list is * always manipulated with the lock held, so the list is private and * per-thread. Userspace also maintains a per-thread 'list_op_pending' * field, to allow the kernel to clean up if the thread dies after * acquiring the lock, but just before it could have added itself to * the list. There can only be one such pending lock. *//** * sys_set_robust_list - set the robust-futex list head of a task * @head: pointer to the list-head * @len: length of the list-head, as userspace expects */asmlinkage longsys_set_robust_list(struct robust_list_head __user *head,		    size_t len){	if (!futex_cmpxchg_enabled)		return -ENOSYS;	/*	 * The kernel knows only one size for now:	 */	if (unlikely(len != sizeof(*head)))		return -EINVAL;	current->robust_list = head;	return 0;}/** * sys_get_robust_list - get the robust-futex list head of a task * @pid: pid of the process [zero for current task] * @head_ptr: pointer to a list-head pointer, the kernel fills it in * @len_ptr: pointer to a length field, the kernel fills in the header size */asmlinkage longsys_get_robust_list(int pid, struct robust_list_head __user * __user *head_ptr,		    size_t __user *len_ptr){	struct robust_list_head __user *head;	unsigned long ret;	if (!futex_cmpxchg_enabled)		return -ENOSYS;	if (!pid)		head = current->robust_list;	else {		struct task_struct *p;		ret = -ESRCH;		rcu_read_lock();		p = find_task_by_vpid(pid);		if (!p)			goto err_unlock;		ret = -EPERM;		if ((current->euid != p->euid) && (current->euid != p->uid) &&				!capable(CAP_SYS_PTRACE))			goto err_unlock;		head = p->robust_list;		rcu_read_unlock();	}	if (put_user(sizeof(*head), len_ptr))		return -EFAULT;	return put_user(head, head_ptr);err_unlock:	rcu_read_unlock();	return ret;}/* * Process a futex-list entry, check whether it's owned by the * dying task, and do notification if so: */int handle_futex_death(u32 __user *uaddr, struct task_struct *curr, int pi){	u32 uval, nval, mval;retry:	if (get_user(uval, uaddr))		return -1;	if ((uval & FUTEX_TID_MASK) == task_pid_vnr(curr)) {		/*		 * Ok, this dying thread is truly holding a futex		 * of interest. Set the OWNER_DIED bit atomically		 * via cmpxchg, and if the value had FUTEX_WAITERS		 * set, wake up a waiter (if any). (We have to do a		 * futex_wake() even if OWNER_DIED is already set -		 * to handle the rare but possible case of recursive		 * thread-death.) The rest of the cleanup is done in		 * userspace.		 */		mval = (uval & FUTEX_WAITERS) | FUTEX_OWNER_DIED;		nval = futex_atomic_cmpxchg_inatomic(uaddr, uval, mval);		if (nval == -EFAULT)			return -1;		if (nval != uval)			goto retry;		/*		 * Wake robust non-PI futexes here. The wakeup of		 * PI futexes happens in exit_pi_state():		 */		if (!pi && (uval & FUTEX_WAITERS))			futex_wake(uaddr, &curr->mm->mmap_sem, 1,				   FUTEX_BITSET_MATCH_ANY);	}	return 0;}/* * Fetch a robust-list pointer. Bit 0 signals PI futexes: */static inline int fetch_robust_entry(struct robust_list __user **entry,				     struct robust_list __user * __user *head,				     int *pi){	unsigned long uentry;	if (get_user(uentry, (unsigned long __user *)head))		return -EFAULT;	*entry = (void __user *)(uentry & ~1UL);	*pi = uentry & 1;	return 0;}/* * Walk curr->robust_list (very carefully, it's a userspace list!) * and mark any locks found there dead, and notify any waiters. * * We silently return on any sign of list-walking problem. */void exit_robust_list(struct task_struct *curr){	struct robust_list_head __user *head = curr->robust_list;	struct robust_list __user *entry, *next_entry, *pending;	unsigned int limit = ROBUST_LIST_LIMIT, pi, next_pi, pip;	unsigned long futex_offset;	int rc;	if (!futex_cmpxchg_enabled)		return;	/*	 * Fetch the list head (which was registered earlier, via	 * sys_set_robust_list()):	 */	if (fetch_robust_entry(&entry, &head->list.next, &pi))		return;	/*	 * Fetch the relative futex offset:	 */	if (get_user(futex_offset, &head->futex_offset))		return;	/*	 * Fetch any possibly pending lock-add first, and handle it	 * if it exists:	 */	if (fetch_robust_entry(&pending, &head->list_op_pending, &pip))		return;	next_entry = NULL;	/* avoid warning with gcc */	while (entry != &head->list) {		/*		 * Fetch the next entry in the list before calling		 * handle_futex_death:		 */		rc = fetch_robust_entry(&next_entry, &entry->next, &next_pi);		/*		 * A pending lock might already be on the list, so		 * don't process it twice:		 */		if (entry != pending)			if (handle_futex_death((void __user *)entry + futex_offset,						curr, pi))				return;		if (rc)			return;		entry = next_entry;		pi = next_pi;		/*		 * Avoid excessively long or circular lists:		 */		if (!--limit)			break;		cond_resched();	}	if (pending)		handle_futex_death((void __user *)pending + futex_offset,				   curr, pip);}long do_futex(u32 __user *uaddr, int op, u32 val, ktime_t *timeout,		u32 __user *uaddr2, u32 val2, u32 val3){	int ret = -ENOSYS;	int cmd = op & FUTEX_CMD_MASK;	struct rw_semaphore *fshared = NULL;	if (!(op & FUTEX_PRIVATE_FLAG))		fshared = &current->mm->mmap_sem;	switch (cmd) {	case FUTEX_WAIT:		val3 = FUTEX_BITSET_MATCH_ANY;	case FUTEX_WAIT_BITSET:		ret = futex_wait(uaddr, fshared, val, timeout, val3);		break;	case FUTEX_WAKE:		val3 = FUTEX_BITSET_MATCH_ANY;	case FUTEX_WAKE_BITSET:		ret = futex_wake(uaddr, fshared, val, val3);		break;	case FUTEX_REQUEUE:		ret = futex_requeue(uaddr, fshared, uaddr2, val, val2, NULL);		break;	case FUTEX_CMP_REQUEUE:		ret = futex_requeue(uaddr, fshared, uaddr2, val, val2, &val3);		break;	case FUTEX_WAKE_OP:		ret = futex_wake_op(uaddr, fshared, uaddr2, val, val2, val3);		break;	case FUTEX_LOCK_PI:		if (futex_cmpxchg_enabled)			ret = futex_lock_pi(uaddr, fshared, val, timeout, 0);		break;	case FUTEX_UNLOCK_PI:		if (futex_cmpxchg_enabled)			ret = futex_unlock_pi(uaddr, fshared);		break;	case FUTEX_TRYLOCK_PI:		if (futex_cmpxchg_enabled)			ret = futex_lock_pi(uaddr, fshared, 0, timeout, 1);		break;	default:		ret = -ENOSYS;	}	return ret;}asmlinkage long sys_futex(u32 __user *uaddr, int op, u32 val,			  struct timespec __user *utime, u32 __user *uaddr2,			  u32 val3){	struct timespec ts;	ktime_t t, *tp = NULL;	u32 val2 = 0;	int cmd = op & FUTEX_CMD_MASK;	if (utime && (cmd == FUTEX_WAIT || cmd == FUTEX_LOCK_PI ||		      cmd == FUTEX_WAIT_BITSET)) {		if (copy_from_user(&ts, utime, sizeof(ts)) != 0)			return -EFAULT;		if (!timespec_valid(&ts))			return -EINVAL;		t = timespec_to_ktime(ts);		if (cmd == FUTEX_WAIT)			t = ktime_add_safe(ktime_get(), t);		tp = &t;	}	/*	 * requeue parameter in 'utime' if cmd == FUTEX_REQUEUE.	 * number of waiters to wake in 'utime' if cmd == FUTEX_WAKE_OP.	 */	if (cmd == FUTEX_REQUEUE || cmd == FUTEX_CMP_REQUEUE ||	    cmd == FUTEX_WAKE_OP)		val2 = (u32) (unsigned long) utime;	return do_futex(uaddr, op, val, tp, uaddr2, val2, val3);}static int __init futex_init(void){	u32 curval;	int i;	/*	 * This will fail and we want it. Some arch implementations do	 * runtime detection of the futex_atomic_cmpxchg_inatomic()	 * functionality. We want to know that before we call in any	 * of the complex code paths. Also we want to prevent	 * registration of robust lists in that case. NULL is	 * guaranteed to fault and we get -EFAULT on functional	 * implementation, the non functional ones will return	 * -ENOSYS.	 */	curval = cmpxchg_futex_value_locked(NULL, 0, 0);	if (curval == -EFAULT)		futex_cmpxchg_enabled = 1;	for (i = 0; i < ARRAY_SIZE(futex_queues); i++) {		plist_head_init(&futex_queues[i].chain, &futex_queues[i].lock);		spin_lock_init(&futex_queues[i].lock);	}	return 0;}__initcall(futex_init);

⌨️ 快捷键说明

复制代码Ctrl + C
搜索代码Ctrl + F
全屏模式F11
增大字号Ctrl + =
减小字号Ctrl + -
显示快捷键?