nfssvc.c 21.3 KB
Newer Older
Linus Torvalds's avatar
Linus Torvalds committed
1 2 3 4 5 6 7 8
/*
 * Central processing for nfsd.
 *
 * Authors:	Olaf Kirch (okir@monad.swb.de)
 *
 * Copyright (C) 1995, 1996, 1997 Olaf Kirch <okir@monad.swb.de>
 */

9
#include <linux/sched/signal.h>
10
#include <linux/freezer.h>
11
#include <linux/module.h>
Linus Torvalds's avatar
Linus Torvalds committed
12
#include <linux/fs_struct.h>
13
#include <linux/swap.h>
Linus Torvalds's avatar
Linus Torvalds committed
14 15 16

#include <linux/sunrpc/stats.h>
#include <linux/sunrpc/svcsock.h>
17
#include <linux/sunrpc/svc_xprt.h>
Linus Torvalds's avatar
Linus Torvalds committed
18
#include <linux/lockd/bind.h>
19
#include <linux/nfsacl.h>
20
#include <linux/seq_file.h>
21 22 23
#include <linux/inetdevice.h>
#include <net/addrconf.h>
#include <net/ipv6.h>
24
#include <net/net_namespace.h>
25 26
#include "nfsd.h"
#include "cache.h"
27
#include "vfs.h"
28
#include "netns.h"
Linus Torvalds's avatar
Linus Torvalds committed
29 30 31 32

#define NFSDDBG_FACILITY	NFSDDBG_SVC

extern struct svc_program	nfsd_program;
33
static int			nfsd(void *vrqstp);
Linus Torvalds's avatar
Linus Torvalds committed
34

35
/*
36
 * nfsd_mutex protects nn->nfsd_serv -- both the pointer itself and the members
37 38 39
 * of the svc_serv struct. In particular, ->sv_nrthreads but also to some
 * extent ->sv_temp_socks and ->sv_permsocks. It also protects nfsdstats.th_cnt
 *
40
 * If (out side the lock) nn->nfsd_serv is non-NULL, then it must point to a
41 42 43 44 45 46 47
 * properly initialised 'struct svc_serv' with ->sv_nrthreads > 0. That number
 * of nfsd threads must exist and each must listed in ->sp_all_threads in each
 * entry of ->sv_pools[].
 *
 * Transitions of the thread count between zero and non-zero are of particular
 * interest since the svc_serv needs to be created and initialized at that
 * point, or freed.
48 49 50 51 52 53 54 55
 *
 * Finally, the nfsd_mutex also protects some of the global variables that are
 * accessed when nfsd starts and that are settable via the write_* routines in
 * nfsctl.c. In particular:
 *
 *	user_recovery_dirname
 *	user_lease_time
 *	nfsd_versions
56 57 58
 */
DEFINE_MUTEX(nfsd_mutex);

59 60 61 62 63 64 65
/*
 * nfsd_drc_lock protects nfsd_drc_max_pages and nfsd_drc_pages_used.
 * nfsd_drc_max_pages limits the total amount of memory available for
 * version 4.1 DRC caches.
 * nfsd_drc_pages_used tracks the current version 4.1 DRC memory usage.
 */
spinlock_t	nfsd_drc_lock;
66 67
unsigned long	nfsd_drc_max_mem;
unsigned long	nfsd_drc_mem_used;
68

69 70 71 72 73 74 75 76
#if defined(CONFIG_NFSD_V2_ACL) || defined(CONFIG_NFSD_V3_ACL)
static struct svc_stat	nfsd_acl_svcstats;
static struct svc_version *	nfsd_acl_version[] = {
	[2] = &nfsd_acl_version2,
	[3] = &nfsd_acl_version3,
};

#define NFSD_ACL_MINVERS            2
77
#define NFSD_ACL_NRVERS		ARRAY_SIZE(nfsd_acl_version)
78 79 80 81 82 83
static struct svc_version *nfsd_acl_versions[NFSD_ACL_NRVERS];

static struct svc_program	nfsd_acl_program = {
	.pg_prog		= NFS_ACL_PROGRAM,
	.pg_nvers		= NFSD_ACL_NRVERS,
	.pg_vers		= nfsd_acl_versions,
84
	.pg_name		= "nfsacl",
85 86 87 88 89 90 91 92 93 94
	.pg_class		= "nfsd",
	.pg_stats		= &nfsd_acl_svcstats,
	.pg_authenticate	= &svc_set_client,
};

static struct svc_stat	nfsd_acl_svcstats = {
	.program	= &nfsd_acl_program,
};
#endif /* defined(CONFIG_NFSD_V2_ACL) || defined(CONFIG_NFSD_V3_ACL) */

95 96 97 98 99 100 101 102 103 104 105
static struct svc_version *	nfsd_version[] = {
	[2] = &nfsd_version2,
#if defined(CONFIG_NFSD_V3)
	[3] = &nfsd_version3,
#endif
#if defined(CONFIG_NFSD_V4)
	[4] = &nfsd_version4,
#endif
};

#define NFSD_MINVERS    	2
106
#define NFSD_NRVERS		ARRAY_SIZE(nfsd_version)
107 108 109
static struct svc_version *nfsd_versions[NFSD_NRVERS];

struct svc_program		nfsd_program = {
110 111 112
#if defined(CONFIG_NFSD_V2_ACL) || defined(CONFIG_NFSD_V3_ACL)
	.pg_next		= &nfsd_acl_program,
#endif
113 114 115 116 117 118 119 120 121 122
	.pg_prog		= NFS_PROGRAM,		/* program number */
	.pg_nvers		= NFSD_NRVERS,		/* nr of entries in nfsd_version */
	.pg_vers		= nfsd_versions,	/* version table */
	.pg_name		= "nfsd",		/* program name */
	.pg_class		= "nfsd",		/* authentication class */
	.pg_stats		= &nfsd_svcstats,	/* version table */
	.pg_authenticate	= &svc_set_client,	/* export authentication */

};

123 124 125
static bool nfsd_supported_minorversions[NFSD_SUPPORTED_MINOR_VERSION + 1] = {
	[0] = 1,
	[1] = 1,
126
	[2] = 1,
127
};
128

129 130 131
int nfsd_vers(int vers, enum vers_op change)
{
	if (vers < NFSD_MINVERS || vers >= NFSD_NRVERS)
132
		return 0;
133 134 135 136 137
	switch(change) {
	case NFSD_SET:
		nfsd_versions[vers] = nfsd_version[vers];
#if defined(CONFIG_NFSD_V2_ACL) || defined(CONFIG_NFSD_V3_ACL)
		if (vers < NFSD_ACL_NRVERS)
138
			nfsd_acl_versions[vers] = nfsd_acl_version[vers];
139
#endif
140
		break;
141 142 143 144
	case NFSD_CLEAR:
		nfsd_versions[vers] = NULL;
#if defined(CONFIG_NFSD_V2_ACL) || defined(CONFIG_NFSD_V3_ACL)
		if (vers < NFSD_ACL_NRVERS)
145
			nfsd_acl_versions[vers] = NULL;
146 147 148 149 150 151 152 153 154
#endif
		break;
	case NFSD_TEST:
		return nfsd_versions[vers] != NULL;
	case NFSD_AVAIL:
		return nfsd_version[vers] != NULL;
	}
	return 0;
}
155

156 157 158 159 160 161 162 163 164 165 166 167
static void
nfsd_adjust_nfsd_versions4(void)
{
	unsigned i;

	for (i = 0; i <= NFSD_SUPPORTED_MINOR_VERSION; i++) {
		if (nfsd_supported_minorversions[i])
			return;
	}
	nfsd_vers(4, NFSD_CLEAR);
}

168 169
int nfsd_minorversion(u32 minorversion, enum vers_op change)
{
170 171
	if (minorversion > NFSD_SUPPORTED_MINOR_VERSION &&
	    change != NFSD_AVAIL)
172 173 174
		return -1;
	switch(change) {
	case NFSD_SET:
175
		nfsd_supported_minorversions[minorversion] = true;
176
		nfsd_vers(4, NFSD_SET);
177 178
		break;
	case NFSD_CLEAR:
179
		nfsd_supported_minorversions[minorversion] = false;
180
		nfsd_adjust_nfsd_versions4();
181 182
		break;
	case NFSD_TEST:
183
		return nfsd_supported_minorversions[minorversion];
184 185 186 187 188 189
	case NFSD_AVAIL:
		return minorversion <= NFSD_SUPPORTED_MINOR_VERSION;
	}
	return 0;
}

Linus Torvalds's avatar
Linus Torvalds committed
190 191 192 193 194
/*
 * Maximum number of nfsd processes
 */
#define	NFSD_MAXSERVS		8192

195
int nfsd_nrthreads(struct net *net)
Linus Torvalds's avatar
Linus Torvalds committed
196
{
197
	int rv = 0;
198 199
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);

200
	mutex_lock(&nfsd_mutex);
201 202
	if (nn->nfsd_serv)
		rv = nn->nfsd_serv->sv_nrthreads;
203 204
	mutex_unlock(&nfsd_mutex);
	return rv;
Linus Torvalds's avatar
Linus Torvalds committed
205 206
}

207
static int nfsd_init_socks(struct net *net)
208 209
{
	int error;
210 211 212
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);

	if (!list_empty(&nn->nfsd_serv->sv_permsocks))
213 214
		return 0;

215
	error = svc_create_xprt(nn->nfsd_serv, "udp", net, PF_INET, NFS_PORT,
216 217 218 219
					SVC_SOCK_DEFAULTS);
	if (error < 0)
		return error;

220
	error = svc_create_xprt(nn->nfsd_serv, "tcp", net, PF_INET, NFS_PORT,
221 222 223 224 225 226 227
					SVC_SOCK_DEFAULTS);
	if (error < 0)
		return error;

	return 0;
}

228
static int nfsd_users = 0;
229

230 231 232 233
static int nfsd_startup_generic(int nrservs)
{
	int ret;

234
	if (nfsd_users++)
235 236 237 238 239 240 241 242 243
		return 0;

	/*
	 * Readahead param cache - will no-op if it already exists.
	 * (Note therefore results will be suboptimal if number of
	 * threads is modified after nfsd start.)
	 */
	ret = nfsd_racache_init(2*nrservs);
	if (ret)
244 245
		goto dec_users;

246 247 248 249 250 251 252
	ret = nfs4_state_start();
	if (ret)
		goto out_racache;
	return 0;

out_racache:
	nfsd_racache_shutdown();
253 254
dec_users:
	nfsd_users--;
255 256 257 258 259
	return ret;
}

static void nfsd_shutdown_generic(void)
{
260 261 262
	if (--nfsd_users)
		return;

263 264 265 266
	nfs4_state_shutdown();
	nfsd_racache_shutdown();
}

267 268
static bool nfsd_needs_lockd(void)
{
269
#if defined(CONFIG_NFSD_V3)
270
	return (nfsd_versions[2] != NULL) || (nfsd_versions[3] != NULL);
271 272 273
#else
	return (nfsd_versions[2] != NULL);
#endif
274 275
}

276
static int nfsd_startup_net(int nrservs, struct net *net)
277
{
278
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
279 280
	int ret;

281 282 283
	if (nn->nfsd_net_up)
		return 0;

284
	ret = nfsd_startup_generic(nrservs);
285 286
	if (ret)
		return ret;
287 288 289
	ret = nfsd_init_socks(net);
	if (ret)
		goto out_socks;
290 291 292 293 294 295 296 297

	if (nfsd_needs_lockd() && !nn->lockd_up) {
		ret = lockd_up(net);
		if (ret)
			goto out_socks;
		nn->lockd_up = 1;
	}

298 299 300 301
	ret = nfs4_state_start_net(net);
	if (ret)
		goto out_lockd;

302
	nn->nfsd_net_up = true;
303 304 305
	return 0;

out_lockd:
306 307 308 309
	if (nn->lockd_up) {
		lockd_down(net);
		nn->lockd_up = 0;
	}
310
out_socks:
311
	nfsd_shutdown_generic();
312 313 314
	return ret;
}

315 316
static void nfsd_shutdown_net(struct net *net)
{
317 318
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);

319
	nfs4_state_shutdown_net(net);
320 321 322 323
	if (nn->lockd_up) {
		lockd_down(net);
		nn->lockd_up = 0;
	}
324
	nn->nfsd_net_up = false;
325
	nfsd_shutdown_generic();
326 327
}

328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371
static int nfsd_inetaddr_event(struct notifier_block *this, unsigned long event,
	void *ptr)
{
	struct in_ifaddr *ifa = (struct in_ifaddr *)ptr;
	struct net_device *dev = ifa->ifa_dev->dev;
	struct net *net = dev_net(dev);
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
	struct sockaddr_in sin;

	if (event != NETDEV_DOWN)
		goto out;

	if (nn->nfsd_serv) {
		dprintk("nfsd_inetaddr_event: removed %pI4\n", &ifa->ifa_local);
		sin.sin_family = AF_INET;
		sin.sin_addr.s_addr = ifa->ifa_local;
		svc_age_temp_xprts_now(nn->nfsd_serv, (struct sockaddr *)&sin);
	}

out:
	return NOTIFY_DONE;
}

static struct notifier_block nfsd_inetaddr_notifier = {
	.notifier_call = nfsd_inetaddr_event,
};

#if IS_ENABLED(CONFIG_IPV6)
static int nfsd_inet6addr_event(struct notifier_block *this,
	unsigned long event, void *ptr)
{
	struct inet6_ifaddr *ifa = (struct inet6_ifaddr *)ptr;
	struct net_device *dev = ifa->idev->dev;
	struct net *net = dev_net(dev);
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
	struct sockaddr_in6 sin6;

	if (event != NETDEV_DOWN)
		goto out;

	if (nn->nfsd_serv) {
		dprintk("nfsd_inet6addr_event: removed %pI6\n", &ifa->addr);
		sin6.sin6_family = AF_INET6;
		sin6.sin6_addr = ifa->addr;
372 373
		if (ipv6_addr_type(&sin6.sin6_addr) & IPV6_ADDR_LINKLOCAL)
			sin6.sin6_scope_id = ifa->idev->dev->ifindex;
374 375 376 377 378 379 380 381 382 383 384 385
		svc_age_temp_xprts_now(nn->nfsd_serv, (struct sockaddr *)&sin6);
	}

out:
	return NOTIFY_DONE;
}

static struct notifier_block nfsd_inet6addr_notifier = {
	.notifier_call = nfsd_inet6addr_event,
};
#endif

386 387 388
/* Only used under nfsd_mutex, so this atomic may be overkill: */
static atomic_t nfsd_notifier_refcount = ATOMIC_INIT(0);

389
static void nfsd_last_thread(struct svc_serv *serv, struct net *net)
390
{
391 392
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);

393 394 395
	/* check if the notifier still has clients */
	if (atomic_dec_return(&nfsd_notifier_refcount) == 0) {
		unregister_inetaddr_notifier(&nfsd_inetaddr_notifier);
396
#if IS_ENABLED(CONFIG_IPV6)
397
		unregister_inet6addr_notifier(&nfsd_inet6addr_notifier);
398
#endif
399 400
	}

401 402 403 404
	/*
	 * write_ports can create the server without actually starting
	 * any threads--if we get shut down before any threads are
	 * started, then nfsd_last_thread will be run before any of this
405
	 * other initialization has been done except the rpcb information.
406
	 */
407
	svc_rpcb_cleanup(serv, net);
408
	if (!nn->nfsd_net_up)
409
		return;
410

411
	nfsd_shutdown_net(net);
412 413
	printk(KERN_WARNING "nfsd: last server has exited, flushing export "
			    "cache\n");
414
	nfsd_export_flush(net);
415
}
416 417 418 419 420

void nfsd_reset_versions(void)
{
	int i;

421 422 423
	for (i = 0; i < NFSD_NRVERS; i++)
		if (nfsd_vers(i, NFSD_TEST))
			return;
424

425 426 427 428 429 430 431 432
	for (i = 0; i < NFSD_NRVERS; i++)
		if (i != 4)
			nfsd_vers(i, NFSD_SET);
		else {
			int minor = 0;
			while (nfsd_minorversion(minor, NFSD_SET) >= 0)
				minor++;
		}
433 434
}

435 436 437 438 439 440 441 442 443 444 445 446 447 448
/*
 * Each session guarantees a negotiated per slot memory cache for replies
 * which in turn consumes memory beyond the v2/v3/v4.0 server. A dedicated
 * NFSv4.1 server might want to use more memory for a DRC than a machine
 * with mutiple services.
 *
 * Impose a hard limit on the number of pages for the DRC which varies
 * according to the machines free pages. This is of course only a default.
 *
 * For now this is a #defined shift which could be under admin control
 * in the future.
 */
static void set_max_drc(void)
{
449
	#define NFSD_DRC_SIZE_SHIFT	10
450 451 452
	nfsd_drc_max_mem = (nr_free_buffer_pages()
					>> NFSD_DRC_SIZE_SHIFT) * PAGE_SIZE;
	nfsd_drc_mem_used = 0;
453
	spin_lock_init(&nfsd_drc_lock);
454
	dprintk("%s nfsd_drc_max_mem %lu \n", __func__, nfsd_drc_max_mem);
455
}
456

457
static int nfsd_get_default_max_blksize(void)
458
{
459 460 461
	struct sysinfo i;
	unsigned long long target;
	unsigned long ret;
462

463
	si_meminfo(&i);
464
	target = (i.totalram - i.totalhigh) << PAGE_SHIFT;
465 466 467 468 469 470 471 472 473 474 475 476 477
	/*
	 * Aim for 1/4096 of memory per thread This gives 1MB on 4Gig
	 * machines, but only uses 32K on 128M machines.  Bottom out at
	 * 8K on 32M and smaller.  Of course, this is only a default.
	 */
	target >>= 12;

	ret = NFSSVC_MAXBLKSIZE;
	while (ret > target && ret >= 8*1024*2)
		ret /= 2;
	return ret;
}

478 479 480 481
static struct svc_serv_ops nfsd_thread_sv_ops = {
	.svo_shutdown		= nfsd_last_thread,
	.svo_function		= nfsd,
	.svo_enqueue_xprt	= svc_xprt_do_enqueue,
482
	.svo_setup		= svc_set_num_threads,
483
	.svo_module		= THIS_MODULE,
484 485
};

486
int nfsd_create_serv(struct net *net)
487
{
488
	int error;
489
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
490

491
	WARN_ON(!mutex_is_locked(&nfsd_mutex));
492 493
	if (nn->nfsd_serv) {
		svc_get(nn->nfsd_serv);
494 495
		return 0;
	}
496 497
	if (nfsd_max_blksize == 0)
		nfsd_max_blksize = nfsd_get_default_max_blksize();
498
	nfsd_reset_versions();
499
	nn->nfsd_serv = svc_create_pooled(&nfsd_program, nfsd_max_blksize,
500
						&nfsd_thread_sv_ops);
501
	if (nn->nfsd_serv == NULL)
502
		return -ENOMEM;
503

504
	nn->nfsd_serv->sv_maxconn = nn->max_connections;
505
	error = svc_bind(nn->nfsd_serv, net);
506
	if (error < 0) {
507
		svc_destroy(nn->nfsd_serv);
508 509 510
		return error;
	}

511
	set_max_drc();
512 513 514
	/* check if the notifier is already set */
	if (atomic_inc_return(&nfsd_notifier_refcount) == 1) {
		register_inetaddr_notifier(&nfsd_inetaddr_notifier);
515
#if IS_ENABLED(CONFIG_IPV6)
516
		register_inet6addr_notifier(&nfsd_inet6addr_notifier);
517
#endif
518
	}
519
	do_gettimeofday(&nn->nfssvc_boot);		/* record boot time */
520
	return 0;
521 522
}

523
int nfsd_nrpools(struct net *net)
524
{
525 526 527
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);

	if (nn->nfsd_serv == NULL)
528 529
		return 0;
	else
530
		return nn->nfsd_serv->sv_nrpools;
531 532
}

533
int nfsd_get_nrthreads(int n, int *nthreads, struct net *net)
534 535
{
	int i = 0;
536
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
537

538 539 540
	if (nn->nfsd_serv != NULL) {
		for (i = 0; i < nn->nfsd_serv->sv_nrpools && i < n; i++)
			nthreads[i] = nn->nfsd_serv->sv_pools[i].sp_nrthreads;
541 542 543 544 545
	}

	return 0;
}

546 547 548 549 550 551 552 553 554 555 556 557
void nfsd_destroy(struct net *net)
{
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
	int destroy = (nn->nfsd_serv->sv_nrthreads == 1);

	if (destroy)
		svc_shutdown_net(nn->nfsd_serv, net);
	svc_destroy(nn->nfsd_serv);
	if (destroy)
		nn->nfsd_serv = NULL;
}

558
int nfsd_set_nrthreads(int n, int *nthreads, struct net *net)
559 560 561 562
{
	int i = 0;
	int tot = 0;
	int err = 0;
563
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
564

565 566
	WARN_ON(!mutex_is_locked(&nfsd_mutex));

567
	if (nn->nfsd_serv == NULL || n <= 0)
568 569
		return 0;

570 571
	if (n > nn->nfsd_serv->sv_nrpools)
		n = nn->nfsd_serv->sv_nrpools;
572 573 574 575

	/* enforce a global maximum number of threads */
	tot = 0;
	for (i = 0; i < n; i++) {
576
		nthreads[i] = min(nthreads[i], NFSD_MAXSERVS);
577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599
		tot += nthreads[i];
	}
	if (tot > NFSD_MAXSERVS) {
		/* total too large: scale down requested numbers */
		for (i = 0; i < n && tot > 0; i++) {
		    	int new = nthreads[i] * NFSD_MAXSERVS / tot;
			tot -= (nthreads[i] - new);
			nthreads[i] = new;
		}
		for (i = 0; i < n && tot > 0; i++) {
			nthreads[i]--;
			tot--;
		}
	}

	/*
	 * There must always be a thread in pool 0; the admin
	 * can't shut down NFS completely using pool_threads.
	 */
	if (nthreads[0] == 0)
		nthreads[0] = 1;

	/* apply the new numbers */
600
	svc_get(nn->nfsd_serv);
601
	for (i = 0; i < n; i++) {
602 603
		err = nn->nfsd_serv->sv_ops->svo_setup(nn->nfsd_serv,
				&nn->nfsd_serv->sv_pools[i], nthreads[i]);
604 605 606
		if (err)
			break;
	}
607
	nfsd_destroy(net);
608 609 610
	return err;
}

611 612 613 614 615
/*
 * Adjust the number of threads and return the new number of threads.
 * This is also the function that starts the server if necessary, if
 * this is the first time nrservs is nonzero.
 */
Linus Torvalds's avatar
Linus Torvalds committed
616
int
617
nfsd_svc(int nrservs, struct net *net)
Linus Torvalds's avatar
Linus Torvalds committed
618 619
{
	int	error;
620
	bool	nfsd_up_before;
621
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
622 623

	mutex_lock(&nfsd_mutex);
624
	dprintk("nfsd: creating service\n");
625 626 627

	nrservs = max(nrservs, 0);
	nrservs = min(nrservs, NFSD_MAXSERVS);
628
	error = 0;
629

630
	if (nrservs == 0 && nn->nfsd_serv == NULL)
631 632
		goto out;

633
	error = nfsd_create_serv(net);
634
	if (error)
635 636
		goto out;

637
	nfsd_up_before = nn->nfsd_net_up;
638

639
	error = nfsd_startup_net(nrservs, net);
640 641
	if (error)
		goto out_destroy;
642 643
	error = nn->nfsd_serv->sv_ops->svo_setup(nn->nfsd_serv,
			NULL, nrservs);
644 645
	if (error)
		goto out_shutdown;
646
	/* We are holding a reference to nn->nfsd_serv which
647 648 649
	 * we don't want to count in the return value,
	 * so subtract 1
	 */
650
	error = nn->nfsd_serv->sv_nrthreads - 1;
651
out_shutdown:
652
	if (error < 0 && !nfsd_up_before)
653
		nfsd_shutdown_net(net);
654
out_destroy:
655
	nfsd_destroy(net);		/* Release server */
656
out:
657
	mutex_unlock(&nfsd_mutex);
Linus Torvalds's avatar
Linus Torvalds committed
658 659 660 661 662 663 664
	return error;
}


/*
 * This is the NFS server kernel thread
 */
665 666
static int
nfsd(void *vrqstp)
Linus Torvalds's avatar
Linus Torvalds committed
667
{
668
	struct svc_rqst *rqstp = (struct svc_rqst *) vrqstp;
669 670
	struct svc_xprt *perm_sock = list_entry(rqstp->rq_server->sv_permsocks.next, typeof(struct svc_xprt), xpt_list);
	struct net *net = perm_sock->xpt_net;
671
	struct nfsd_net *nn = net_generic(net, nfsd_net_id);
672
	int err;
Linus Torvalds's avatar
Linus Torvalds committed
673 674

	/* Lock module and set up kernel thread */
675
	mutex_lock(&nfsd_mutex);
Linus Torvalds's avatar
Linus Torvalds committed
676

677
	/* At this point, the thread shares current->fs
678 679
	 * with the init process. We need to create files with the
	 * umask as defined by the client instead of init's umask. */
680
	if (unshare_fs_struct() < 0) {
Linus Torvalds's avatar
Linus Torvalds committed
681 682 683
		printk("Unable to start nfsd thread: out of memory\n");
		goto out;
	}
684

Linus Torvalds's avatar
Linus Torvalds committed
685 686
	current->fs->umask = 0;

687 688
	/*
	 * thread is spawned with all signals set to SIG_IGN, re-enable
689
	 * the ones that will bring down the thread
690
	 */
691 692 693 694
	allow_signal(SIGKILL);
	allow_signal(SIGHUP);
	allow_signal(SIGINT);
	allow_signal(SIGQUIT);
695

Linus Torvalds's avatar
Linus Torvalds committed
696
	nfsdstats.th_cnt++;
697 698
	mutex_unlock(&nfsd_mutex);

699
	set_freezable();
Linus Torvalds's avatar
Linus Torvalds committed
700 701 702 703 704

	/*
	 * The main request loop
	 */
	for (;;) {
705 706 707
		/* Update sv_maxconn if it has changed */
		rqstp->rq_server->sv_maxconn = nn->max_connections;

Linus Torvalds's avatar
Linus Torvalds committed
708 709 710 711
		/*
		 * Find a socket with data available and call its
		 * recvfrom routine.
		 */
712
		while ((err = svc_recv(rqstp, 60*60*HZ)) == -EAGAIN)
Linus Torvalds's avatar
Linus Torvalds committed
713
			;
714
		if (err == -EINTR)
Linus Torvalds's avatar
Linus Torvalds committed
715
			break;
716
		validate_process_creds();
717
		svc_process(rqstp);
718
		validate_process_creds();
Linus Torvalds's avatar
Linus Torvalds committed
719 720
	}

721
	/* Clear signals before calling svc_exit_thread() */
722
	flush_signals(current);
Linus Torvalds's avatar
Linus Torvalds committed
723

724
	mutex_lock(&nfsd_mutex);
Linus Torvalds's avatar
Linus Torvalds committed
725 726 727
	nfsdstats.th_cnt --;

out:
728
	rqstp->rq_server = NULL;
729

Linus Torvalds's avatar
Linus Torvalds committed
730 731 732
	/* Release the thread */
	svc_exit_thread(rqstp);

733
	nfsd_destroy(net);
734

Linus Torvalds's avatar
Linus Torvalds committed
735
	/* Release module */
736
	mutex_unlock(&nfsd_mutex);
Linus Torvalds's avatar
Linus Torvalds committed
737
	module_put_and_exit(0);
738
	return 0;
Linus Torvalds's avatar
Linus Torvalds committed
739 740
}

741 742 743 744 745 746 747 748 749
static __be32 map_new_errors(u32 vers, __be32 nfserr)
{
	if (nfserr == nfserr_jukebox && vers == 2)
		return nfserr_dropit;
	if (nfserr == nfserr_wrongsec && vers < 4)
		return nfserr_acces;
	return nfserr;
}

750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780
/*
 * A write procedure can have a large argument, and a read procedure can
 * have a large reply, but no NFSv2 or NFSv3 procedure has argument and
 * reply that can both be larger than a page.  The xdr code has taken
 * advantage of this assumption to be a sloppy about bounds checking in
 * some cases.  Pending a rewrite of the NFSv2/v3 xdr code to fix that
 * problem, we enforce these assumptions here:
 */
static bool nfs_request_too_big(struct svc_rqst *rqstp,
				struct svc_procedure *proc)
{
	/*
	 * The ACL code has more careful bounds-checking and is not
	 * susceptible to this problem:
	 */
	if (rqstp->rq_prog != NFS_PROGRAM)
		return false;
	/*
	 * Ditto NFSv4 (which can in theory have argument and reply both
	 * more than a page):
	 */
	if (rqstp->rq_vers >= 4)
		return false;
	/* The reply will be small, we're OK: */
	if (proc->pc_xdrressize > 0 &&
	    proc->pc_xdrressize < XDR_QUADLEN(PAGE_SIZE))
		return false;

	return rqstp->rq_arg.len > PAGE_SIZE;
}

Linus Torvalds's avatar
Linus Torvalds committed
781
int
782
nfsd_dispatch(struct svc_rqst *rqstp, __be32 *statp)
Linus Torvalds's avatar
Linus Torvalds committed
783 784 785
{
	struct svc_procedure	*proc;
	kxdrproc_t		xdr;
786 787
	__be32			nfserr;
	__be32			*nfserrp;
Linus Torvalds's avatar
Linus Torvalds committed
788 789 790 791 792

	dprintk("nfsd_dispatch: vers %d proc %d\n",
				rqstp->rq_vers, rqstp->rq_proc);
	proc = rqstp->rq_procinfo;

793 794 795 796 797
	if (nfs_request_too_big(rqstp, proc)) {
		dprintk("nfsd: NFSv%d argument too large\n", rqstp->rq_vers);
		*statp = rpc_garbage_args;
		return 1;
	}
798 799 800 801 802 803 804 805 806 807 808 809 810 811
	/*
	 * Give the xdr decoder a chance to change this if it wants
	 * (necessary in the NFSv4.0 compound case)
	 */
	rqstp->rq_cachetype = proc->pc_cachetype;
	/* Decode arguments */
	xdr = proc->pc_decode;
	if (xdr && !xdr(rqstp, (__be32*)rqstp->rq_arg.head[0].iov_base,
			rqstp->rq_argp)) {
		dprintk("nfsd: failed to decode arguments!\n");
		*statp = rpc_garbage_args;
		return 1;
	}

Linus Torvalds's avatar
Linus Torvalds committed
812
	/* Check whether we have this call in the cache. */
813
	switch (nfsd_cache_lookup(rqstp)) {
Linus Torvalds's avatar
Linus Torvalds committed
814 815 816 817 818 819 820 821 822 823 824 825 826
	case RC_DROPIT:
		return 0;
	case RC_REPLY:
		return 1;
	case RC_DOIT:;
		/* do it */
	}

	/* need to grab the location to store the status, as
	 * nfsv4 does some encoding while processing 
	 */
	nfserrp = rqstp->rq_res.head[0].iov_base
		+ rqstp->rq_res.head[0].iov_len;
827
	rqstp->rq_res.head[0].iov_len += sizeof(__be32);
Linus Torvalds's avatar
Linus Torvalds committed
828 829 830

	/* Now call the procedure handler, and encode NFS status. */
	nfserr = proc->pc_func(rqstp, rqstp->rq_argp, rqstp->rq_resp);
831
	nfserr = map_new_errors(rqstp->rq_vers, nfserr);
832
	if (nfserr == nfserr_dropit || test_bit(RQ_DROPME, &rqstp->rq_flags)) {
833
		dprintk("nfsd: Dropping request; may be revisited later\n");
Linus Torvalds's avatar
Linus Torvalds committed
834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 851 852 853 854 855 856
		nfsd_cache_update(rqstp, RC_NOCACHE, NULL);
		return 0;
	}

	if (rqstp->rq_proc != 0)
		*nfserrp++ = nfserr;

	/* Encode result.
	 * For NFSv2, additional info is never returned in case of an error.
	 */
	if (!(nfserr && rqstp->rq_vers == 2)) {
		xdr = proc->pc_encode;
		if (xdr && !xdr(rqstp, nfserrp,
				rqstp->rq_resp)) {
			/* Failed to encode result. Release cache entry */
			dprintk("nfsd: failed to encode result!\n");
			nfsd_cache_update(rqstp, RC_NOCACHE, NULL);
			*statp = rpc_system_err;
			return 1;
		}
	}

	/* Store reply in cache. */
J. Bruce Fields's avatar
J. Bruce Fields committed
857
	nfsd_cache_update(rqstp, rqstp->rq_cachetype, statp + 1);
Linus Torvalds's avatar
Linus Torvalds committed
858 859
	return 1;
}
860 861 862

int nfsd_pool_stats_open(struct inode *inode, struct file *file)
{
863
	int ret;
864
	struct nfsd_net *nn = net_generic(inode->i_sb->s_fs_info, nfsd_net_id);
865

866
	mutex_lock(&nfsd_mutex);
867
	if (nn->nfsd_serv == NULL) {
868
		mutex_unlock(&nfsd_mutex);
869
		return -ENODEV;
870 871
	}
	/* bump up the psudo refcount while traversing */
872 873
	svc_get(nn->nfsd_serv);
	ret = svc_pool_stats_open(nn->nfsd_serv, file);
874 875 876 877 878 879 880
	mutex_unlock(&nfsd_mutex);
	return ret;
}

int nfsd_pool_stats_release(struct inode *inode, struct file *file)
{
	int ret = seq_release(inode, file);
881
	struct net *net = inode->i_sb->s_fs_info;
882

883 884
	mutex_lock(&nfsd_mutex);
	/* this function really, really should have been called svc_put() */
885
	nfsd_destroy(net);
886 887
	mutex_unlock(&nfsd_mutex);
	return ret;
888
}