Blame view

kernel/pid_namespace.c 6.63 KB
74bd59bb3   Pavel Emelyanov   namespaces: clean...
1
2
3
4
5
6
7
8
9
10
11
12
13
14
  /*
   * Pid namespaces
   *
   * Authors:
   *    (C) 2007 Pavel Emelyanov <xemul@openvz.org>, OpenVZ, SWsoft Inc.
   *    (C) 2007 Sukadev Bhattiprolu <sukadev@us.ibm.com>, IBM
   *     Many thanks to Oleg Nesterov for comments and help
   *
   */
  
  #include <linux/pid.h>
  #include <linux/pid_namespace.h>
  #include <linux/syscalls.h>
  #include <linux/err.h>
0b6b030fc   Pavel Emelyanov   bsdacct: switch f...
15
  #include <linux/acct.h>
5a0e3ad6a   Tejun Heo   include cleanup: ...
16
  #include <linux/slab.h>
4308eebbe   Eric W. Biederman   pidns: call pid_n...
17
  #include <linux/proc_fs.h>
cf3f89214   Daniel Lezcano   pidns: add reboot...
18
  #include <linux/reboot.h>
74bd59bb3   Pavel Emelyanov   namespaces: clean...
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
  
  #define BITS_PER_PAGE		(PAGE_SIZE*8)
  
  struct pid_cache {
  	int nr_ids;
  	char name[16];
  	struct kmem_cache *cachep;
  	struct list_head list;
  };
  
  static LIST_HEAD(pid_caches_lh);
  static DEFINE_MUTEX(pid_caches_mutex);
  static struct kmem_cache *pid_ns_cachep;
  
  /*
   * creates the kmem cache to allocate pids from.
   * @nr_ids: the number of numerical ids this pid will have to carry
   */
  
  static struct kmem_cache *create_pid_cachep(int nr_ids)
  {
  	struct pid_cache *pcache;
  	struct kmem_cache *cachep;
  
  	mutex_lock(&pid_caches_mutex);
  	list_for_each_entry(pcache, &pid_caches_lh, list)
  		if (pcache->nr_ids == nr_ids)
  			goto out;
  
  	pcache = kmalloc(sizeof(struct pid_cache), GFP_KERNEL);
  	if (pcache == NULL)
  		goto err_alloc;
  
  	snprintf(pcache->name, sizeof(pcache->name), "pid_%d", nr_ids);
  	cachep = kmem_cache_create(pcache->name,
  			sizeof(struct pid) + (nr_ids - 1) * sizeof(struct upid),
  			0, SLAB_HWCACHE_ALIGN, NULL);
  	if (cachep == NULL)
  		goto err_cachep;
  
  	pcache->nr_ids = nr_ids;
  	pcache->cachep = cachep;
  	list_add(&pcache->list, &pid_caches_lh);
  out:
  	mutex_unlock(&pid_caches_mutex);
  	return pcache->cachep;
  
  err_cachep:
  	kfree(pcache);
  err_alloc:
  	mutex_unlock(&pid_caches_mutex);
  	return NULL;
  }
ed469a63c   Alexey Dobriyan   pidns: make creat...
72
  static struct pid_namespace *create_pid_namespace(struct pid_namespace *parent_pid_ns)
74bd59bb3   Pavel Emelyanov   namespaces: clean...
73
74
  {
  	struct pid_namespace *ns;
ed469a63c   Alexey Dobriyan   pidns: make creat...
75
  	unsigned int level = parent_pid_ns->level + 1;
4308eebbe   Eric W. Biederman   pidns: call pid_n...
76
  	int i, err = -ENOMEM;
74bd59bb3   Pavel Emelyanov   namespaces: clean...
77

84406c153   Pavel Emelyanov   pidns: use kzallo...
78
  	ns = kmem_cache_zalloc(pid_ns_cachep, GFP_KERNEL);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
79
80
81
82
83
84
85
86
87
88
89
90
  	if (ns == NULL)
  		goto out;
  
  	ns->pidmap[0].page = kzalloc(PAGE_SIZE, GFP_KERNEL);
  	if (!ns->pidmap[0].page)
  		goto out_free;
  
  	ns->pid_cachep = create_pid_cachep(level + 1);
  	if (ns->pid_cachep == NULL)
  		goto out_free_map;
  
  	kref_init(&ns->kref);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
91
  	ns->level = level;
ed469a63c   Alexey Dobriyan   pidns: make creat...
92
  	ns->parent = get_pid_ns(parent_pid_ns);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
93
94
95
  
  	set_bit(0, ns->pidmap[0].page);
  	atomic_set(&ns->pidmap[0].nr_free, BITS_PER_PAGE - 1);
84406c153   Pavel Emelyanov   pidns: use kzallo...
96
  	for (i = 1; i < PIDMAP_ENTRIES; i++)
74bd59bb3   Pavel Emelyanov   namespaces: clean...
97
  		atomic_set(&ns->pidmap[i].nr_free, BITS_PER_PAGE);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
98

4308eebbe   Eric W. Biederman   pidns: call pid_n...
99
100
101
  	err = pid_ns_prepare_proc(ns);
  	if (err)
  		goto out_put_parent_pid_ns;
74bd59bb3   Pavel Emelyanov   namespaces: clean...
102
  	return ns;
4308eebbe   Eric W. Biederman   pidns: call pid_n...
103
104
  out_put_parent_pid_ns:
  	put_pid_ns(parent_pid_ns);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
105
106
107
108
109
  out_free_map:
  	kfree(ns->pidmap[0].page);
  out_free:
  	kmem_cache_free(pid_ns_cachep, ns);
  out:
4308eebbe   Eric W. Biederman   pidns: call pid_n...
110
  	return ERR_PTR(err);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
111
112
113
114
115
116
117
118
119
120
121
122
123
  }
  
  static void destroy_pid_namespace(struct pid_namespace *ns)
  {
  	int i;
  
  	for (i = 0; i < PIDMAP_ENTRIES; i++)
  		kfree(ns->pidmap[i].page);
  	kmem_cache_free(pid_ns_cachep, ns);
  }
  
  struct pid_namespace *copy_pid_ns(unsigned long flags, struct pid_namespace *old_ns)
  {
74bd59bb3   Pavel Emelyanov   namespaces: clean...
124
  	if (!(flags & CLONE_NEWPID))
dca4a9796   Alexey Dobriyan   pidns: rewrite co...
125
  		return get_pid_ns(old_ns);
e5a473869   Sukadev Bhattiprolu   pidns: deny CLONE...
126
  	if (flags & (CLONE_THREAD|CLONE_PARENT))
dca4a9796   Alexey Dobriyan   pidns: rewrite co...
127
128
  		return ERR_PTR(-EINVAL);
  	return create_pid_namespace(old_ns);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
  }
  
  void free_pid_ns(struct kref *kref)
  {
  	struct pid_namespace *ns, *parent;
  
  	ns = container_of(kref, struct pid_namespace, kref);
  
  	parent = ns->parent;
  	destroy_pid_namespace(ns);
  
  	if (parent != NULL)
  		put_pid_ns(parent);
  }
  
  void zap_pid_ns_processes(struct pid_namespace *pid_ns)
  {
  	int nr;
  	int rc;
00c10bc13   Eric W. Biederman   pidns: make kille...
148
149
150
151
152
153
  	struct task_struct *task, *me = current;
  
  	/* Ignore SIGCHLD causing any terminated children to autoreap */
  	spin_lock_irq(&me->sighand->siglock);
  	me->sighand->action[SIGCHLD - 1].sa.sa_handler = SIG_IGN;
  	spin_unlock_irq(&me->sighand->siglock);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
  
  	/*
  	 * The last thread in the cgroup-init thread group is terminating.
  	 * Find remaining pid_ts in the namespace, signal and wait for them
  	 * to exit.
  	 *
  	 * Note:  This signals each threads in the namespace - even those that
  	 * 	  belong to the same thread group, To avoid this, we would have
  	 * 	  to walk the entire tasklist looking a processes in this
  	 * 	  namespace, but that could be unnecessarily expensive if the
  	 * 	  pid namespace has just a few processes. Or we need to
  	 * 	  maintain a tasklist for each pid namespace.
  	 *
  	 */
  	read_lock(&tasklist_lock);
  	nr = next_pidmap(pid_ns, 1);
  	while (nr > 0) {
e4da026f9   Sukadev Bhattiprolu   signals: zap_pid_...
171
  		rcu_read_lock();
e4da026f9   Sukadev Bhattiprolu   signals: zap_pid_...
172
  		task = pid_task(find_vpid(nr), PIDTYPE_PID);
a02d6fd64   Oleg Nesterov   signal: zap_pid_n...
173
174
  		if (task && !__fatal_signal_pending(task))
  			send_sig_info(SIGKILL, SEND_SIG_FORCED, task);
e4da026f9   Sukadev Bhattiprolu   signals: zap_pid_...
175
176
  
  		rcu_read_unlock();
74bd59bb3   Pavel Emelyanov   namespaces: clean...
177
178
179
  		nr = next_pidmap(pid_ns, nr);
  	}
  	read_unlock(&tasklist_lock);
6347e9009   Eric W. Biederman   pidns: guarantee ...
180
  	/* Firstly reap the EXIT_ZOMBIE children we may have. */
74bd59bb3   Pavel Emelyanov   namespaces: clean...
181
182
183
184
  	do {
  		clear_thread_flag(TIF_SIGPENDING);
  		rc = sys_wait4(-1, NULL, __WALL, NULL);
  	} while (rc != -ECHILD);
6347e9009   Eric W. Biederman   pidns: guarantee ...
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
  	/*
  	 * sys_wait4() above can't reap the TASK_DEAD children.
  	 * Make sure they all go away, see __unhash_process().
  	 */
  	for (;;) {
  		bool need_wait = false;
  
  		read_lock(&tasklist_lock);
  		if (!list_empty(&current->children)) {
  			__set_current_state(TASK_UNINTERRUPTIBLE);
  			need_wait = true;
  		}
  		read_unlock(&tasklist_lock);
  
  		if (!need_wait)
  			break;
  		schedule();
  	}
cf3f89214   Daniel Lezcano   pidns: add reboot...
203
204
  	if (pid_ns->reboot)
  		current->signal->group_exit_code = pid_ns->reboot;
0b6b030fc   Pavel Emelyanov   bsdacct: switch f...
205
  	acct_exit_ns(pid_ns);
74bd59bb3   Pavel Emelyanov   namespaces: clean...
206
207
  	return;
  }
98ed57eef   Cyrill Gorcunov   sysctl: make kern...
208
  #ifdef CONFIG_CHECKPOINT_RESTORE
b8f566b04   Pavel Emelyanov   sysctl: add the k...
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
  static int pid_ns_ctl_handler(struct ctl_table *table, int write,
  		void __user *buffer, size_t *lenp, loff_t *ppos)
  {
  	struct ctl_table tmp = *table;
  
  	if (write && !capable(CAP_SYS_ADMIN))
  		return -EPERM;
  
  	/*
  	 * Writing directly to ns' last_pid field is OK, since this field
  	 * is volatile in a living namespace anyway and a code writing to
  	 * it should synchronize its usage with external means.
  	 */
  
  	tmp.data = &current->nsproxy->pid_ns->last_pid;
  	return proc_dointvec(&tmp, write, buffer, lenp, ppos);
  }
  
  static struct ctl_table pid_ns_ctl_table[] = {
  	{
  		.procname = "ns_last_pid",
  		.maxlen = sizeof(int),
  		.mode = 0666, /* permissions are checked in the handler */
  		.proc_handler = pid_ns_ctl_handler,
  	},
  	{ }
  };
b8f566b04   Pavel Emelyanov   sysctl: add the k...
236
  static struct ctl_path kern_path[] = { { .procname = "kernel", }, { } };
98ed57eef   Cyrill Gorcunov   sysctl: make kern...
237
  #endif	/* CONFIG_CHECKPOINT_RESTORE */
b8f566b04   Pavel Emelyanov   sysctl: add the k...
238

cf3f89214   Daniel Lezcano   pidns: add reboot...
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
  int reboot_pid_ns(struct pid_namespace *pid_ns, int cmd)
  {
  	if (pid_ns == &init_pid_ns)
  		return 0;
  
  	switch (cmd) {
  	case LINUX_REBOOT_CMD_RESTART2:
  	case LINUX_REBOOT_CMD_RESTART:
  		pid_ns->reboot = SIGHUP;
  		break;
  
  	case LINUX_REBOOT_CMD_POWER_OFF:
  	case LINUX_REBOOT_CMD_HALT:
  		pid_ns->reboot = SIGINT;
  		break;
  	default:
  		return -EINVAL;
  	}
  
  	read_lock(&tasklist_lock);
  	force_sig(SIGKILL, pid_ns->child_reaper);
  	read_unlock(&tasklist_lock);
  
  	do_exit(0);
  
  	/* Not reached */
  	return 0;
  }
74bd59bb3   Pavel Emelyanov   namespaces: clean...
267
268
269
  static __init int pid_namespaces_init(void)
  {
  	pid_ns_cachep = KMEM_CACHE(pid_namespace, SLAB_PANIC);
98ed57eef   Cyrill Gorcunov   sysctl: make kern...
270
271
  
  #ifdef CONFIG_CHECKPOINT_RESTORE
b8f566b04   Pavel Emelyanov   sysctl: add the k...
272
  	register_sysctl_paths(kern_path, pid_ns_ctl_table);
98ed57eef   Cyrill Gorcunov   sysctl: make kern...
273
  #endif
74bd59bb3   Pavel Emelyanov   namespaces: clean...
274
275
276
277
  	return 0;
  }
  
  __initcall(pid_namespaces_init);