user_namespace.c 36 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863864865866867868869870871872873874875876877878879880881882883884885886887888889890891892893894895896897898899900901902903904905906907908909910911912913914915916917918919920921922923924925926927928929930931932933934935936937938939940941942943944945946947948949950951952953954955956957958959960961962963964965966967968969970971972973974975976977978979980981982983984985986987988989990991992993994995996997998999100010011002100310041005100610071008100910101011101210131014101510161017101810191020102110221023102410251026102710281029103010311032103310341035103610371038103910401041104210431044104510461047104810491050105110521053105410551056105710581059106010611062106310641065106610671068106910701071107210731074107510761077107810791080108110821083108410851086108710881089109010911092109310941095109610971098109911001101110211031104110511061107110811091110111111121113111411151116111711181119112011211122112311241125112611271128112911301131113211331134113511361137113811391140114111421143114411451146114711481149115011511152115311541155115611571158115911601161116211631164116511661167116811691170117111721173117411751176117711781179118011811182118311841185118611871188118911901191119211931194119511961197119811991200120112021203120412051206120712081209121012111212121312141215121612171218121912201221122212231224122512261227122812291230123112321233123412351236123712381239124012411242124312441245124612471248124912501251125212531254125512561257125812591260126112621263126412651266126712681269127012711272127312741275127612771278127912801281128212831284128512861287128812891290129112921293129412951296129712981299130013011302130313041305130613071308130913101311131213131314131513161317131813191320132113221323132413251326132713281329133013311332133313341335133613371338133913401341134213431344134513461347134813491350135113521353135413551356135713581359136013611362136313641365136613671368136913701371137213731374137513761377137813791380138113821383138413851386138713881389139013911392139313941395139613971398139914001401140214031404140514061407140814091410
  1. // SPDX-License-Identifier: GPL-2.0-only
  2. #include <linux/export.h>
  3. #include <linux/nsproxy.h>
  4. #include <linux/slab.h>
  5. #include <linux/sched/signal.h>
  6. #include <linux/user_namespace.h>
  7. #include <linux/proc_ns.h>
  8. #include <linux/highuid.h>
  9. #include <linux/cred.h>
  10. #include <linux/securebits.h>
  11. #include <linux/security.h>
  12. #include <linux/keyctl.h>
  13. #include <linux/key-type.h>
  14. #include <keys/user-type.h>
  15. #include <linux/seq_file.h>
  16. #include <linux/fs.h>
  17. #include <linux/uaccess.h>
  18. #include <linux/ctype.h>
  19. #include <linux/projid.h>
  20. #include <linux/fs_struct.h>
  21. #include <linux/bsearch.h>
  22. #include <linux/sort.h>
  23. static struct kmem_cache *user_ns_cachep __ro_after_init;
  24. static DEFINE_MUTEX(userns_state_mutex);
  25. static bool new_idmap_permitted(const struct file *file,
  26. struct user_namespace *ns, int cap_setid,
  27. struct uid_gid_map *map);
  28. static void free_user_ns(struct work_struct *work);
  29. static struct ucounts *inc_user_namespaces(struct user_namespace *ns, kuid_t uid)
  30. {
  31. return inc_ucount(ns, uid, UCOUNT_USER_NAMESPACES);
  32. }
  33. static void dec_user_namespaces(struct ucounts *ucounts)
  34. {
  35. return dec_ucount(ucounts, UCOUNT_USER_NAMESPACES);
  36. }
  37. static void set_cred_user_ns(struct cred *cred, struct user_namespace *user_ns)
  38. {
  39. /* Start with the same capabilities as init but useless for doing
  40. * anything as the capabilities are bound to the new user namespace.
  41. */
  42. cred->securebits = SECUREBITS_DEFAULT;
  43. cred->cap_inheritable = CAP_EMPTY_SET;
  44. cred->cap_permitted = CAP_FULL_SET;
  45. cred->cap_effective = CAP_FULL_SET;
  46. cred->cap_ambient = CAP_EMPTY_SET;
  47. cred->cap_bset = CAP_FULL_SET;
  48. #ifdef CONFIG_KEYS
  49. key_put(cred->request_key_auth);
  50. cred->request_key_auth = NULL;
  51. #endif
  52. /* tgcred will be cleared in our caller bc CLONE_THREAD won't be set */
  53. cred->user_ns = user_ns;
  54. }
  55. static unsigned long enforced_nproc_rlimit(void)
  56. {
  57. unsigned long limit = RLIM_INFINITY;
  58. /* Is RLIMIT_NPROC currently enforced? */
  59. if (!uid_eq(current_uid(), GLOBAL_ROOT_UID) ||
  60. (current_user_ns() != &init_user_ns))
  61. limit = rlimit(RLIMIT_NPROC);
  62. return limit;
  63. }
  64. /*
  65. * Create a new user namespace, deriving the creator from the user in the
  66. * passed credentials, and replacing that user with the new root user for the
  67. * new namespace.
  68. *
  69. * This is called by copy_creds(), which will finish setting the target task's
  70. * credentials.
  71. */
  72. int create_user_ns(struct cred *new)
  73. {
  74. struct user_namespace *ns, *parent_ns = new->user_ns;
  75. kuid_t owner = new->euid;
  76. kgid_t group = new->egid;
  77. struct ucounts *ucounts;
  78. int ret, i;
  79. ret = -ENOSPC;
  80. if (parent_ns->level > 32)
  81. goto fail;
  82. ucounts = inc_user_namespaces(parent_ns, owner);
  83. if (!ucounts)
  84. goto fail;
  85. /*
  86. * Verify that we can not violate the policy of which files
  87. * may be accessed that is specified by the root directory,
  88. * by verifying that the root directory is at the root of the
  89. * mount namespace which allows all files to be accessed.
  90. */
  91. ret = -EPERM;
  92. if (current_chrooted())
  93. goto fail_dec;
  94. /* The creator needs a mapping in the parent user namespace
  95. * or else we won't be able to reasonably tell userspace who
  96. * created a user_namespace.
  97. */
  98. ret = -EPERM;
  99. if (!kuid_has_mapping(parent_ns, owner) ||
  100. !kgid_has_mapping(parent_ns, group))
  101. goto fail_dec;
  102. ret = security_create_user_ns(new);
  103. if (ret < 0)
  104. goto fail_dec;
  105. ret = -ENOMEM;
  106. ns = kmem_cache_zalloc(user_ns_cachep, GFP_KERNEL);
  107. if (!ns)
  108. goto fail_dec;
  109. ns->parent_could_setfcap = cap_raised(new->cap_effective, CAP_SETFCAP);
  110. ret = ns_alloc_inum(&ns->ns);
  111. if (ret)
  112. goto fail_free;
  113. ns->ns.ops = &userns_operations;
  114. refcount_set(&ns->ns.count, 1);
  115. /* Leave the new->user_ns reference with the new user namespace. */
  116. ns->parent = parent_ns;
  117. ns->level = parent_ns->level + 1;
  118. ns->owner = owner;
  119. ns->group = group;
  120. INIT_WORK(&ns->work, free_user_ns);
  121. for (i = 0; i < UCOUNT_COUNTS; i++) {
  122. ns->ucount_max[i] = INT_MAX;
  123. }
  124. set_userns_rlimit_max(ns, UCOUNT_RLIMIT_NPROC, enforced_nproc_rlimit());
  125. set_userns_rlimit_max(ns, UCOUNT_RLIMIT_MSGQUEUE, rlimit(RLIMIT_MSGQUEUE));
  126. set_userns_rlimit_max(ns, UCOUNT_RLIMIT_SIGPENDING, rlimit(RLIMIT_SIGPENDING));
  127. set_userns_rlimit_max(ns, UCOUNT_RLIMIT_MEMLOCK, rlimit(RLIMIT_MEMLOCK));
  128. ns->ucounts = ucounts;
  129. /* Inherit USERNS_SETGROUPS_ALLOWED from our parent */
  130. mutex_lock(&userns_state_mutex);
  131. ns->flags = parent_ns->flags;
  132. mutex_unlock(&userns_state_mutex);
  133. #ifdef CONFIG_KEYS
  134. INIT_LIST_HEAD(&ns->keyring_name_list);
  135. init_rwsem(&ns->keyring_sem);
  136. #endif
  137. ret = -ENOMEM;
  138. if (!setup_userns_sysctls(ns))
  139. goto fail_keyring;
  140. set_cred_user_ns(new, ns);
  141. return 0;
  142. fail_keyring:
  143. #ifdef CONFIG_PERSISTENT_KEYRINGS
  144. key_put(ns->persistent_keyring_register);
  145. #endif
  146. ns_free_inum(&ns->ns);
  147. fail_free:
  148. kmem_cache_free(user_ns_cachep, ns);
  149. fail_dec:
  150. dec_user_namespaces(ucounts);
  151. fail:
  152. return ret;
  153. }
  154. int unshare_userns(unsigned long unshare_flags, struct cred **new_cred)
  155. {
  156. struct cred *cred;
  157. int err = -ENOMEM;
  158. if (!(unshare_flags & CLONE_NEWUSER))
  159. return 0;
  160. cred = prepare_creds();
  161. if (cred) {
  162. err = create_user_ns(cred);
  163. if (err)
  164. put_cred(cred);
  165. else
  166. *new_cred = cred;
  167. }
  168. return err;
  169. }
  170. static void free_user_ns(struct work_struct *work)
  171. {
  172. struct user_namespace *parent, *ns =
  173. container_of(work, struct user_namespace, work);
  174. do {
  175. struct ucounts *ucounts = ns->ucounts;
  176. parent = ns->parent;
  177. if (ns->gid_map.nr_extents > UID_GID_MAP_MAX_BASE_EXTENTS) {
  178. kfree(ns->gid_map.forward);
  179. kfree(ns->gid_map.reverse);
  180. }
  181. if (ns->uid_map.nr_extents > UID_GID_MAP_MAX_BASE_EXTENTS) {
  182. kfree(ns->uid_map.forward);
  183. kfree(ns->uid_map.reverse);
  184. }
  185. if (ns->projid_map.nr_extents > UID_GID_MAP_MAX_BASE_EXTENTS) {
  186. kfree(ns->projid_map.forward);
  187. kfree(ns->projid_map.reverse);
  188. }
  189. #if IS_ENABLED(CONFIG_BINFMT_MISC)
  190. kfree(ns->binfmt_misc);
  191. #endif
  192. retire_userns_sysctls(ns);
  193. key_free_user_ns(ns);
  194. ns_free_inum(&ns->ns);
  195. kmem_cache_free(user_ns_cachep, ns);
  196. dec_user_namespaces(ucounts);
  197. ns = parent;
  198. } while (refcount_dec_and_test(&parent->ns.count));
  199. }
  200. void __put_user_ns(struct user_namespace *ns)
  201. {
  202. schedule_work(&ns->work);
  203. }
  204. EXPORT_SYMBOL(__put_user_ns);
  205. /*
  206. * struct idmap_key - holds the information necessary to find an idmapping in a
  207. * sorted idmap array. It is passed to cmp_map_id() as first argument.
  208. */
  209. struct idmap_key {
  210. bool map_up; /* true -> id from kid; false -> kid from id */
  211. u32 id; /* id to find */
  212. u32 count; /* == 0 unless used with map_id_range_down() */
  213. };
  214. /*
  215. * cmp_map_id - Function to be passed to bsearch() to find the requested
  216. * idmapping. Expects struct idmap_key to be passed via @k.
  217. */
  218. static int cmp_map_id(const void *k, const void *e)
  219. {
  220. u32 first, last, id2;
  221. const struct idmap_key *key = k;
  222. const struct uid_gid_extent *el = e;
  223. id2 = key->id + key->count - 1;
  224. /* handle map_id_{down,up}() */
  225. if (key->map_up)
  226. first = el->lower_first;
  227. else
  228. first = el->first;
  229. last = first + el->count - 1;
  230. if (key->id >= first && key->id <= last &&
  231. (id2 >= first && id2 <= last))
  232. return 0;
  233. if (key->id < first || id2 < first)
  234. return -1;
  235. return 1;
  236. }
  237. /*
  238. * map_id_range_down_max - Find idmap via binary search in ordered idmap array.
  239. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS.
  240. */
  241. static struct uid_gid_extent *
  242. map_id_range_down_max(unsigned extents, struct uid_gid_map *map, u32 id, u32 count)
  243. {
  244. struct idmap_key key;
  245. key.map_up = false;
  246. key.count = count;
  247. key.id = id;
  248. return bsearch(&key, map->forward, extents,
  249. sizeof(struct uid_gid_extent), cmp_map_id);
  250. }
  251. /*
  252. * map_id_range_down_base - Find idmap via binary search in static extent array.
  253. * Can only be called if number of mappings is equal or less than
  254. * UID_GID_MAP_MAX_BASE_EXTENTS.
  255. */
  256. static struct uid_gid_extent *
  257. map_id_range_down_base(unsigned extents, struct uid_gid_map *map, u32 id, u32 count)
  258. {
  259. unsigned idx;
  260. u32 first, last, id2;
  261. id2 = id + count - 1;
  262. /* Find the matching extent */
  263. for (idx = 0; idx < extents; idx++) {
  264. first = map->extent[idx].first;
  265. last = first + map->extent[idx].count - 1;
  266. if (id >= first && id <= last &&
  267. (id2 >= first && id2 <= last))
  268. return &map->extent[idx];
  269. }
  270. return NULL;
  271. }
  272. static u32 map_id_range_down(struct uid_gid_map *map, u32 id, u32 count)
  273. {
  274. struct uid_gid_extent *extent;
  275. unsigned extents = map->nr_extents;
  276. smp_rmb();
  277. if (extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  278. extent = map_id_range_down_base(extents, map, id, count);
  279. else
  280. extent = map_id_range_down_max(extents, map, id, count);
  281. /* Map the id or note failure */
  282. if (extent)
  283. id = (id - extent->first) + extent->lower_first;
  284. else
  285. id = (u32) -1;
  286. return id;
  287. }
  288. u32 map_id_down(struct uid_gid_map *map, u32 id)
  289. {
  290. return map_id_range_down(map, id, 1);
  291. }
  292. /*
  293. * map_id_up_base - Find idmap via binary search in static extent array.
  294. * Can only be called if number of mappings is equal or less than
  295. * UID_GID_MAP_MAX_BASE_EXTENTS.
  296. */
  297. static struct uid_gid_extent *
  298. map_id_up_base(unsigned extents, struct uid_gid_map *map, u32 id)
  299. {
  300. unsigned idx;
  301. u32 first, last;
  302. /* Find the matching extent */
  303. for (idx = 0; idx < extents; idx++) {
  304. first = map->extent[idx].lower_first;
  305. last = first + map->extent[idx].count - 1;
  306. if (id >= first && id <= last)
  307. return &map->extent[idx];
  308. }
  309. return NULL;
  310. }
  311. /*
  312. * map_id_up_max - Find idmap via binary search in ordered idmap array.
  313. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS.
  314. */
  315. static struct uid_gid_extent *
  316. map_id_up_max(unsigned extents, struct uid_gid_map *map, u32 id)
  317. {
  318. struct idmap_key key;
  319. key.map_up = true;
  320. key.count = 1;
  321. key.id = id;
  322. return bsearch(&key, map->reverse, extents,
  323. sizeof(struct uid_gid_extent), cmp_map_id);
  324. }
  325. u32 map_id_up(struct uid_gid_map *map, u32 id)
  326. {
  327. struct uid_gid_extent *extent;
  328. unsigned extents = map->nr_extents;
  329. smp_rmb();
  330. if (extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  331. extent = map_id_up_base(extents, map, id);
  332. else
  333. extent = map_id_up_max(extents, map, id);
  334. /* Map the id or note failure */
  335. if (extent)
  336. id = (id - extent->lower_first) + extent->first;
  337. else
  338. id = (u32) -1;
  339. return id;
  340. }
  341. /**
  342. * make_kuid - Map a user-namespace uid pair into a kuid.
  343. * @ns: User namespace that the uid is in
  344. * @uid: User identifier
  345. *
  346. * Maps a user-namespace uid pair into a kernel internal kuid,
  347. * and returns that kuid.
  348. *
  349. * When there is no mapping defined for the user-namespace uid
  350. * pair INVALID_UID is returned. Callers are expected to test
  351. * for and handle INVALID_UID being returned. INVALID_UID
  352. * may be tested for using uid_valid().
  353. */
  354. kuid_t make_kuid(struct user_namespace *ns, uid_t uid)
  355. {
  356. /* Map the uid to a global kernel uid */
  357. return KUIDT_INIT(map_id_down(&ns->uid_map, uid));
  358. }
  359. EXPORT_SYMBOL(make_kuid);
  360. /**
  361. * from_kuid - Create a uid from a kuid user-namespace pair.
  362. * @targ: The user namespace we want a uid in.
  363. * @kuid: The kernel internal uid to start with.
  364. *
  365. * Map @kuid into the user-namespace specified by @targ and
  366. * return the resulting uid.
  367. *
  368. * There is always a mapping into the initial user_namespace.
  369. *
  370. * If @kuid has no mapping in @targ (uid_t)-1 is returned.
  371. */
  372. uid_t from_kuid(struct user_namespace *targ, kuid_t kuid)
  373. {
  374. /* Map the uid from a global kernel uid */
  375. return map_id_up(&targ->uid_map, __kuid_val(kuid));
  376. }
  377. EXPORT_SYMBOL(from_kuid);
  378. /**
  379. * from_kuid_munged - Create a uid from a kuid user-namespace pair.
  380. * @targ: The user namespace we want a uid in.
  381. * @kuid: The kernel internal uid to start with.
  382. *
  383. * Map @kuid into the user-namespace specified by @targ and
  384. * return the resulting uid.
  385. *
  386. * There is always a mapping into the initial user_namespace.
  387. *
  388. * Unlike from_kuid from_kuid_munged never fails and always
  389. * returns a valid uid. This makes from_kuid_munged appropriate
  390. * for use in syscalls like stat and getuid where failing the
  391. * system call and failing to provide a valid uid are not an
  392. * options.
  393. *
  394. * If @kuid has no mapping in @targ overflowuid is returned.
  395. */
  396. uid_t from_kuid_munged(struct user_namespace *targ, kuid_t kuid)
  397. {
  398. uid_t uid;
  399. uid = from_kuid(targ, kuid);
  400. if (uid == (uid_t) -1)
  401. uid = overflowuid;
  402. return uid;
  403. }
  404. EXPORT_SYMBOL(from_kuid_munged);
  405. /**
  406. * make_kgid - Map a user-namespace gid pair into a kgid.
  407. * @ns: User namespace that the gid is in
  408. * @gid: group identifier
  409. *
  410. * Maps a user-namespace gid pair into a kernel internal kgid,
  411. * and returns that kgid.
  412. *
  413. * When there is no mapping defined for the user-namespace gid
  414. * pair INVALID_GID is returned. Callers are expected to test
  415. * for and handle INVALID_GID being returned. INVALID_GID may be
  416. * tested for using gid_valid().
  417. */
  418. kgid_t make_kgid(struct user_namespace *ns, gid_t gid)
  419. {
  420. /* Map the gid to a global kernel gid */
  421. return KGIDT_INIT(map_id_down(&ns->gid_map, gid));
  422. }
  423. EXPORT_SYMBOL(make_kgid);
  424. /**
  425. * from_kgid - Create a gid from a kgid user-namespace pair.
  426. * @targ: The user namespace we want a gid in.
  427. * @kgid: The kernel internal gid to start with.
  428. *
  429. * Map @kgid into the user-namespace specified by @targ and
  430. * return the resulting gid.
  431. *
  432. * There is always a mapping into the initial user_namespace.
  433. *
  434. * If @kgid has no mapping in @targ (gid_t)-1 is returned.
  435. */
  436. gid_t from_kgid(struct user_namespace *targ, kgid_t kgid)
  437. {
  438. /* Map the gid from a global kernel gid */
  439. return map_id_up(&targ->gid_map, __kgid_val(kgid));
  440. }
  441. EXPORT_SYMBOL(from_kgid);
  442. /**
  443. * from_kgid_munged - Create a gid from a kgid user-namespace pair.
  444. * @targ: The user namespace we want a gid in.
  445. * @kgid: The kernel internal gid to start with.
  446. *
  447. * Map @kgid into the user-namespace specified by @targ and
  448. * return the resulting gid.
  449. *
  450. * There is always a mapping into the initial user_namespace.
  451. *
  452. * Unlike from_kgid from_kgid_munged never fails and always
  453. * returns a valid gid. This makes from_kgid_munged appropriate
  454. * for use in syscalls like stat and getgid where failing the
  455. * system call and failing to provide a valid gid are not options.
  456. *
  457. * If @kgid has no mapping in @targ overflowgid is returned.
  458. */
  459. gid_t from_kgid_munged(struct user_namespace *targ, kgid_t kgid)
  460. {
  461. gid_t gid;
  462. gid = from_kgid(targ, kgid);
  463. if (gid == (gid_t) -1)
  464. gid = overflowgid;
  465. return gid;
  466. }
  467. EXPORT_SYMBOL(from_kgid_munged);
  468. /**
  469. * make_kprojid - Map a user-namespace projid pair into a kprojid.
  470. * @ns: User namespace that the projid is in
  471. * @projid: Project identifier
  472. *
  473. * Maps a user-namespace uid pair into a kernel internal kuid,
  474. * and returns that kuid.
  475. *
  476. * When there is no mapping defined for the user-namespace projid
  477. * pair INVALID_PROJID is returned. Callers are expected to test
  478. * for and handle INVALID_PROJID being returned. INVALID_PROJID
  479. * may be tested for using projid_valid().
  480. */
  481. kprojid_t make_kprojid(struct user_namespace *ns, projid_t projid)
  482. {
  483. /* Map the uid to a global kernel uid */
  484. return KPROJIDT_INIT(map_id_down(&ns->projid_map, projid));
  485. }
  486. EXPORT_SYMBOL(make_kprojid);
  487. /**
  488. * from_kprojid - Create a projid from a kprojid user-namespace pair.
  489. * @targ: The user namespace we want a projid in.
  490. * @kprojid: The kernel internal project identifier to start with.
  491. *
  492. * Map @kprojid into the user-namespace specified by @targ and
  493. * return the resulting projid.
  494. *
  495. * There is always a mapping into the initial user_namespace.
  496. *
  497. * If @kprojid has no mapping in @targ (projid_t)-1 is returned.
  498. */
  499. projid_t from_kprojid(struct user_namespace *targ, kprojid_t kprojid)
  500. {
  501. /* Map the uid from a global kernel uid */
  502. return map_id_up(&targ->projid_map, __kprojid_val(kprojid));
  503. }
  504. EXPORT_SYMBOL(from_kprojid);
  505. /**
  506. * from_kprojid_munged - Create a projiid from a kprojid user-namespace pair.
  507. * @targ: The user namespace we want a projid in.
  508. * @kprojid: The kernel internal projid to start with.
  509. *
  510. * Map @kprojid into the user-namespace specified by @targ and
  511. * return the resulting projid.
  512. *
  513. * There is always a mapping into the initial user_namespace.
  514. *
  515. * Unlike from_kprojid from_kprojid_munged never fails and always
  516. * returns a valid projid. This makes from_kprojid_munged
  517. * appropriate for use in syscalls like stat and where
  518. * failing the system call and failing to provide a valid projid are
  519. * not an options.
  520. *
  521. * If @kprojid has no mapping in @targ OVERFLOW_PROJID is returned.
  522. */
  523. projid_t from_kprojid_munged(struct user_namespace *targ, kprojid_t kprojid)
  524. {
  525. projid_t projid;
  526. projid = from_kprojid(targ, kprojid);
  527. if (projid == (projid_t) -1)
  528. projid = OVERFLOW_PROJID;
  529. return projid;
  530. }
  531. EXPORT_SYMBOL(from_kprojid_munged);
  532. static int uid_m_show(struct seq_file *seq, void *v)
  533. {
  534. struct user_namespace *ns = seq->private;
  535. struct uid_gid_extent *extent = v;
  536. struct user_namespace *lower_ns;
  537. uid_t lower;
  538. lower_ns = seq_user_ns(seq);
  539. if ((lower_ns == ns) && lower_ns->parent)
  540. lower_ns = lower_ns->parent;
  541. lower = from_kuid(lower_ns, KUIDT_INIT(extent->lower_first));
  542. seq_printf(seq, "%10u %10u %10u\n",
  543. extent->first,
  544. lower,
  545. extent->count);
  546. return 0;
  547. }
  548. static int gid_m_show(struct seq_file *seq, void *v)
  549. {
  550. struct user_namespace *ns = seq->private;
  551. struct uid_gid_extent *extent = v;
  552. struct user_namespace *lower_ns;
  553. gid_t lower;
  554. lower_ns = seq_user_ns(seq);
  555. if ((lower_ns == ns) && lower_ns->parent)
  556. lower_ns = lower_ns->parent;
  557. lower = from_kgid(lower_ns, KGIDT_INIT(extent->lower_first));
  558. seq_printf(seq, "%10u %10u %10u\n",
  559. extent->first,
  560. lower,
  561. extent->count);
  562. return 0;
  563. }
  564. static int projid_m_show(struct seq_file *seq, void *v)
  565. {
  566. struct user_namespace *ns = seq->private;
  567. struct uid_gid_extent *extent = v;
  568. struct user_namespace *lower_ns;
  569. projid_t lower;
  570. lower_ns = seq_user_ns(seq);
  571. if ((lower_ns == ns) && lower_ns->parent)
  572. lower_ns = lower_ns->parent;
  573. lower = from_kprojid(lower_ns, KPROJIDT_INIT(extent->lower_first));
  574. seq_printf(seq, "%10u %10u %10u\n",
  575. extent->first,
  576. lower,
  577. extent->count);
  578. return 0;
  579. }
  580. static void *m_start(struct seq_file *seq, loff_t *ppos,
  581. struct uid_gid_map *map)
  582. {
  583. loff_t pos = *ppos;
  584. unsigned extents = map->nr_extents;
  585. smp_rmb();
  586. if (pos >= extents)
  587. return NULL;
  588. if (extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  589. return &map->extent[pos];
  590. return &map->forward[pos];
  591. }
  592. static void *uid_m_start(struct seq_file *seq, loff_t *ppos)
  593. {
  594. struct user_namespace *ns = seq->private;
  595. return m_start(seq, ppos, &ns->uid_map);
  596. }
  597. static void *gid_m_start(struct seq_file *seq, loff_t *ppos)
  598. {
  599. struct user_namespace *ns = seq->private;
  600. return m_start(seq, ppos, &ns->gid_map);
  601. }
  602. static void *projid_m_start(struct seq_file *seq, loff_t *ppos)
  603. {
  604. struct user_namespace *ns = seq->private;
  605. return m_start(seq, ppos, &ns->projid_map);
  606. }
  607. static void *m_next(struct seq_file *seq, void *v, loff_t *pos)
  608. {
  609. (*pos)++;
  610. return seq->op->start(seq, pos);
  611. }
  612. static void m_stop(struct seq_file *seq, void *v)
  613. {
  614. return;
  615. }
  616. const struct seq_operations proc_uid_seq_operations = {
  617. .start = uid_m_start,
  618. .stop = m_stop,
  619. .next = m_next,
  620. .show = uid_m_show,
  621. };
  622. const struct seq_operations proc_gid_seq_operations = {
  623. .start = gid_m_start,
  624. .stop = m_stop,
  625. .next = m_next,
  626. .show = gid_m_show,
  627. };
  628. const struct seq_operations proc_projid_seq_operations = {
  629. .start = projid_m_start,
  630. .stop = m_stop,
  631. .next = m_next,
  632. .show = projid_m_show,
  633. };
  634. static bool mappings_overlap(struct uid_gid_map *new_map,
  635. struct uid_gid_extent *extent)
  636. {
  637. u32 upper_first, lower_first, upper_last, lower_last;
  638. unsigned idx;
  639. upper_first = extent->first;
  640. lower_first = extent->lower_first;
  641. upper_last = upper_first + extent->count - 1;
  642. lower_last = lower_first + extent->count - 1;
  643. for (idx = 0; idx < new_map->nr_extents; idx++) {
  644. u32 prev_upper_first, prev_lower_first;
  645. u32 prev_upper_last, prev_lower_last;
  646. struct uid_gid_extent *prev;
  647. if (new_map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  648. prev = &new_map->extent[idx];
  649. else
  650. prev = &new_map->forward[idx];
  651. prev_upper_first = prev->first;
  652. prev_lower_first = prev->lower_first;
  653. prev_upper_last = prev_upper_first + prev->count - 1;
  654. prev_lower_last = prev_lower_first + prev->count - 1;
  655. /* Does the upper range intersect a previous extent? */
  656. if ((prev_upper_first <= upper_last) &&
  657. (prev_upper_last >= upper_first))
  658. return true;
  659. /* Does the lower range intersect a previous extent? */
  660. if ((prev_lower_first <= lower_last) &&
  661. (prev_lower_last >= lower_first))
  662. return true;
  663. }
  664. return false;
  665. }
  666. /*
  667. * insert_extent - Safely insert a new idmap extent into struct uid_gid_map.
  668. * Takes care to allocate a 4K block of memory if the number of mappings exceeds
  669. * UID_GID_MAP_MAX_BASE_EXTENTS.
  670. */
  671. static int insert_extent(struct uid_gid_map *map, struct uid_gid_extent *extent)
  672. {
  673. struct uid_gid_extent *dest;
  674. if (map->nr_extents == UID_GID_MAP_MAX_BASE_EXTENTS) {
  675. struct uid_gid_extent *forward;
  676. /* Allocate memory for 340 mappings. */
  677. forward = kmalloc_array(UID_GID_MAP_MAX_EXTENTS,
  678. sizeof(struct uid_gid_extent),
  679. GFP_KERNEL);
  680. if (!forward)
  681. return -ENOMEM;
  682. /* Copy over memory. Only set up memory for the forward pointer.
  683. * Defer the memory setup for the reverse pointer.
  684. */
  685. memcpy(forward, map->extent,
  686. map->nr_extents * sizeof(map->extent[0]));
  687. map->forward = forward;
  688. map->reverse = NULL;
  689. }
  690. if (map->nr_extents < UID_GID_MAP_MAX_BASE_EXTENTS)
  691. dest = &map->extent[map->nr_extents];
  692. else
  693. dest = &map->forward[map->nr_extents];
  694. *dest = *extent;
  695. map->nr_extents++;
  696. return 0;
  697. }
  698. /* cmp function to sort() forward mappings */
  699. static int cmp_extents_forward(const void *a, const void *b)
  700. {
  701. const struct uid_gid_extent *e1 = a;
  702. const struct uid_gid_extent *e2 = b;
  703. if (e1->first < e2->first)
  704. return -1;
  705. if (e1->first > e2->first)
  706. return 1;
  707. return 0;
  708. }
  709. /* cmp function to sort() reverse mappings */
  710. static int cmp_extents_reverse(const void *a, const void *b)
  711. {
  712. const struct uid_gid_extent *e1 = a;
  713. const struct uid_gid_extent *e2 = b;
  714. if (e1->lower_first < e2->lower_first)
  715. return -1;
  716. if (e1->lower_first > e2->lower_first)
  717. return 1;
  718. return 0;
  719. }
  720. /*
  721. * sort_idmaps - Sorts an array of idmap entries.
  722. * Can only be called if number of mappings exceeds UID_GID_MAP_MAX_BASE_EXTENTS.
  723. */
  724. static int sort_idmaps(struct uid_gid_map *map)
  725. {
  726. if (map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  727. return 0;
  728. /* Sort forward array. */
  729. sort(map->forward, map->nr_extents, sizeof(struct uid_gid_extent),
  730. cmp_extents_forward, NULL);
  731. /* Only copy the memory from forward we actually need. */
  732. map->reverse = kmemdup_array(map->forward, map->nr_extents,
  733. sizeof(struct uid_gid_extent), GFP_KERNEL);
  734. if (!map->reverse)
  735. return -ENOMEM;
  736. /* Sort reverse array. */
  737. sort(map->reverse, map->nr_extents, sizeof(struct uid_gid_extent),
  738. cmp_extents_reverse, NULL);
  739. return 0;
  740. }
  741. /**
  742. * verify_root_map() - check the uid 0 mapping
  743. * @file: idmapping file
  744. * @map_ns: user namespace of the target process
  745. * @new_map: requested idmap
  746. *
  747. * If a process requests mapping parent uid 0 into the new ns, verify that the
  748. * process writing the map had the CAP_SETFCAP capability as the target process
  749. * will be able to write fscaps that are valid in ancestor user namespaces.
  750. *
  751. * Return: true if the mapping is allowed, false if not.
  752. */
  753. static bool verify_root_map(const struct file *file,
  754. struct user_namespace *map_ns,
  755. struct uid_gid_map *new_map)
  756. {
  757. int idx;
  758. const struct user_namespace *file_ns = file->f_cred->user_ns;
  759. struct uid_gid_extent *extent0 = NULL;
  760. for (idx = 0; idx < new_map->nr_extents; idx++) {
  761. if (new_map->nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  762. extent0 = &new_map->extent[idx];
  763. else
  764. extent0 = &new_map->forward[idx];
  765. if (extent0->lower_first == 0)
  766. break;
  767. extent0 = NULL;
  768. }
  769. if (!extent0)
  770. return true;
  771. if (map_ns == file_ns) {
  772. /* The process unshared its ns and is writing to its own
  773. * /proc/self/uid_map. User already has full capabilites in
  774. * the new namespace. Verify that the parent had CAP_SETFCAP
  775. * when it unshared.
  776. * */
  777. if (!file_ns->parent_could_setfcap)
  778. return false;
  779. } else {
  780. /* Process p1 is writing to uid_map of p2, who is in a child
  781. * user namespace to p1's. Verify that the opener of the map
  782. * file has CAP_SETFCAP against the parent of the new map
  783. * namespace */
  784. if (!file_ns_capable(file, map_ns->parent, CAP_SETFCAP))
  785. return false;
  786. }
  787. return true;
  788. }
  789. static ssize_t map_write(struct file *file, const char __user *buf,
  790. size_t count, loff_t *ppos,
  791. int cap_setid,
  792. struct uid_gid_map *map,
  793. struct uid_gid_map *parent_map)
  794. {
  795. struct seq_file *seq = file->private_data;
  796. struct user_namespace *map_ns = seq->private;
  797. struct uid_gid_map new_map;
  798. unsigned idx;
  799. struct uid_gid_extent extent;
  800. char *kbuf, *pos, *next_line;
  801. ssize_t ret;
  802. /* Only allow < page size writes at the beginning of the file */
  803. if ((*ppos != 0) || (count >= PAGE_SIZE))
  804. return -EINVAL;
  805. /* Slurp in the user data */
  806. kbuf = memdup_user_nul(buf, count);
  807. if (IS_ERR(kbuf))
  808. return PTR_ERR(kbuf);
  809. /*
  810. * The userns_state_mutex serializes all writes to any given map.
  811. *
  812. * Any map is only ever written once.
  813. *
  814. * An id map fits within 1 cache line on most architectures.
  815. *
  816. * On read nothing needs to be done unless you are on an
  817. * architecture with a crazy cache coherency model like alpha.
  818. *
  819. * There is a one time data dependency between reading the
  820. * count of the extents and the values of the extents. The
  821. * desired behavior is to see the values of the extents that
  822. * were written before the count of the extents.
  823. *
  824. * To achieve this smp_wmb() is used on guarantee the write
  825. * order and smp_rmb() is guaranteed that we don't have crazy
  826. * architectures returning stale data.
  827. */
  828. mutex_lock(&userns_state_mutex);
  829. memset(&new_map, 0, sizeof(struct uid_gid_map));
  830. ret = -EPERM;
  831. /* Only allow one successful write to the map */
  832. if (map->nr_extents != 0)
  833. goto out;
  834. /*
  835. * Adjusting namespace settings requires capabilities on the target.
  836. */
  837. if (cap_valid(cap_setid) && !file_ns_capable(file, map_ns, CAP_SYS_ADMIN))
  838. goto out;
  839. /* Parse the user data */
  840. ret = -EINVAL;
  841. pos = kbuf;
  842. for (; pos; pos = next_line) {
  843. /* Find the end of line and ensure I don't look past it */
  844. next_line = strchr(pos, '\n');
  845. if (next_line) {
  846. *next_line = '\0';
  847. next_line++;
  848. if (*next_line == '\0')
  849. next_line = NULL;
  850. }
  851. pos = skip_spaces(pos);
  852. extent.first = simple_strtoul(pos, &pos, 10);
  853. if (!isspace(*pos))
  854. goto out;
  855. pos = skip_spaces(pos);
  856. extent.lower_first = simple_strtoul(pos, &pos, 10);
  857. if (!isspace(*pos))
  858. goto out;
  859. pos = skip_spaces(pos);
  860. extent.count = simple_strtoul(pos, &pos, 10);
  861. if (*pos && !isspace(*pos))
  862. goto out;
  863. /* Verify there is not trailing junk on the line */
  864. pos = skip_spaces(pos);
  865. if (*pos != '\0')
  866. goto out;
  867. /* Verify we have been given valid starting values */
  868. if ((extent.first == (u32) -1) ||
  869. (extent.lower_first == (u32) -1))
  870. goto out;
  871. /* Verify count is not zero and does not cause the
  872. * extent to wrap
  873. */
  874. if ((extent.first + extent.count) <= extent.first)
  875. goto out;
  876. if ((extent.lower_first + extent.count) <=
  877. extent.lower_first)
  878. goto out;
  879. /* Do the ranges in extent overlap any previous extents? */
  880. if (mappings_overlap(&new_map, &extent))
  881. goto out;
  882. if ((new_map.nr_extents + 1) == UID_GID_MAP_MAX_EXTENTS &&
  883. (next_line != NULL))
  884. goto out;
  885. ret = insert_extent(&new_map, &extent);
  886. if (ret < 0)
  887. goto out;
  888. ret = -EINVAL;
  889. }
  890. /* Be very certain the new map actually exists */
  891. if (new_map.nr_extents == 0)
  892. goto out;
  893. ret = -EPERM;
  894. /* Validate the user is allowed to use user id's mapped to. */
  895. if (!new_idmap_permitted(file, map_ns, cap_setid, &new_map))
  896. goto out;
  897. ret = -EPERM;
  898. /* Map the lower ids from the parent user namespace to the
  899. * kernel global id space.
  900. */
  901. for (idx = 0; idx < new_map.nr_extents; idx++) {
  902. struct uid_gid_extent *e;
  903. u32 lower_first;
  904. if (new_map.nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS)
  905. e = &new_map.extent[idx];
  906. else
  907. e = &new_map.forward[idx];
  908. lower_first = map_id_range_down(parent_map,
  909. e->lower_first,
  910. e->count);
  911. /* Fail if we can not map the specified extent to
  912. * the kernel global id space.
  913. */
  914. if (lower_first == (u32) -1)
  915. goto out;
  916. e->lower_first = lower_first;
  917. }
  918. /*
  919. * If we want to use binary search for lookup, this clones the extent
  920. * array and sorts both copies.
  921. */
  922. ret = sort_idmaps(&new_map);
  923. if (ret < 0)
  924. goto out;
  925. /* Install the map */
  926. if (new_map.nr_extents <= UID_GID_MAP_MAX_BASE_EXTENTS) {
  927. memcpy(map->extent, new_map.extent,
  928. new_map.nr_extents * sizeof(new_map.extent[0]));
  929. } else {
  930. map->forward = new_map.forward;
  931. map->reverse = new_map.reverse;
  932. }
  933. smp_wmb();
  934. map->nr_extents = new_map.nr_extents;
  935. *ppos = count;
  936. ret = count;
  937. out:
  938. if (ret < 0 && new_map.nr_extents > UID_GID_MAP_MAX_BASE_EXTENTS) {
  939. kfree(new_map.forward);
  940. kfree(new_map.reverse);
  941. map->forward = NULL;
  942. map->reverse = NULL;
  943. map->nr_extents = 0;
  944. }
  945. mutex_unlock(&userns_state_mutex);
  946. kfree(kbuf);
  947. return ret;
  948. }
  949. ssize_t proc_uid_map_write(struct file *file, const char __user *buf,
  950. size_t size, loff_t *ppos)
  951. {
  952. struct seq_file *seq = file->private_data;
  953. struct user_namespace *ns = seq->private;
  954. struct user_namespace *seq_ns = seq_user_ns(seq);
  955. if (!ns->parent)
  956. return -EPERM;
  957. if ((seq_ns != ns) && (seq_ns != ns->parent))
  958. return -EPERM;
  959. return map_write(file, buf, size, ppos, CAP_SETUID,
  960. &ns->uid_map, &ns->parent->uid_map);
  961. }
  962. ssize_t proc_gid_map_write(struct file *file, const char __user *buf,
  963. size_t size, loff_t *ppos)
  964. {
  965. struct seq_file *seq = file->private_data;
  966. struct user_namespace *ns = seq->private;
  967. struct user_namespace *seq_ns = seq_user_ns(seq);
  968. if (!ns->parent)
  969. return -EPERM;
  970. if ((seq_ns != ns) && (seq_ns != ns->parent))
  971. return -EPERM;
  972. return map_write(file, buf, size, ppos, CAP_SETGID,
  973. &ns->gid_map, &ns->parent->gid_map);
  974. }
  975. ssize_t proc_projid_map_write(struct file *file, const char __user *buf,
  976. size_t size, loff_t *ppos)
  977. {
  978. struct seq_file *seq = file->private_data;
  979. struct user_namespace *ns = seq->private;
  980. struct user_namespace *seq_ns = seq_user_ns(seq);
  981. if (!ns->parent)
  982. return -EPERM;
  983. if ((seq_ns != ns) && (seq_ns != ns->parent))
  984. return -EPERM;
  985. /* Anyone can set any valid project id no capability needed */
  986. return map_write(file, buf, size, ppos, -1,
  987. &ns->projid_map, &ns->parent->projid_map);
  988. }
  989. static bool new_idmap_permitted(const struct file *file,
  990. struct user_namespace *ns, int cap_setid,
  991. struct uid_gid_map *new_map)
  992. {
  993. const struct cred *cred = file->f_cred;
  994. if (cap_setid == CAP_SETUID && !verify_root_map(file, ns, new_map))
  995. return false;
  996. /* Don't allow mappings that would allow anything that wouldn't
  997. * be allowed without the establishment of unprivileged mappings.
  998. */
  999. if ((new_map->nr_extents == 1) && (new_map->extent[0].count == 1) &&
  1000. uid_eq(ns->owner, cred->euid)) {
  1001. u32 id = new_map->extent[0].lower_first;
  1002. if (cap_setid == CAP_SETUID) {
  1003. kuid_t uid = make_kuid(ns->parent, id);
  1004. if (uid_eq(uid, cred->euid))
  1005. return true;
  1006. } else if (cap_setid == CAP_SETGID) {
  1007. kgid_t gid = make_kgid(ns->parent, id);
  1008. if (!(ns->flags & USERNS_SETGROUPS_ALLOWED) &&
  1009. gid_eq(gid, cred->egid))
  1010. return true;
  1011. }
  1012. }
  1013. /* Allow anyone to set a mapping that doesn't require privilege */
  1014. if (!cap_valid(cap_setid))
  1015. return true;
  1016. /* Allow the specified ids if we have the appropriate capability
  1017. * (CAP_SETUID or CAP_SETGID) over the parent user namespace.
  1018. * And the opener of the id file also has the appropriate capability.
  1019. */
  1020. if (ns_capable(ns->parent, cap_setid) &&
  1021. file_ns_capable(file, ns->parent, cap_setid))
  1022. return true;
  1023. return false;
  1024. }
  1025. int proc_setgroups_show(struct seq_file *seq, void *v)
  1026. {
  1027. struct user_namespace *ns = seq->private;
  1028. unsigned long userns_flags = READ_ONCE(ns->flags);
  1029. seq_printf(seq, "%s\n",
  1030. (userns_flags & USERNS_SETGROUPS_ALLOWED) ?
  1031. "allow" : "deny");
  1032. return 0;
  1033. }
  1034. ssize_t proc_setgroups_write(struct file *file, const char __user *buf,
  1035. size_t count, loff_t *ppos)
  1036. {
  1037. struct seq_file *seq = file->private_data;
  1038. struct user_namespace *ns = seq->private;
  1039. char kbuf[8], *pos;
  1040. bool setgroups_allowed;
  1041. ssize_t ret;
  1042. /* Only allow a very narrow range of strings to be written */
  1043. ret = -EINVAL;
  1044. if ((*ppos != 0) || (count >= sizeof(kbuf)))
  1045. goto out;
  1046. /* What was written? */
  1047. ret = -EFAULT;
  1048. if (copy_from_user(kbuf, buf, count))
  1049. goto out;
  1050. kbuf[count] = '\0';
  1051. pos = kbuf;
  1052. /* What is being requested? */
  1053. ret = -EINVAL;
  1054. if (strncmp(pos, "allow", 5) == 0) {
  1055. pos += 5;
  1056. setgroups_allowed = true;
  1057. }
  1058. else if (strncmp(pos, "deny", 4) == 0) {
  1059. pos += 4;
  1060. setgroups_allowed = false;
  1061. }
  1062. else
  1063. goto out;
  1064. /* Verify there is not trailing junk on the line */
  1065. pos = skip_spaces(pos);
  1066. if (*pos != '\0')
  1067. goto out;
  1068. ret = -EPERM;
  1069. mutex_lock(&userns_state_mutex);
  1070. if (setgroups_allowed) {
  1071. /* Enabling setgroups after setgroups has been disabled
  1072. * is not allowed.
  1073. */
  1074. if (!(ns->flags & USERNS_SETGROUPS_ALLOWED))
  1075. goto out_unlock;
  1076. } else {
  1077. /* Permanently disabling setgroups after setgroups has
  1078. * been enabled by writing the gid_map is not allowed.
  1079. */
  1080. if (ns->gid_map.nr_extents != 0)
  1081. goto out_unlock;
  1082. ns->flags &= ~USERNS_SETGROUPS_ALLOWED;
  1083. }
  1084. mutex_unlock(&userns_state_mutex);
  1085. /* Report a successful write */
  1086. *ppos = count;
  1087. ret = count;
  1088. out:
  1089. return ret;
  1090. out_unlock:
  1091. mutex_unlock(&userns_state_mutex);
  1092. goto out;
  1093. }
  1094. bool userns_may_setgroups(const struct user_namespace *ns)
  1095. {
  1096. bool allowed;
  1097. mutex_lock(&userns_state_mutex);
  1098. /* It is not safe to use setgroups until a gid mapping in
  1099. * the user namespace has been established.
  1100. */
  1101. allowed = ns->gid_map.nr_extents != 0;
  1102. /* Is setgroups allowed? */
  1103. allowed = allowed && (ns->flags & USERNS_SETGROUPS_ALLOWED);
  1104. mutex_unlock(&userns_state_mutex);
  1105. return allowed;
  1106. }
  1107. /*
  1108. * Returns true if @child is the same namespace or a descendant of
  1109. * @ancestor.
  1110. */
  1111. bool in_userns(const struct user_namespace *ancestor,
  1112. const struct user_namespace *child)
  1113. {
  1114. const struct user_namespace *ns;
  1115. for (ns = child; ns->level > ancestor->level; ns = ns->parent)
  1116. ;
  1117. return (ns == ancestor);
  1118. }
  1119. bool current_in_userns(const struct user_namespace *target_ns)
  1120. {
  1121. return in_userns(target_ns, current_user_ns());
  1122. }
  1123. EXPORT_SYMBOL(current_in_userns);
  1124. static inline struct user_namespace *to_user_ns(struct ns_common *ns)
  1125. {
  1126. return container_of(ns, struct user_namespace, ns);
  1127. }
  1128. static struct ns_common *userns_get(struct task_struct *task)
  1129. {
  1130. struct user_namespace *user_ns;
  1131. rcu_read_lock();
  1132. user_ns = get_user_ns(__task_cred(task)->user_ns);
  1133. rcu_read_unlock();
  1134. return user_ns ? &user_ns->ns : NULL;
  1135. }
  1136. static void userns_put(struct ns_common *ns)
  1137. {
  1138. put_user_ns(to_user_ns(ns));
  1139. }
  1140. static int userns_install(struct nsset *nsset, struct ns_common *ns)
  1141. {
  1142. struct user_namespace *user_ns = to_user_ns(ns);
  1143. struct cred *cred;
  1144. /* Don't allow gaining capabilities by reentering
  1145. * the same user namespace.
  1146. */
  1147. if (user_ns == current_user_ns())
  1148. return -EINVAL;
  1149. /* Tasks that share a thread group must share a user namespace */
  1150. if (!thread_group_empty(current))
  1151. return -EINVAL;
  1152. if (current->fs->users != 1)
  1153. return -EINVAL;
  1154. if (!ns_capable(user_ns, CAP_SYS_ADMIN))
  1155. return -EPERM;
  1156. cred = nsset_cred(nsset);
  1157. if (!cred)
  1158. return -EINVAL;
  1159. put_user_ns(cred->user_ns);
  1160. set_cred_user_ns(cred, get_user_ns(user_ns));
  1161. if (set_cred_ucounts(cred) < 0)
  1162. return -EINVAL;
  1163. return 0;
  1164. }
  1165. struct ns_common *ns_get_owner(struct ns_common *ns)
  1166. {
  1167. struct user_namespace *my_user_ns = current_user_ns();
  1168. struct user_namespace *owner, *p;
  1169. /* See if the owner is in the current user namespace */
  1170. owner = p = ns->ops->owner(ns);
  1171. for (;;) {
  1172. if (!p)
  1173. return ERR_PTR(-EPERM);
  1174. if (p == my_user_ns)
  1175. break;
  1176. p = p->parent;
  1177. }
  1178. return &get_user_ns(owner)->ns;
  1179. }
  1180. static struct user_namespace *userns_owner(struct ns_common *ns)
  1181. {
  1182. return to_user_ns(ns)->parent;
  1183. }
  1184. const struct proc_ns_operations userns_operations = {
  1185. .name = "user",
  1186. .type = CLONE_NEWUSER,
  1187. .get = userns_get,
  1188. .put = userns_put,
  1189. .install = userns_install,
  1190. .owner = userns_owner,
  1191. .get_parent = ns_get_owner,
  1192. };
  1193. static __init int user_namespaces_init(void)
  1194. {
  1195. user_ns_cachep = KMEM_CACHE(user_namespace, SLAB_PANIC | SLAB_ACCOUNT);
  1196. return 0;
  1197. }
  1198. subsys_initcall(user_namespaces_init);