root-tree.c 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528
  1. // SPDX-License-Identifier: GPL-2.0
  2. /*
  3. * Copyright (C) 2007 Oracle. All rights reserved.
  4. */
  5. #include <linux/err.h>
  6. #include <linux/uuid.h>
  7. #include "ctree.h"
  8. #include "fs.h"
  9. #include "messages.h"
  10. #include "transaction.h"
  11. #include "disk-io.h"
  12. #include "qgroup.h"
  13. #include "space-info.h"
  14. #include "accessors.h"
  15. #include "root-tree.h"
  16. #include "orphan.h"
  17. /*
  18. * Read a root item from the tree. In case we detect a root item smaller then
  19. * sizeof(root_item), we know it's an old version of the root structure and
  20. * initialize all new fields to zero. The same happens if we detect mismatching
  21. * generation numbers as then we know the root was once mounted with an older
  22. * kernel that was not aware of the root item structure change.
  23. */
  24. static void btrfs_read_root_item(struct extent_buffer *eb, int slot,
  25. struct btrfs_root_item *item)
  26. {
  27. u32 len;
  28. int need_reset = 0;
  29. len = btrfs_item_size(eb, slot);
  30. read_extent_buffer(eb, item, btrfs_item_ptr_offset(eb, slot),
  31. min_t(u32, len, sizeof(*item)));
  32. if (len < sizeof(*item))
  33. need_reset = 1;
  34. if (!need_reset && btrfs_root_generation(item)
  35. != btrfs_root_generation_v2(item)) {
  36. if (btrfs_root_generation_v2(item) != 0) {
  37. btrfs_warn(eb->fs_info,
  38. "mismatching generation and generation_v2 found in root item. This root was probably mounted with an older kernel. Resetting all new fields.");
  39. }
  40. need_reset = 1;
  41. }
  42. if (need_reset) {
  43. /* Clear all members from generation_v2 onwards. */
  44. memset_startat(item, 0, generation_v2);
  45. generate_random_guid(item->uuid);
  46. }
  47. }
  48. /*
  49. * Lookup the root by the key.
  50. *
  51. * root: the root of the root tree
  52. * search_key: the key to search
  53. * path: the path we search
  54. * root_item: the root item of the tree we look for
  55. * root_key: the root key of the tree we look for
  56. *
  57. * If ->offset of 'search_key' is -1ULL, it means we are not sure the offset
  58. * of the search key, just lookup the root with the highest offset for a
  59. * given objectid.
  60. *
  61. * If we find something return 0, otherwise > 0, < 0 on error.
  62. */
  63. int btrfs_find_root(struct btrfs_root *root, const struct btrfs_key *search_key,
  64. struct btrfs_path *path, struct btrfs_root_item *root_item,
  65. struct btrfs_key *root_key)
  66. {
  67. struct btrfs_key found_key;
  68. struct extent_buffer *l;
  69. int ret;
  70. int slot;
  71. ret = btrfs_search_slot(NULL, root, search_key, path, 0, 0);
  72. if (ret < 0)
  73. return ret;
  74. if (search_key->offset != -1ULL) { /* the search key is exact */
  75. if (ret > 0)
  76. goto out;
  77. } else {
  78. /*
  79. * Key with offset -1 found, there would have to exist a root
  80. * with such id, but this is out of the valid range.
  81. */
  82. if (unlikely(ret == 0)) {
  83. ret = -EUCLEAN;
  84. goto out;
  85. }
  86. if (path->slots[0] == 0)
  87. goto out;
  88. path->slots[0]--;
  89. ret = 0;
  90. }
  91. l = path->nodes[0];
  92. slot = path->slots[0];
  93. btrfs_item_key_to_cpu(l, &found_key, slot);
  94. if (found_key.objectid != search_key->objectid ||
  95. found_key.type != BTRFS_ROOT_ITEM_KEY) {
  96. ret = 1;
  97. goto out;
  98. }
  99. if (root_item)
  100. btrfs_read_root_item(l, slot, root_item);
  101. if (root_key)
  102. memcpy(root_key, &found_key, sizeof(found_key));
  103. out:
  104. btrfs_release_path(path);
  105. return ret;
  106. }
  107. void btrfs_set_root_node(struct btrfs_root_item *item,
  108. struct extent_buffer *node)
  109. {
  110. btrfs_set_root_bytenr(item, node->start);
  111. btrfs_set_root_level(item, btrfs_header_level(node));
  112. btrfs_set_root_generation(item, btrfs_header_generation(node));
  113. }
  114. /*
  115. * copy the data in 'item' into the btree
  116. */
  117. int btrfs_update_root(struct btrfs_trans_handle *trans, struct btrfs_root
  118. *root, struct btrfs_key *key, struct btrfs_root_item
  119. *item)
  120. {
  121. struct btrfs_fs_info *fs_info = root->fs_info;
  122. BTRFS_PATH_AUTO_FREE(path);
  123. struct extent_buffer *l;
  124. int ret;
  125. int slot;
  126. unsigned long ptr;
  127. u32 old_len;
  128. path = btrfs_alloc_path();
  129. if (!path)
  130. return -ENOMEM;
  131. ret = btrfs_search_slot(trans, root, key, path, 0, 1);
  132. if (ret < 0)
  133. return ret;
  134. if (unlikely(ret > 0)) {
  135. btrfs_crit(fs_info,
  136. "unable to find root key " BTRFS_KEY_FMT " in tree %llu",
  137. BTRFS_KEY_FMT_VALUE(key), btrfs_root_id(root));
  138. ret = -EUCLEAN;
  139. btrfs_abort_transaction(trans, ret);
  140. return ret;
  141. }
  142. l = path->nodes[0];
  143. slot = path->slots[0];
  144. ptr = btrfs_item_ptr_offset(l, slot);
  145. old_len = btrfs_item_size(l, slot);
  146. /*
  147. * If this is the first time we update the root item which originated
  148. * from an older kernel, we need to enlarge the item size to make room
  149. * for the added fields.
  150. */
  151. if (old_len < sizeof(*item)) {
  152. btrfs_release_path(path);
  153. ret = btrfs_search_slot(trans, root, key, path,
  154. -1, 1);
  155. if (unlikely(ret < 0)) {
  156. btrfs_abort_transaction(trans, ret);
  157. return ret;
  158. }
  159. ret = btrfs_del_item(trans, root, path);
  160. if (unlikely(ret < 0)) {
  161. btrfs_abort_transaction(trans, ret);
  162. return ret;
  163. }
  164. btrfs_release_path(path);
  165. ret = btrfs_insert_empty_item(trans, root, path,
  166. key, sizeof(*item));
  167. if (unlikely(ret < 0)) {
  168. btrfs_abort_transaction(trans, ret);
  169. return ret;
  170. }
  171. l = path->nodes[0];
  172. slot = path->slots[0];
  173. ptr = btrfs_item_ptr_offset(l, slot);
  174. }
  175. /*
  176. * Update generation_v2 so at the next mount we know the new root
  177. * fields are valid.
  178. */
  179. btrfs_set_root_generation_v2(item, btrfs_root_generation(item));
  180. write_extent_buffer(l, item, ptr, sizeof(*item));
  181. return ret;
  182. }
  183. int btrfs_insert_root(struct btrfs_trans_handle *trans, struct btrfs_root *root,
  184. const struct btrfs_key *key, struct btrfs_root_item *item)
  185. {
  186. /*
  187. * Make sure generation v1 and v2 match. See update_root for details.
  188. */
  189. btrfs_set_root_generation_v2(item, btrfs_root_generation(item));
  190. return btrfs_insert_item(trans, root, key, item, sizeof(*item));
  191. }
  192. int btrfs_find_orphan_roots(struct btrfs_fs_info *fs_info)
  193. {
  194. struct btrfs_root *tree_root = fs_info->tree_root;
  195. struct extent_buffer *leaf;
  196. BTRFS_PATH_AUTO_FREE(path);
  197. struct btrfs_key key;
  198. struct btrfs_root *root;
  199. path = btrfs_alloc_path();
  200. if (!path)
  201. return -ENOMEM;
  202. key.objectid = BTRFS_ORPHAN_OBJECTID;
  203. key.type = BTRFS_ORPHAN_ITEM_KEY;
  204. key.offset = 0;
  205. while (1) {
  206. u64 root_objectid;
  207. int ret;
  208. ret = btrfs_search_slot(NULL, tree_root, &key, path, 0, 0);
  209. if (ret < 0)
  210. return ret;
  211. leaf = path->nodes[0];
  212. if (path->slots[0] >= btrfs_header_nritems(leaf)) {
  213. ret = btrfs_next_leaf(tree_root, path);
  214. if (ret < 0)
  215. return ret;
  216. else if (ret > 0)
  217. return 0;
  218. leaf = path->nodes[0];
  219. }
  220. btrfs_item_key_to_cpu(leaf, &key, path->slots[0]);
  221. btrfs_release_path(path);
  222. if (key.objectid != BTRFS_ORPHAN_OBJECTID ||
  223. key.type != BTRFS_ORPHAN_ITEM_KEY)
  224. return 0;
  225. root_objectid = key.offset;
  226. key.offset++;
  227. root = btrfs_get_fs_root(fs_info, root_objectid, false);
  228. ret = PTR_ERR_OR_ZERO(root);
  229. if (ret && ret != -ENOENT) {
  230. return ret;
  231. } else if (ret == -ENOENT) {
  232. struct btrfs_trans_handle *trans;
  233. trans = btrfs_join_transaction(tree_root);
  234. if (IS_ERR(trans)) {
  235. ret = PTR_ERR(trans);
  236. btrfs_err(fs_info,
  237. "failed to join transaction to delete orphan item: %d",
  238. ret);
  239. return ret;
  240. }
  241. ret = btrfs_del_orphan_item(trans, tree_root, root_objectid);
  242. btrfs_end_transaction(trans);
  243. if (ret) {
  244. btrfs_err(fs_info,
  245. "failed to delete root orphan item: %d", ret);
  246. return ret;
  247. }
  248. continue;
  249. }
  250. WARN_ON(!test_bit(BTRFS_ROOT_ORPHAN_ITEM_INSERTED, &root->state));
  251. if (btrfs_root_refs(&root->root_item) == 0) {
  252. struct btrfs_key drop_key;
  253. btrfs_disk_key_to_cpu(&drop_key, &root->root_item.drop_progress);
  254. /*
  255. * If we have a non-zero drop_progress then we know we
  256. * made it partly through deleting this snapshot, and
  257. * thus we need to make sure we block any balance from
  258. * happening until this snapshot is completely dropped.
  259. */
  260. if (drop_key.objectid != 0 || drop_key.type != 0 ||
  261. drop_key.offset != 0) {
  262. set_bit(BTRFS_FS_UNFINISHED_DROPS, &fs_info->flags);
  263. set_bit(BTRFS_ROOT_UNFINISHED_DROP, &root->state);
  264. }
  265. set_bit(BTRFS_ROOT_DEAD_TREE, &root->state);
  266. btrfs_add_dead_root(root);
  267. }
  268. btrfs_put_root(root);
  269. }
  270. return 0;
  271. }
  272. /* drop the root item for 'key' from the tree root */
  273. int btrfs_del_root(struct btrfs_trans_handle *trans,
  274. const struct btrfs_key *key)
  275. {
  276. struct btrfs_root *root = trans->fs_info->tree_root;
  277. BTRFS_PATH_AUTO_FREE(path);
  278. int ret;
  279. path = btrfs_alloc_path();
  280. if (!path)
  281. return -ENOMEM;
  282. ret = btrfs_search_slot(trans, root, key, path, -1, 1);
  283. if (ret < 0)
  284. return ret;
  285. if (unlikely(ret > 0))
  286. /* The root must exist but we did not find it by the key. */
  287. return -EUCLEAN;
  288. return btrfs_del_item(trans, root, path);
  289. }
  290. int btrfs_del_root_ref(struct btrfs_trans_handle *trans, u64 root_id,
  291. u64 ref_id, u64 dirid, u64 *sequence,
  292. const struct fscrypt_str *name)
  293. {
  294. struct btrfs_root *tree_root = trans->fs_info->tree_root;
  295. BTRFS_PATH_AUTO_FREE(path);
  296. struct btrfs_root_ref *ref;
  297. struct extent_buffer *leaf;
  298. struct btrfs_key key;
  299. unsigned long ptr;
  300. int ret;
  301. path = btrfs_alloc_path();
  302. if (!path)
  303. return -ENOMEM;
  304. key.objectid = root_id;
  305. key.type = BTRFS_ROOT_BACKREF_KEY;
  306. key.offset = ref_id;
  307. again:
  308. ret = btrfs_search_slot(trans, tree_root, &key, path, -1, 1);
  309. if (ret < 0) {
  310. return ret;
  311. } else if (ret == 0) {
  312. leaf = path->nodes[0];
  313. ref = btrfs_item_ptr(leaf, path->slots[0],
  314. struct btrfs_root_ref);
  315. ptr = (unsigned long)(ref + 1);
  316. if ((btrfs_root_ref_dirid(leaf, ref) != dirid) ||
  317. (btrfs_root_ref_name_len(leaf, ref) != name->len) ||
  318. memcmp_extent_buffer(leaf, name->name, ptr, name->len))
  319. return -ENOENT;
  320. *sequence = btrfs_root_ref_sequence(leaf, ref);
  321. ret = btrfs_del_item(trans, tree_root, path);
  322. if (ret)
  323. return ret;
  324. } else {
  325. return -ENOENT;
  326. }
  327. if (key.type == BTRFS_ROOT_BACKREF_KEY) {
  328. btrfs_release_path(path);
  329. key.objectid = ref_id;
  330. key.type = BTRFS_ROOT_REF_KEY;
  331. key.offset = root_id;
  332. goto again;
  333. }
  334. return ret;
  335. }
  336. /*
  337. * add a btrfs_root_ref item. type is either BTRFS_ROOT_REF_KEY
  338. * or BTRFS_ROOT_BACKREF_KEY.
  339. *
  340. * The dirid, sequence, name and name_len refer to the directory entry
  341. * that is referencing the root.
  342. *
  343. * For a forward ref, the root_id is the id of the tree referencing
  344. * the root and ref_id is the id of the subvol or snapshot.
  345. *
  346. * For a back ref the root_id is the id of the subvol or snapshot and
  347. * ref_id is the id of the tree referencing it.
  348. *
  349. * Will return 0, -ENOMEM, or anything from the CoW path
  350. */
  351. int btrfs_add_root_ref(struct btrfs_trans_handle *trans, u64 root_id,
  352. u64 ref_id, u64 dirid, u64 sequence,
  353. const struct fscrypt_str *name)
  354. {
  355. struct btrfs_root *tree_root = trans->fs_info->tree_root;
  356. struct btrfs_key key;
  357. int ret;
  358. BTRFS_PATH_AUTO_FREE(path);
  359. struct btrfs_root_ref *ref;
  360. struct extent_buffer *leaf;
  361. unsigned long ptr;
  362. path = btrfs_alloc_path();
  363. if (!path)
  364. return -ENOMEM;
  365. key.objectid = root_id;
  366. key.type = BTRFS_ROOT_BACKREF_KEY;
  367. key.offset = ref_id;
  368. again:
  369. ret = btrfs_insert_empty_item(trans, tree_root, path, &key,
  370. sizeof(*ref) + name->len);
  371. if (unlikely(ret)) {
  372. btrfs_abort_transaction(trans, ret);
  373. return ret;
  374. }
  375. leaf = path->nodes[0];
  376. ref = btrfs_item_ptr(leaf, path->slots[0], struct btrfs_root_ref);
  377. btrfs_set_root_ref_dirid(leaf, ref, dirid);
  378. btrfs_set_root_ref_sequence(leaf, ref, sequence);
  379. btrfs_set_root_ref_name_len(leaf, ref, name->len);
  380. ptr = (unsigned long)(ref + 1);
  381. write_extent_buffer(leaf, name->name, ptr, name->len);
  382. if (key.type == BTRFS_ROOT_BACKREF_KEY) {
  383. btrfs_release_path(path);
  384. key.objectid = ref_id;
  385. key.type = BTRFS_ROOT_REF_KEY;
  386. key.offset = root_id;
  387. goto again;
  388. }
  389. return 0;
  390. }
  391. /*
  392. * Old btrfs forgets to init root_item->flags and root_item->byte_limit
  393. * for subvolumes. To work around this problem, we steal a bit from
  394. * root_item->inode_item->flags, and use it to indicate if those fields
  395. * have been properly initialized.
  396. */
  397. void btrfs_check_and_init_root_item(struct btrfs_root_item *root_item)
  398. {
  399. u64 inode_flags = btrfs_stack_inode_flags(&root_item->inode);
  400. if (!(inode_flags & BTRFS_INODE_ROOT_ITEM_INIT)) {
  401. inode_flags |= BTRFS_INODE_ROOT_ITEM_INIT;
  402. btrfs_set_stack_inode_flags(&root_item->inode, inode_flags);
  403. btrfs_set_root_flags(root_item, 0);
  404. btrfs_set_root_limit(root_item, 0);
  405. }
  406. }
  407. void btrfs_update_root_times(struct btrfs_trans_handle *trans,
  408. struct btrfs_root *root)
  409. {
  410. struct btrfs_root_item *item = &root->root_item;
  411. struct timespec64 ct;
  412. ktime_get_real_ts64(&ct);
  413. spin_lock(&root->root_item_lock);
  414. btrfs_set_root_ctransid(item, trans->transid);
  415. btrfs_set_stack_timespec_sec(&item->ctime, ct.tv_sec);
  416. btrfs_set_stack_timespec_nsec(&item->ctime, ct.tv_nsec);
  417. spin_unlock(&root->root_item_lock);
  418. }
  419. /*
  420. * Reserve space for subvolume operation.
  421. *
  422. * root: the root of the parent directory
  423. * rsv: block reservation
  424. * items: the number of items that we need do reservation
  425. * use_global_rsv: allow fallback to the global block reservation
  426. *
  427. * This function is used to reserve the space for snapshot/subvolume
  428. * creation and deletion. Those operations are different with the
  429. * common file/directory operations, they change two fs/file trees
  430. * and root tree, the number of items that the qgroup reserves is
  431. * different with the free space reservation. So we can not use
  432. * the space reservation mechanism in start_transaction().
  433. */
  434. int btrfs_subvolume_reserve_metadata(struct btrfs_root *root,
  435. struct btrfs_block_rsv *rsv, int items,
  436. bool use_global_rsv)
  437. {
  438. u64 qgroup_num_bytes = 0;
  439. u64 num_bytes;
  440. int ret;
  441. struct btrfs_fs_info *fs_info = root->fs_info;
  442. struct btrfs_block_rsv *global_rsv = &fs_info->global_block_rsv;
  443. if (btrfs_qgroup_enabled(fs_info)) {
  444. /* One for parent inode, two for dir entries */
  445. qgroup_num_bytes = 3 * fs_info->nodesize;
  446. ret = btrfs_qgroup_reserve_meta_prealloc(root,
  447. qgroup_num_bytes, true,
  448. false);
  449. if (ret)
  450. return ret;
  451. }
  452. num_bytes = btrfs_calc_insert_metadata_size(fs_info, items);
  453. rsv->space_info = btrfs_find_space_info(fs_info,
  454. BTRFS_BLOCK_GROUP_METADATA);
  455. ret = btrfs_block_rsv_add(fs_info, rsv, num_bytes,
  456. BTRFS_RESERVE_FLUSH_ALL);
  457. if (ret == -ENOSPC && use_global_rsv)
  458. ret = btrfs_block_rsv_migrate(global_rsv, rsv, num_bytes, true);
  459. if (ret && qgroup_num_bytes)
  460. btrfs_qgroup_free_meta_prealloc(root, qgroup_num_bytes);
  461. if (!ret) {
  462. spin_lock(&rsv->lock);
  463. rsv->qgroup_rsv_reserved += qgroup_num_bytes;
  464. spin_unlock(&rsv->lock);
  465. }
  466. return ret;
  467. }