1 // SPDX-License-Identifier: GPL-2.0-only
3 * Overlayfs NFS export support.
5 * Amir Goldstein <amir73il@gmail.com>
7 * Copyright (C) 2017-2018 CTERA Networks. All Rights Reserved.
11 #include <linux/cred.h>
12 #include <linux/mount.h>
13 #include <linux/namei.h>
14 #include <linux/xattr.h>
15 #include <linux/exportfs.h>
16 #include <linux/ratelimit.h>
17 #include "overlayfs.h"
19 static int ovl_encode_maybe_copy_up(struct dentry
*dentry
)
23 if (ovl_dentry_upper(dentry
))
26 err
= ovl_copy_up(dentry
);
28 pr_warn_ratelimited("failed to copy up on encode (%pd2, err=%i)\n",
36 * Before encoding a non-upper directory file handle from real layer N, we need
37 * to check if it will be possible to reconnect an overlay dentry from the real
38 * lower decoded dentry. This is done by following the overlay ancestry up to a
39 * "layer N connected" ancestor and verifying that all parents along the way are
40 * "layer N connectable". If an ancestor that is NOT "layer N connectable" is
41 * found, we need to copy up an ancestor, which is "layer N connectable", thus
42 * making that ancestor "layer N connected". For example:
47 * The overlay dentry /a is NOT "layer 2 connectable", because if dir /a is
48 * copied up and renamed, upper dir /a will be indexed by lower dir /a from
49 * layer 1. The dir /a from layer 2 will never be indexed, so the algorithm (*)
50 * in ovl_lookup_real_ancestor() will not be able to lookup a connected overlay
51 * dentry from the connected lower dentry /a/b/c.
53 * To avoid this problem on decode time, we need to copy up an ancestor of
54 * /a/b/c, which is "layer 2 connectable", on encode time. That ancestor is
55 * /a/b. After copy up (and index) of /a/b, it will become "layer 2 connected"
56 * and when the time comes to decode the file handle from lower dentry /a/b/c,
57 * ovl_lookup_real_ancestor() will find the indexed ancestor /a/b and decoding
58 * a connected overlay dentry will be accomplished.
60 * (*) the algorithm in ovl_lookup_real_ancestor() can be improved to lookup an
61 * entry /a in the lower layers above layer N and find the indexed dir /a from
62 * layer 1. If that improvement is made, then the check for "layer N connected"
63 * will need to verify there are no redirects in lower layers above N. In the
64 * example above, /a will be "layer 2 connectable". However, if layer 2 dir /a
65 * is a target of a layer 1 redirect, then /a will NOT be "layer 2 connectable":
67 * layer 1: /A (redirect = /a)
71 /* Return the lowest layer for encoding a connectable file handle */
72 static int ovl_connectable_layer(struct dentry
*dentry
)
74 struct ovl_entry
*oe
= OVL_E(dentry
);
76 /* We can get overlay root from root of any layer */
77 if (dentry
== dentry
->d_sb
->s_root
)
78 return ovl_numlower(oe
);
81 * If it's an unindexed merge dir, then it's not connectable with any
84 if (ovl_dentry_upper(dentry
) &&
85 !ovl_test_flag(OVL_INDEX
, d_inode(dentry
)))
88 /* We can get upper/overlay path from indexed/lower dentry */
89 return ovl_lowerstack(oe
)->layer
->idx
;
93 * @dentry is "connected" if all ancestors up to root or a "connected" ancestor
94 * have the same uppermost lower layer as the origin's layer. We may need to
95 * copy up a "connectable" ancestor to make it "connected". A "connected" dentry
96 * cannot become non "connected", so cache positive result in dentry flags.
98 * Return the connected origin layer or < 0 on error.
100 static int ovl_connect_layer(struct dentry
*dentry
)
102 struct dentry
*next
, *parent
= NULL
;
103 struct ovl_entry
*oe
= OVL_E(dentry
);
107 if (WARN_ON(dentry
== dentry
->d_sb
->s_root
) ||
108 WARN_ON(!ovl_dentry_lower(dentry
)))
111 origin_layer
= ovl_lowerstack(oe
)->layer
->idx
;
112 if (ovl_dentry_test_flag(OVL_E_CONNECTED
, dentry
))
115 /* Find the topmost origin layer connectable ancestor of @dentry */
118 parent
= dget_parent(next
);
119 if (WARN_ON(parent
== next
)) {
125 * If @parent is not origin layer connectable, then copy up
126 * @next which is origin layer connectable and we are done.
128 if (ovl_connectable_layer(parent
) < origin_layer
) {
129 err
= ovl_encode_maybe_copy_up(next
);
133 /* If @parent is connected or indexed we are done */
134 if (ovl_dentry_test_flag(OVL_E_CONNECTED
, parent
) ||
135 ovl_test_flag(OVL_INDEX
, d_inode(parent
)))
146 ovl_dentry_set_flag(OVL_E_CONNECTED
, dentry
);
148 return err
?: origin_layer
;
152 * We only need to encode origin if there is a chance that the same object was
153 * encoded pre copy up and then we need to stay consistent with the same
154 * encoding also after copy up. If non-pure upper is not indexed, then it was
155 * copied up before NFS export was enabled. In that case we don't need to worry
156 * about staying consistent with pre copy up encoding and we encode an upper
157 * file handle. Overlay root dentry is a private case of non-indexed upper.
159 * The following table summarizes the different file handle encodings used for
160 * different overlay object types:
162 * Object type | Encoding
163 * --------------------------------
165 * Non-indexed upper | U
166 * Indexed upper | L (*)
169 * U = upper file handle
170 * L = lower file handle
172 * (*) Decoding a connected overlay dir from real lower dentry is not always
173 * possible when there are redirects in lower layers and non-indexed merge dirs.
174 * To mitigate those case, we may copy up the lower dir ancestor before encode
175 * of a decodable file handle for non-upper dir.
177 * Return 0 for upper file handle, > 0 for lower file handle or < 0 on error.
179 static int ovl_check_encode_origin(struct dentry
*dentry
)
181 struct ovl_fs
*ofs
= OVL_FS(dentry
->d_sb
);
182 bool decodable
= ofs
->config
.nfs_export
;
184 /* No upper layer? */
185 if (!ovl_upper_mnt(ofs
))
188 /* Lower file handle for non-upper non-decodable */
189 if (!ovl_dentry_upper(dentry
) && !decodable
)
192 /* Upper file handle for pure upper */
193 if (!ovl_dentry_lower(dentry
))
197 * Root is never indexed, so if there's an upper layer, encode upper for
200 if (dentry
== dentry
->d_sb
->s_root
)
204 * Upper decodable file handle for non-indexed upper.
206 if (ovl_dentry_upper(dentry
) && decodable
&&
207 !ovl_test_flag(OVL_INDEX
, d_inode(dentry
)))
211 * Decoding a merge dir, whose origin's ancestor is under a redirected
212 * lower dir or under a non-indexed upper is not always possible.
213 * ovl_connect_layer() will try to make origin's layer "connected" by
214 * copying up a "connectable" ancestor.
216 if (d_is_dir(dentry
) && decodable
)
217 return ovl_connect_layer(dentry
);
219 /* Lower file handle for indexed and non-upper dir/non-dir */
223 static int ovl_dentry_to_fid(struct ovl_fs
*ofs
, struct dentry
*dentry
,
224 u32
*fid
, int buflen
)
226 struct ovl_fh
*fh
= NULL
;
231 * Check if we should encode a lower or upper file handle and maybe
232 * copy up an ancestor to make lower file handle connectable.
234 err
= enc_lower
= ovl_check_encode_origin(dentry
);
238 /* Encode an upper or lower file handle */
239 fh
= ovl_encode_real_fh(ofs
, enc_lower
? ovl_dentry_lower(dentry
) :
240 ovl_dentry_upper(dentry
), !enc_lower
);
244 len
= OVL_FH_LEN(fh
);
246 memcpy(fid
, fh
, len
);
254 pr_warn_ratelimited("failed to encode file handle (%pd2, err=%i)\n",
259 static int ovl_encode_fh(struct inode
*inode
, u32
*fid
, int *max_len
,
260 struct inode
*parent
)
262 struct ovl_fs
*ofs
= OVL_FS(inode
->i_sb
);
263 struct dentry
*dentry
;
264 int bytes
, buflen
= *max_len
<< 2;
266 /* TODO: encode connectable file handles */
268 return FILEID_INVALID
;
270 dentry
= d_find_any_alias(inode
);
272 return FILEID_INVALID
;
274 bytes
= ovl_dentry_to_fid(ofs
, dentry
, fid
, buflen
);
277 return FILEID_INVALID
;
279 *max_len
= bytes
>> 2;
281 return FILEID_INVALID
;
283 return OVL_FILEID_V1
;
287 * Find or instantiate an overlay dentry from real dentries and index.
289 static struct dentry
*ovl_obtain_alias(struct super_block
*sb
,
290 struct dentry
*upper_alias
,
291 struct ovl_path
*lowerpath
,
292 struct dentry
*index
)
294 struct dentry
*lower
= lowerpath
? lowerpath
->dentry
: NULL
;
295 struct dentry
*upper
= upper_alias
?: index
;
296 struct inode
*inode
= NULL
;
297 struct ovl_entry
*oe
;
298 struct ovl_inode_params oip
= {
302 /* We get overlay directory dentries with ovl_lookup_real() */
303 if (d_is_dir(upper
?: lower
))
304 return ERR_PTR(-EIO
);
306 oe
= ovl_alloc_entry(!!lower
);
308 return ERR_PTR(-ENOMEM
);
310 oip
.upperdentry
= dget(upper
);
312 ovl_lowerstack(oe
)->dentry
= dget(lower
);
313 ovl_lowerstack(oe
)->layer
= lowerpath
->layer
;
316 inode
= ovl_get_inode(sb
, &oip
);
320 return ERR_CAST(inode
);
324 ovl_set_flag(OVL_UPPERDATA
, inode
);
326 return d_obtain_alias(inode
);
329 /* Get the upper or lower dentry in stack whose on layer @idx */
330 static struct dentry
*ovl_dentry_real_at(struct dentry
*dentry
, int idx
)
332 struct ovl_entry
*oe
= OVL_E(dentry
);
333 struct ovl_path
*lowerstack
= ovl_lowerstack(oe
);
337 return ovl_dentry_upper(dentry
);
339 for (i
= 0; i
< ovl_numlower(oe
); i
++) {
340 if (lowerstack
[i
].layer
->idx
== idx
)
341 return lowerstack
[i
].dentry
;
348 * Lookup a child overlay dentry to get a connected overlay dentry whose real
349 * dentry is @real. If @real is on upper layer, we lookup a child overlay
350 * dentry with the same name as the real dentry. Otherwise, we need to consult
353 static struct dentry
*ovl_lookup_real_one(struct dentry
*connected
,
355 const struct ovl_layer
*layer
)
357 struct inode
*dir
= d_inode(connected
);
358 struct dentry
*this, *parent
= NULL
;
359 struct name_snapshot name
;
363 * Lookup child overlay dentry by real name. The dir mutex protects us
364 * from racing with overlay rename. If the overlay dentry that is above
365 * real has already been moved to a parent that is not under the
366 * connected overlay dir, we return -ECHILD and restart the lookup of
367 * connected real path from the top.
369 inode_lock_nested(dir
, I_MUTEX_PARENT
);
371 parent
= dget_parent(real
);
372 if (ovl_dentry_real_at(connected
, layer
->idx
) != parent
)
376 * We also need to take a snapshot of real dentry name to protect us
377 * from racing with underlying layer rename. In this case, we don't
378 * care about returning ESTALE, only from dereferencing a free name
379 * pointer because we hold no lock on the real dentry.
381 take_dentry_name_snapshot(&name
, real
);
383 * No idmap handling here: it's an internal lookup. Could skip
384 * permission checking altogether, but for now just use non-idmap
387 this = lookup_one_len(name
.name
.name
, connected
, name
.name
.len
);
388 release_dentry_name_snapshot(&name
);
392 } else if (!this || !this->d_inode
) {
396 } else if (ovl_dentry_real_at(this, layer
->idx
) != real
) {
408 pr_warn_ratelimited("failed to lookup one by real (%pd2, layer=%d, connected=%pd2, err=%i)\n",
409 real
, layer
->idx
, connected
, err
);
414 static struct dentry
*ovl_lookup_real(struct super_block
*sb
,
416 const struct ovl_layer
*layer
);
419 * Lookup an indexed or hashed overlay dentry by real inode.
421 static struct dentry
*ovl_lookup_real_inode(struct super_block
*sb
,
423 const struct ovl_layer
*layer
)
425 struct ovl_fs
*ofs
= OVL_FS(sb
);
426 struct dentry
*index
= NULL
;
427 struct dentry
*this = NULL
;
431 * Decoding upper dir from index is expensive, so first try to lookup
432 * overlay dentry in inode/dcache.
434 inode
= ovl_lookup_inode(sb
, real
, !layer
->idx
);
436 return ERR_CAST(inode
);
438 this = d_find_any_alias(inode
);
443 * For decoded lower dir file handle, lookup index by origin to check
444 * if lower dir was copied up and and/or removed.
446 if (!this && layer
->idx
&& ovl_indexdir(sb
) && !WARN_ON(!d_is_dir(real
))) {
447 index
= ovl_lookup_index(ofs
, NULL
, real
, false);
452 /* Get connected upper overlay dir from index */
454 struct dentry
*upper
= ovl_index_upper(ofs
, index
, true);
457 if (IS_ERR_OR_NULL(upper
))
461 * ovl_lookup_real() in lower layer may call recursively once to
462 * ovl_lookup_real() in upper layer. The first level call walks
463 * back lower parents to the topmost indexed parent. The second
464 * recursive call walks back from indexed upper to the topmost
465 * connected/hashed upper parent (or up to root).
467 this = ovl_lookup_real(sb
, upper
, &ofs
->layers
[0]);
471 if (IS_ERR_OR_NULL(this))
474 if (ovl_dentry_real_at(this, layer
->idx
) != real
) {
476 this = ERR_PTR(-EIO
);
483 * Lookup an indexed or hashed overlay dentry, whose real dentry is an
486 static struct dentry
*ovl_lookup_real_ancestor(struct super_block
*sb
,
488 const struct ovl_layer
*layer
)
490 struct dentry
*next
, *parent
= NULL
;
491 struct dentry
*ancestor
= ERR_PTR(-EIO
);
493 if (real
== layer
->mnt
->mnt_root
)
494 return dget(sb
->s_root
);
496 /* Find the topmost indexed or hashed ancestor */
499 parent
= dget_parent(next
);
502 * Lookup a matching overlay dentry in inode/dentry
503 * cache or in index by real inode.
505 ancestor
= ovl_lookup_real_inode(sb
, next
, layer
);
509 if (parent
== layer
->mnt
->mnt_root
) {
510 ancestor
= dget(sb
->s_root
);
515 * If @real has been moved out of the layer root directory,
516 * we will eventully hit the real fs root. This cannot happen
517 * by legit overlay rename, so we return error in that case.
519 if (parent
== next
) {
520 ancestor
= ERR_PTR(-EXDEV
);
535 * Lookup a connected overlay dentry whose real dentry is @real.
536 * If @real is on upper layer, we lookup a child overlay dentry with the same
537 * path the real dentry. Otherwise, we need to consult index for lookup.
539 static struct dentry
*ovl_lookup_real(struct super_block
*sb
,
541 const struct ovl_layer
*layer
)
543 struct dentry
*connected
;
546 connected
= ovl_lookup_real_ancestor(sb
, real
, layer
);
547 if (IS_ERR(connected
))
551 struct dentry
*next
, *this;
552 struct dentry
*parent
= NULL
;
553 struct dentry
*real_connected
= ovl_dentry_real_at(connected
,
556 if (real_connected
== real
)
559 /* Find the topmost dentry not yet connected */
562 parent
= dget_parent(next
);
564 if (parent
== real_connected
)
568 * If real has been moved out of 'real_connected',
569 * we will not find 'real_connected' and hit the layer
570 * root. In that case, we need to restart connecting.
571 * This game can go on forever in the worst case. We
572 * may want to consider taking s_vfs_rename_mutex if
573 * this happens more than once.
575 if (parent
== layer
->mnt
->mnt_root
) {
577 connected
= dget(sb
->s_root
);
582 * If real file has been moved out of the layer root
583 * directory, we will eventully hit the real fs root.
584 * This cannot happen by legit overlay rename, so we
585 * return error in that case.
587 if (parent
== next
) {
597 this = ovl_lookup_real_one(connected
, next
, layer
);
602 * Lookup of child in overlay can fail when racing with
603 * overlay rename of child away from 'connected' parent.
604 * In this case, we need to restart the lookup from the
605 * top, because we cannot trust that 'real_connected' is
606 * still an ancestor of 'real'. There is a good chance
607 * that the renamed overlay ancestor is now in cache, so
608 * ovl_lookup_real_ancestor() will find it and we can
609 * continue to connect exactly from where lookup failed.
611 if (err
== -ECHILD
) {
612 this = ovl_lookup_real_ancestor(sb
, real
,
614 err
= PTR_ERR_OR_ZERO(this);
632 pr_warn_ratelimited("failed to lookup by real (%pd2, layer=%d, connected=%pd2, err=%i)\n",
633 real
, layer
->idx
, connected
, err
);
639 * Get an overlay dentry from upper/lower real dentries and index.
641 static struct dentry
*ovl_get_dentry(struct super_block
*sb
,
642 struct dentry
*upper
,
643 struct ovl_path
*lowerpath
,
644 struct dentry
*index
)
646 struct ovl_fs
*ofs
= OVL_FS(sb
);
647 const struct ovl_layer
*layer
= upper
? &ofs
->layers
[0] : lowerpath
->layer
;
648 struct dentry
*real
= upper
?: (index
?: lowerpath
->dentry
);
651 * Obtain a disconnected overlay dentry from a non-dir real dentry
655 return ovl_obtain_alias(sb
, upper
, lowerpath
, index
);
657 /* Removed empty directory? */
658 if ((real
->d_flags
& DCACHE_DISCONNECTED
) || d_unhashed(real
))
659 return ERR_PTR(-ENOENT
);
662 * If real dentry is connected and hashed, get a connected overlay
663 * dentry whose real dentry is @real.
665 return ovl_lookup_real(sb
, real
, layer
);
668 static struct dentry
*ovl_upper_fh_to_d(struct super_block
*sb
,
671 struct ovl_fs
*ofs
= OVL_FS(sb
);
672 struct dentry
*dentry
;
673 struct dentry
*upper
;
675 if (!ovl_upper_mnt(ofs
))
676 return ERR_PTR(-EACCES
);
678 upper
= ovl_decode_real_fh(ofs
, fh
, ovl_upper_mnt(ofs
), true);
679 if (IS_ERR_OR_NULL(upper
))
682 dentry
= ovl_get_dentry(sb
, upper
, NULL
, NULL
);
688 static struct dentry
*ovl_lower_fh_to_d(struct super_block
*sb
,
691 struct ovl_fs
*ofs
= OVL_FS(sb
);
692 struct ovl_path origin
= { };
693 struct ovl_path
*stack
= &origin
;
694 struct dentry
*dentry
= NULL
;
695 struct dentry
*index
= NULL
;
699 /* First lookup overlay inode in inode cache by origin fh */
700 err
= ovl_check_origin_fh(ofs
, fh
, false, NULL
, &stack
);
704 if (!d_is_dir(origin
.dentry
) ||
705 !(origin
.dentry
->d_flags
& DCACHE_DISCONNECTED
)) {
706 inode
= ovl_lookup_inode(sb
, origin
.dentry
, false);
707 err
= PTR_ERR(inode
);
711 dentry
= d_find_any_alias(inode
);
718 /* Then lookup indexed upper/whiteout by origin fh */
719 if (ovl_indexdir(sb
)) {
720 index
= ovl_get_index_fh(ofs
, fh
);
721 err
= PTR_ERR(index
);
728 /* Then try to get a connected upper dir by index */
729 if (index
&& d_is_dir(index
)) {
730 struct dentry
*upper
= ovl_index_upper(ofs
, index
, true);
732 err
= PTR_ERR(upper
);
733 if (IS_ERR_OR_NULL(upper
))
736 dentry
= ovl_get_dentry(sb
, upper
, NULL
, NULL
);
741 /* Find origin.dentry again with ovl_acceptable() layer check */
742 if (d_is_dir(origin
.dentry
)) {
744 origin
.dentry
= NULL
;
745 err
= ovl_check_origin_fh(ofs
, fh
, true, NULL
, &stack
);
750 err
= ovl_verify_origin(ofs
, index
, origin
.dentry
, false);
755 /* Get a connected non-upper dir or disconnected non-dir */
756 dentry
= ovl_get_dentry(sb
, NULL
, &origin
, index
);
764 dentry
= ERR_PTR(err
);
768 static struct ovl_fh
*ovl_fid_to_fh(struct fid
*fid
, int buflen
, int fh_type
)
772 /* If on-wire inner fid is aligned - nothing to do */
773 if (fh_type
== OVL_FILEID_V1
)
774 return (struct ovl_fh
*)fid
;
776 if (fh_type
!= OVL_FILEID_V0
)
777 return ERR_PTR(-EINVAL
);
779 if (buflen
<= OVL_FH_WIRE_OFFSET
)
780 return ERR_PTR(-EINVAL
);
782 fh
= kzalloc(buflen
, GFP_KERNEL
);
784 return ERR_PTR(-ENOMEM
);
786 /* Copy unaligned inner fh into aligned buffer */
787 memcpy(fh
->buf
, fid
, buflen
- OVL_FH_WIRE_OFFSET
);
791 static struct dentry
*ovl_fh_to_dentry(struct super_block
*sb
, struct fid
*fid
,
792 int fh_len
, int fh_type
)
794 struct dentry
*dentry
= NULL
;
795 struct ovl_fh
*fh
= NULL
;
796 int len
= fh_len
<< 2;
797 unsigned int flags
= 0;
800 fh
= ovl_fid_to_fh(fid
, len
, fh_type
);
805 err
= ovl_check_fh_len(fh
, len
);
809 flags
= fh
->fb
.flags
;
810 dentry
= (flags
& OVL_FH_FLAG_PATH_UPPER
) ?
811 ovl_upper_fh_to_d(sb
, fh
) :
812 ovl_lower_fh_to_d(sb
, fh
);
813 err
= PTR_ERR(dentry
);
814 if (IS_ERR(dentry
) && err
!= -ESTALE
)
818 /* We may have needed to re-align OVL_FILEID_V0 */
819 if (!IS_ERR_OR_NULL(fh
) && fh
!= (void *)fid
)
825 pr_warn_ratelimited("failed to decode file handle (len=%d, type=%d, flags=%x, err=%i)\n",
826 fh_len
, fh_type
, flags
, err
);
827 dentry
= ERR_PTR(err
);
831 static struct dentry
*ovl_fh_to_parent(struct super_block
*sb
, struct fid
*fid
,
832 int fh_len
, int fh_type
)
834 pr_warn_ratelimited("connectable file handles not supported; use 'no_subtree_check' exportfs option.\n");
835 return ERR_PTR(-EACCES
);
838 static int ovl_get_name(struct dentry
*parent
, char *name
,
839 struct dentry
*child
)
842 * ovl_fh_to_dentry() returns connected dir overlay dentries and
843 * ovl_fh_to_parent() is not implemented, so we should not get here.
849 static struct dentry
*ovl_get_parent(struct dentry
*dentry
)
852 * ovl_fh_to_dentry() returns connected dir overlay dentries, so we
853 * should not get here.
856 return ERR_PTR(-EIO
);
859 const struct export_operations ovl_export_operations
= {
860 .encode_fh
= ovl_encode_fh
,
861 .fh_to_dentry
= ovl_fh_to_dentry
,
862 .fh_to_parent
= ovl_fh_to_parent
,
863 .get_name
= ovl_get_name
,
864 .get_parent
= ovl_get_parent
,
867 /* encode_fh() encodes non-decodable file handles with nfs_export=off */
868 const struct export_operations ovl_export_fid_operations
= {
869 .encode_fh
= ovl_encode_fh
,