2 * Overlayfs NFS export support.
4 * Amir Goldstein <amir73il@gmail.com>
6 * Copyright (C) 2017-2018 CTERA Networks. All Rights Reserved.
8 * This program is free software; you can redistribute it and/or modify it
9 * under the terms of the GNU General Public License version 2 as published by
10 * the Free Software Foundation.
14 #include <linux/cred.h>
15 #include <linux/mount.h>
16 #include <linux/namei.h>
17 #include <linux/xattr.h>
18 #include <linux/exportfs.h>
19 #include <linux/ratelimit.h>
20 #include "overlayfs.h"
22 static int ovl_encode_maybe_copy_up(struct dentry
*dentry
)
26 if (ovl_dentry_upper(dentry
))
29 err
= ovl_want_write(dentry
);
31 err
= ovl_copy_up(dentry
);
32 ovl_drop_write(dentry
);
36 pr_warn_ratelimited("overlayfs: failed to copy up on encode (%pd2, err=%i)\n",
44 * Before encoding a non-upper directory file handle from real layer N, we need
45 * to check if it will be possible to reconnect an overlay dentry from the real
46 * lower decoded dentry. This is done by following the overlay ancestry up to a
47 * "layer N connected" ancestor and verifying that all parents along the way are
48 * "layer N connectable". If an ancestor that is NOT "layer N connectable" is
49 * found, we need to copy up an ancestor, which is "layer N connectable", thus
50 * making that ancestor "layer N connected". For example:
55 * The overlay dentry /a is NOT "layer 2 connectable", because if dir /a is
56 * copied up and renamed, upper dir /a will be indexed by lower dir /a from
57 * layer 1. The dir /a from layer 2 will never be indexed, so the algorithm (*)
58 * in ovl_lookup_real_ancestor() will not be able to lookup a connected overlay
59 * dentry from the connected lower dentry /a/b/c.
61 * To avoid this problem on decode time, we need to copy up an ancestor of
62 * /a/b/c, which is "layer 2 connectable", on encode time. That ancestor is
63 * /a/b. After copy up (and index) of /a/b, it will become "layer 2 connected"
64 * and when the time comes to decode the file handle from lower dentry /a/b/c,
65 * ovl_lookup_real_ancestor() will find the indexed ancestor /a/b and decoding
66 * a connected overlay dentry will be accomplished.
68 * (*) the algorithm in ovl_lookup_real_ancestor() can be improved to lookup an
69 * entry /a in the lower layers above layer N and find the indexed dir /a from
70 * layer 1. If that improvement is made, then the check for "layer N connected"
71 * will need to verify there are no redirects in lower layers above N. In the
72 * example above, /a will be "layer 2 connectable". However, if layer 2 dir /a
73 * is a target of a layer 1 redirect, then /a will NOT be "layer 2 connectable":
75 * layer 1: /A (redirect = /a)
79 /* Return the lowest layer for encoding a connectable file handle */
80 static int ovl_connectable_layer(struct dentry
*dentry
)
82 struct ovl_entry
*oe
= OVL_E(dentry
);
84 /* We can get overlay root from root of any layer */
85 if (dentry
== dentry
->d_sb
->s_root
)
89 * If it's an unindexed merge dir, then it's not connectable with any
92 if (ovl_dentry_upper(dentry
) &&
93 !ovl_test_flag(OVL_INDEX
, d_inode(dentry
)))
96 /* We can get upper/overlay path from indexed/lower dentry */
97 return oe
->lowerstack
[0].layer
->idx
;
101 * @dentry is "connected" if all ancestors up to root or a "connected" ancestor
102 * have the same uppermost lower layer as the origin's layer. We may need to
103 * copy up a "connectable" ancestor to make it "connected". A "connected" dentry
104 * cannot become non "connected", so cache positive result in dentry flags.
106 * Return the connected origin layer or < 0 on error.
108 static int ovl_connect_layer(struct dentry
*dentry
)
110 struct dentry
*next
, *parent
= NULL
;
114 if (WARN_ON(dentry
== dentry
->d_sb
->s_root
) ||
115 WARN_ON(!ovl_dentry_lower(dentry
)))
118 origin_layer
= OVL_E(dentry
)->lowerstack
[0].layer
->idx
;
119 if (ovl_dentry_test_flag(OVL_E_CONNECTED
, dentry
))
122 /* Find the topmost origin layer connectable ancestor of @dentry */
125 parent
= dget_parent(next
);
126 if (WARN_ON(parent
== next
)) {
132 * If @parent is not origin layer connectable, then copy up
133 * @next which is origin layer connectable and we are done.
135 if (ovl_connectable_layer(parent
) < origin_layer
) {
136 err
= ovl_encode_maybe_copy_up(next
);
140 /* If @parent is connected or indexed we are done */
141 if (ovl_dentry_test_flag(OVL_E_CONNECTED
, parent
) ||
142 ovl_test_flag(OVL_INDEX
, d_inode(parent
)))
153 ovl_dentry_set_flag(OVL_E_CONNECTED
, dentry
);
155 return err
?: origin_layer
;
159 * We only need to encode origin if there is a chance that the same object was
160 * encoded pre copy up and then we need to stay consistent with the same
161 * encoding also after copy up. If non-pure upper is not indexed, then it was
162 * copied up before NFS export was enabled. In that case we don't need to worry
163 * about staying consistent with pre copy up encoding and we encode an upper
164 * file handle. Overlay root dentry is a private case of non-indexed upper.
166 * The following table summarizes the different file handle encodings used for
167 * different overlay object types:
169 * Object type | Encoding
170 * --------------------------------
172 * Non-indexed upper | U
173 * Indexed upper | L (*)
176 * U = upper file handle
177 * L = lower file handle
179 * (*) Connecting an overlay dir from real lower dentry is not always
180 * possible when there are redirects in lower layers and non-indexed merge dirs.
181 * To mitigate those case, we may copy up the lower dir ancestor before encode
182 * a lower dir file handle.
184 * Return 0 for upper file handle, > 0 for lower file handle or < 0 on error.
186 static int ovl_check_encode_origin(struct dentry
*dentry
)
188 struct ovl_fs
*ofs
= dentry
->d_sb
->s_fs_info
;
190 /* Upper file handle for pure upper */
191 if (!ovl_dentry_lower(dentry
))
195 * Upper file handle for non-indexed upper.
197 * Root is never indexed, so if there's an upper layer, encode upper for
200 if (ovl_dentry_upper(dentry
) &&
201 !ovl_test_flag(OVL_INDEX
, d_inode(dentry
)))
205 * Decoding a merge dir, whose origin's ancestor is under a redirected
206 * lower dir or under a non-indexed upper is not always possible.
207 * ovl_connect_layer() will try to make origin's layer "connected" by
208 * copying up a "connectable" ancestor.
210 if (d_is_dir(dentry
) && ofs
->upper_mnt
)
211 return ovl_connect_layer(dentry
);
213 /* Lower file handle for indexed and non-upper dir/non-dir */
217 static int ovl_d_to_fh(struct dentry
*dentry
, char *buf
, int buflen
)
219 struct ovl_fh
*fh
= NULL
;
223 * Check if we should encode a lower or upper file handle and maybe
224 * copy up an ancestor to make lower file handle connectable.
226 err
= enc_lower
= ovl_check_encode_origin(dentry
);
230 /* Encode an upper or lower file handle */
231 fh
= ovl_encode_fh(enc_lower
? ovl_dentry_lower(dentry
) :
232 ovl_dentry_upper(dentry
), !enc_lower
);
238 if (fh
->len
> buflen
)
241 memcpy(buf
, (char *)fh
, fh
->len
);
249 pr_warn_ratelimited("overlayfs: failed to encode file handle (%pd2, err=%i, buflen=%d, len=%d, type=%d)\n",
250 dentry
, err
, buflen
, fh
? (int)fh
->len
: 0,
255 static int ovl_dentry_to_fh(struct dentry
*dentry
, u32
*fid
, int *max_len
)
257 int res
, len
= *max_len
<< 2;
259 res
= ovl_d_to_fh(dentry
, (char *)fid
, len
);
261 return FILEID_INVALID
;
265 /* Round up to dwords */
266 *max_len
= (len
+ 3) >> 2;
270 static int ovl_encode_inode_fh(struct inode
*inode
, u32
*fid
, int *max_len
,
271 struct inode
*parent
)
273 struct dentry
*dentry
;
276 /* TODO: encode connectable file handles */
278 return FILEID_INVALID
;
280 dentry
= d_find_any_alias(inode
);
281 if (WARN_ON(!dentry
))
282 return FILEID_INVALID
;
284 type
= ovl_dentry_to_fh(dentry
, fid
, max_len
);
291 * Find or instantiate an overlay dentry from real dentries and index.
293 static struct dentry
*ovl_obtain_alias(struct super_block
*sb
,
294 struct dentry
*upper_alias
,
295 struct ovl_path
*lowerpath
,
296 struct dentry
*index
)
298 struct dentry
*lower
= lowerpath
? lowerpath
->dentry
: NULL
;
299 struct dentry
*upper
= upper_alias
?: index
;
300 struct dentry
*dentry
;
302 struct ovl_entry
*oe
;
304 /* We get overlay directory dentries with ovl_lookup_real() */
305 if (d_is_dir(upper
?: lower
))
306 return ERR_PTR(-EIO
);
308 inode
= ovl_get_inode(sb
, dget(upper
), lower
, index
, !!lower
);
311 return ERR_CAST(inode
);
315 ovl_set_flag(OVL_INDEX
, inode
);
317 dentry
= d_find_any_alias(inode
);
319 dentry
= d_alloc_anon(inode
->i_sb
);
322 oe
= ovl_alloc_entry(lower
? 1 : 0);
327 oe
->lowerstack
->dentry
= dget(lower
);
328 oe
->lowerstack
->layer
= lowerpath
->layer
;
330 dentry
->d_fsdata
= oe
;
332 ovl_dentry_set_upper_alias(dentry
);
335 return d_instantiate_anon(dentry
, inode
);
340 return ERR_PTR(-ENOMEM
);
343 /* Get the upper or lower dentry in stach whose on layer @idx */
344 static struct dentry
*ovl_dentry_real_at(struct dentry
*dentry
, int idx
)
346 struct ovl_entry
*oe
= dentry
->d_fsdata
;
350 return ovl_dentry_upper(dentry
);
352 for (i
= 0; i
< oe
->numlower
; i
++) {
353 if (oe
->lowerstack
[i
].layer
->idx
== idx
)
354 return oe
->lowerstack
[i
].dentry
;
361 * Lookup a child overlay dentry to get a connected overlay dentry whose real
362 * dentry is @real. If @real is on upper layer, we lookup a child overlay
363 * dentry with the same name as the real dentry. Otherwise, we need to consult
366 static struct dentry
*ovl_lookup_real_one(struct dentry
*connected
,
368 struct ovl_layer
*layer
)
370 struct inode
*dir
= d_inode(connected
);
371 struct dentry
*this, *parent
= NULL
;
372 struct name_snapshot name
;
376 * Lookup child overlay dentry by real name. The dir mutex protects us
377 * from racing with overlay rename. If the overlay dentry that is above
378 * real has already been moved to a parent that is not under the
379 * connected overlay dir, we return -ECHILD and restart the lookup of
380 * connected real path from the top.
382 inode_lock_nested(dir
, I_MUTEX_PARENT
);
384 parent
= dget_parent(real
);
385 if (ovl_dentry_real_at(connected
, layer
->idx
) != parent
)
389 * We also need to take a snapshot of real dentry name to protect us
390 * from racing with underlying layer rename. In this case, we don't
391 * care about returning ESTALE, only from dereferencing a free name
392 * pointer because we hold no lock on the real dentry.
394 take_dentry_name_snapshot(&name
, real
);
395 this = lookup_one_len(name
.name
, connected
, strlen(name
.name
));
399 } else if (!this || !this->d_inode
) {
403 } else if (ovl_dentry_real_at(this, layer
->idx
) != real
) {
410 release_dentry_name_snapshot(&name
);
416 pr_warn_ratelimited("overlayfs: failed to lookup one by real (%pd2, layer=%d, connected=%pd2, err=%i)\n",
417 real
, layer
->idx
, connected
, err
);
422 static struct dentry
*ovl_lookup_real(struct super_block
*sb
,
424 struct ovl_layer
*layer
);
427 * Lookup an indexed or hashed overlay dentry by real inode.
429 static struct dentry
*ovl_lookup_real_inode(struct super_block
*sb
,
431 struct ovl_layer
*layer
)
433 struct ovl_fs
*ofs
= sb
->s_fs_info
;
434 struct ovl_layer upper_layer
= { .mnt
= ofs
->upper_mnt
};
435 struct dentry
*index
= NULL
;
436 struct dentry
*this = NULL
;
440 * Decoding upper dir from index is expensive, so first try to lookup
441 * overlay dentry in inode/dcache.
443 inode
= ovl_lookup_inode(sb
, real
, !layer
->idx
);
445 return ERR_CAST(inode
);
447 this = d_find_any_alias(inode
);
452 * For decoded lower dir file handle, lookup index by origin to check
453 * if lower dir was copied up and and/or removed.
455 if (!this && layer
->idx
&& ofs
->indexdir
&& !WARN_ON(!d_is_dir(real
))) {
456 index
= ovl_lookup_index(ofs
, NULL
, real
, false);
461 /* Get connected upper overlay dir from index */
463 struct dentry
*upper
= ovl_index_upper(ofs
, index
);
466 if (IS_ERR_OR_NULL(upper
))
470 * ovl_lookup_real() in lower layer may call recursively once to
471 * ovl_lookup_real() in upper layer. The first level call walks
472 * back lower parents to the topmost indexed parent. The second
473 * recursive call walks back from indexed upper to the topmost
474 * connected/hashed upper parent (or up to root).
476 this = ovl_lookup_real(sb
, upper
, &upper_layer
);
480 if (IS_ERR_OR_NULL(this))
483 if (WARN_ON(ovl_dentry_real_at(this, layer
->idx
) != real
)) {
485 this = ERR_PTR(-EIO
);
492 * Lookup an indexed or hashed overlay dentry, whose real dentry is an
495 static struct dentry
*ovl_lookup_real_ancestor(struct super_block
*sb
,
497 struct ovl_layer
*layer
)
499 struct dentry
*next
, *parent
= NULL
;
500 struct dentry
*ancestor
= ERR_PTR(-EIO
);
502 if (real
== layer
->mnt
->mnt_root
)
503 return dget(sb
->s_root
);
505 /* Find the topmost indexed or hashed ancestor */
508 parent
= dget_parent(next
);
511 * Lookup a matching overlay dentry in inode/dentry
512 * cache or in index by real inode.
514 ancestor
= ovl_lookup_real_inode(sb
, next
, layer
);
518 if (parent
== layer
->mnt
->mnt_root
) {
519 ancestor
= dget(sb
->s_root
);
524 * If @real has been moved out of the layer root directory,
525 * we will eventully hit the real fs root. This cannot happen
526 * by legit overlay rename, so we return error in that case.
528 if (parent
== next
) {
529 ancestor
= ERR_PTR(-EXDEV
);
544 * Lookup a connected overlay dentry whose real dentry is @real.
545 * If @real is on upper layer, we lookup a child overlay dentry with the same
546 * path the real dentry. Otherwise, we need to consult index for lookup.
548 static struct dentry
*ovl_lookup_real(struct super_block
*sb
,
550 struct ovl_layer
*layer
)
552 struct dentry
*connected
;
555 connected
= ovl_lookup_real_ancestor(sb
, real
, layer
);
556 if (IS_ERR(connected
))
560 struct dentry
*next
, *this;
561 struct dentry
*parent
= NULL
;
562 struct dentry
*real_connected
= ovl_dentry_real_at(connected
,
565 if (real_connected
== real
)
568 /* Find the topmost dentry not yet connected */
571 parent
= dget_parent(next
);
573 if (parent
== real_connected
)
577 * If real has been moved out of 'real_connected',
578 * we will not find 'real_connected' and hit the layer
579 * root. In that case, we need to restart connecting.
580 * This game can go on forever in the worst case. We
581 * may want to consider taking s_vfs_rename_mutex if
582 * this happens more than once.
584 if (parent
== layer
->mnt
->mnt_root
) {
586 connected
= dget(sb
->s_root
);
591 * If real file has been moved out of the layer root
592 * directory, we will eventully hit the real fs root.
593 * This cannot happen by legit overlay rename, so we
594 * return error in that case.
596 if (parent
== next
) {
606 this = ovl_lookup_real_one(connected
, next
, layer
);
611 * Lookup of child in overlay can fail when racing with
612 * overlay rename of child away from 'connected' parent.
613 * In this case, we need to restart the lookup from the
614 * top, because we cannot trust that 'real_connected' is
615 * still an ancestor of 'real'. There is a good chance
616 * that the renamed overlay ancestor is now in cache, so
617 * ovl_lookup_real_ancestor() will find it and we can
618 * continue to connect exactly from where lookup failed.
620 if (err
== -ECHILD
) {
621 this = ovl_lookup_real_ancestor(sb
, real
,
623 err
= PTR_ERR_OR_ZERO(this);
641 pr_warn_ratelimited("overlayfs: failed to lookup by real (%pd2, layer=%d, connected=%pd2, err=%i)\n",
642 real
, layer
->idx
, connected
, err
);
648 * Get an overlay dentry from upper/lower real dentries and index.
650 static struct dentry
*ovl_get_dentry(struct super_block
*sb
,
651 struct dentry
*upper
,
652 struct ovl_path
*lowerpath
,
653 struct dentry
*index
)
655 struct ovl_fs
*ofs
= sb
->s_fs_info
;
656 struct ovl_layer upper_layer
= { .mnt
= ofs
->upper_mnt
};
657 struct ovl_layer
*layer
= upper
? &upper_layer
: lowerpath
->layer
;
658 struct dentry
*real
= upper
?: (index
?: lowerpath
->dentry
);
661 * Obtain a disconnected overlay dentry from a non-dir real dentry
665 return ovl_obtain_alias(sb
, upper
, lowerpath
, index
);
667 /* Removed empty directory? */
668 if ((real
->d_flags
& DCACHE_DISCONNECTED
) || d_unhashed(real
))
669 return ERR_PTR(-ENOENT
);
672 * If real dentry is connected and hashed, get a connected overlay
673 * dentry whose real dentry is @real.
675 return ovl_lookup_real(sb
, real
, layer
);
678 static struct dentry
*ovl_upper_fh_to_d(struct super_block
*sb
,
681 struct ovl_fs
*ofs
= sb
->s_fs_info
;
682 struct dentry
*dentry
;
683 struct dentry
*upper
;
686 return ERR_PTR(-EACCES
);
688 upper
= ovl_decode_fh(fh
, ofs
->upper_mnt
);
689 if (IS_ERR_OR_NULL(upper
))
692 dentry
= ovl_get_dentry(sb
, upper
, NULL
, NULL
);
698 static struct dentry
*ovl_lower_fh_to_d(struct super_block
*sb
,
701 struct ovl_fs
*ofs
= sb
->s_fs_info
;
702 struct ovl_path origin
= { };
703 struct ovl_path
*stack
= &origin
;
704 struct dentry
*dentry
= NULL
;
705 struct dentry
*index
= NULL
;
706 struct inode
*inode
= NULL
;
707 bool is_deleted
= false;
710 /* First lookup indexed upper by fh */
712 index
= ovl_get_index_fh(ofs
, fh
);
713 err
= PTR_ERR(index
);
718 /* Found a whiteout index - treat as deleted inode */
724 /* Then try to get upper dir by index */
725 if (index
&& d_is_dir(index
)) {
726 struct dentry
*upper
= ovl_index_upper(ofs
, index
);
728 err
= PTR_ERR(upper
);
729 if (IS_ERR_OR_NULL(upper
))
732 dentry
= ovl_get_dentry(sb
, upper
, NULL
, NULL
);
737 /* Then lookup origin by fh */
738 err
= ovl_check_origin_fh(ofs
, fh
, NULL
, &stack
);
742 err
= ovl_verify_origin(index
, origin
.dentry
, false);
745 } else if (is_deleted
) {
746 /* Lookup deleted non-dir by origin inode */
747 if (!d_is_dir(origin
.dentry
))
748 inode
= ovl_lookup_inode(sb
, origin
.dentry
, false);
750 if (!inode
|| atomic_read(&inode
->i_count
) == 1)
753 /* Deleted but still open? */
754 index
= dget(ovl_i_dentry_upper(inode
));
757 dentry
= ovl_get_dentry(sb
, NULL
, &origin
, index
);
766 dentry
= ERR_PTR(err
);
770 static struct dentry
*ovl_fh_to_dentry(struct super_block
*sb
, struct fid
*fid
,
771 int fh_len
, int fh_type
)
773 struct dentry
*dentry
= NULL
;
774 struct ovl_fh
*fh
= (struct ovl_fh
*) fid
;
775 int len
= fh_len
<< 2;
776 unsigned int flags
= 0;
780 if (fh_type
!= OVL_FILEID
)
783 err
= ovl_check_fh_len(fh
, len
);
788 dentry
= (flags
& OVL_FH_FLAG_PATH_UPPER
) ?
789 ovl_upper_fh_to_d(sb
, fh
) :
790 ovl_lower_fh_to_d(sb
, fh
);
791 err
= PTR_ERR(dentry
);
792 if (IS_ERR(dentry
) && err
!= -ESTALE
)
798 pr_warn_ratelimited("overlayfs: failed to decode file handle (len=%d, type=%d, flags=%x, err=%i)\n",
799 len
, fh_type
, flags
, err
);
803 static struct dentry
*ovl_fh_to_parent(struct super_block
*sb
, struct fid
*fid
,
804 int fh_len
, int fh_type
)
806 pr_warn_ratelimited("overlayfs: connectable file handles not supported; use 'no_subtree_check' exportfs option.\n");
807 return ERR_PTR(-EACCES
);
810 static int ovl_get_name(struct dentry
*parent
, char *name
,
811 struct dentry
*child
)
814 * ovl_fh_to_dentry() returns connected dir overlay dentries and
815 * ovl_fh_to_parent() is not implemented, so we should not get here.
821 static struct dentry
*ovl_get_parent(struct dentry
*dentry
)
824 * ovl_fh_to_dentry() returns connected dir overlay dentries, so we
825 * should not get here.
828 return ERR_PTR(-EIO
);
831 const struct export_operations ovl_export_operations
= {
832 .encode_fh
= ovl_encode_inode_fh
,
833 .fh_to_dentry
= ovl_fh_to_dentry
,
834 .fh_to_parent
= ovl_fh_to_parent
,
835 .get_name
= ovl_get_name
,
836 .get_parent
= ovl_get_parent
,