fs/ext4/file.c

   1 /*
   2  *  linux/fs/ext4/file.c
   3  *
   4  * Copyright (C) 1992, 1993, 1994, 1995
   5  * Remy Card (card@masi.ibp.fr)
   6  * Laboratoire MASI - Institut Blaise Pascal
   7  * Universite Pierre et Marie Curie (Paris VI)
   8  *
   9  *  from
  10  *
  11  *  linux/fs/minix/file.c
  12  *
  13  *  Copyright (C) 1991, 1992  Linus Torvalds
  14  *
  15  *  ext4 fs regular file handling primitives
  16  *
  17  *  64-bit file support on 64-bit platforms by Jakub Jelinek
  18  *      (jj@sunsite.ms.mff.cuni.cz)
  19  */
  20
  21 #include <linux/time.h>
  22 #include <linux/fs.h>
  23 #include <linux/jbd2.h>
  24 #include <linux/mount.h>
  25 #include <linux/path.h>
  26 #include <linux/quotaops.h>
  27 #include "ext4.h"
  28 #include "ext4_jbd2.h"
  29 #include "xattr.h"
  30 #include "acl.h"
  31
  32 /*
  33  * Called when an inode is released. Note that this is different
  34  * from ext4_file_open: open gets called at every open, but release
  35  * gets called only when /all/ the files are closed.
  36  */
  37 static int ext4_release_file(struct inode *inode, struct file *filp)
  38 {
  39         if (ext4_test_inode_state(inode, EXT4_STATE_DA_ALLOC_CLOSE)) {
  40                 ext4_alloc_da_blocks(inode);
  41                 ext4_clear_inode_state(inode, EXT4_STATE_DA_ALLOC_CLOSE);
  42         }
  43         /* if we are the last writer on the inode, drop the block reservation */
  44         if ((filp->f_mode & FMODE_WRITE) &&
  45                         (atomic_read(&inode->i_writecount) == 1) &&
  46                         !EXT4_I(inode)->i_reserved_data_blocks)
  47         {
  48                 down_write(&EXT4_I(inode)->i_data_sem);
  49                 ext4_discard_preallocations(inode);
  50                 up_write(&EXT4_I(inode)->i_data_sem);
  51         }
  52         if (is_dx(inode) && filp->private_data)
  53                 ext4_htree_free_dir_info(filp->private_data);
  54
  55         return 0;
  56 }
  57
  58 static void ext4_aiodio_wait(struct inode *inode)
  59 {
  60         wait_queue_head_t *wq = ext4_ioend_wq(inode);
  61
  62         wait_event(*wq, (atomic_read(&EXT4_I(inode)->i_aiodio_unwritten) == 0));
  63 }
  64
  65 /*
  66  * This tests whether the IO in question is block-aligned or not.
  67  * Ext4 utilizes unwritten extents when hole-filling during direct IO, and they
  68  * are converted to written only after the IO is complete.  Until they are
  69  * mapped, these blocks appear as holes, so dio_zero_block() will assume that
  70  * it needs to zero out portions of the start and/or end block.  If 2 AIO
  71  * threads are at work on the same unwritten block, they must be synchronized
  72  * or one thread will zero the other's data, causing corruption.
  73  */
  74 static int
  75 ext4_unaligned_aio(struct inode *inode, const struct iovec *iov,
  76                    unsigned long nr_segs, loff_t pos)
  77 {
  78         struct super_block *sb = inode->i_sb;
  79         int blockmask = sb->s_blocksize - 1;
  80         size_t count = iov_length(iov, nr_segs);
  81         loff_t final_size = pos + count;
  82
  83         if (pos >= inode->i_size)
  84                 return 0;
  85
  86         if ((pos & blockmask) || (final_size & blockmask))
  87                 return 1;
  88
  89         return 0;
  90 }
  91
  92 static ssize_t
  93 ext4_file_write(struct kiocb *iocb, const struct iovec *iov,
  94                 unsigned long nr_segs, loff_t pos)
  95 {
  96         struct inode *inode = iocb->ki_filp->f_path.dentry->d_inode;
  97         int unaligned_aio = 0;
  98         int ret;
  99
 100         /*
 101          * If O_DIRECT is set and we are doing data journalling we
 102          * don't support O_DIRECT so force it off.
 103          */
 104         if ((iocb->ki_filp->f_flags & O_DIRECT) &&
 105             ext4_should_journal_data(inode)) {
 106                 iocb->ki_filp->f_flags &= ~O_DIRECT;
 107                 iocb->ki_filp->f_flags |= O_DSYNC;
 108         }
 109
 110         /*
 111          * If we have encountered a bitmap-format file, the size limit
 112          * is smaller than s_maxbytes, which is for extent-mapped files.
 113          */
 114
 115         if (!(ext4_test_inode_flag(inode, EXT4_INODE_EXTENTS))) {
 116                 struct ext4_sb_info *sbi = EXT4_SB(inode->i_sb);
 117                 size_t length = iov_length(iov, nr_segs);
 118
 119                 if ((pos > sbi->s_bitmap_maxbytes ||
 120                     (pos == sbi->s_bitmap_maxbytes && length > 0)))
 121                         return -EFBIG;
 122
 123                 if (pos + length > sbi->s_bitmap_maxbytes) {
 124                         nr_segs = iov_shorten((struct iovec *)iov, nr_segs,
 125                                               sbi->s_bitmap_maxbytes - pos);
 126                 }
 127         } else if (unlikely((iocb->ki_filp->f_flags & O_DIRECT) &&
 128                    !is_sync_kiocb(iocb))) {
 129                 unaligned_aio = ext4_unaligned_aio(inode, iov, nr_segs, pos);
 130         }
 131
 132         /* Unaligned direct AIO must be serialized; see comment above */
 133         if (unaligned_aio) {
 134                 static unsigned long unaligned_warn_time;
 135
 136                 /* Warn about this once per day */
 137                 if (printk_timed_ratelimit(&unaligned_warn_time, 60*60*24*HZ))
 138                         ext4_msg(inode->i_sb, KERN_WARNING,
 139                                  "Unaligned AIO/DIO on inode %ld by %s; "
 140                                  "performance will be poor.",
 141                                  inode->i_ino, current->comm);
 142                 mutex_lock(ext4_aio_mutex(inode));
 143                 ext4_aiodio_wait(inode);
 144         }
 145
 146         ret = generic_file_aio_write(iocb, iov, nr_segs, pos);
 147
 148         if (unaligned_aio)
 149                 mutex_unlock(ext4_aio_mutex(inode));
 150
 151         return ret;
 152 }
 153
 154 static const struct vm_operations_struct ext4_file_vm_ops = {
 155         .fault          = filemap_fault,
 156         .page_mkwrite   = ext4_page_mkwrite,
 157 };
 158
 159 static int ext4_file_mmap(struct file *file, struct vm_area_struct *vma)
 160 {
 161         struct address_space *mapping = file->f_mapping;
 162
 163         if (!mapping->a_ops->readpage)
 164                 return -ENOEXEC;
 165         file_accessed(file);
 166         vma->vm_ops = &ext4_file_vm_ops;
 167         vma->vm_flags |= VM_CAN_NONLINEAR;
 168         return 0;
 169 }
 170
 171 static int ext4_file_open(struct inode * inode, struct file * filp)
 172 {
 173         struct super_block *sb = inode->i_sb;
 174         struct ext4_sb_info *sbi = EXT4_SB(inode->i_sb);
 175         struct ext4_inode_info *ei = EXT4_I(inode);
 176         struct vfsmount *mnt = filp->f_path.mnt;
 177         struct path path;
 178         char buf[64], *cp;
 179
 180         if (unlikely(!(sbi->s_mount_flags & EXT4_MF_MNTDIR_SAMPLED) &&
 181                      !(sb->s_flags & MS_RDONLY))) {
 182                 sbi->s_mount_flags |= EXT4_MF_MNTDIR_SAMPLED;
 183                 /*
 184                  * Sample where the filesystem has been mounted and
 185                  * store it in the superblock for sysadmin convenience
 186                  * when trying to sort through large numbers of block
 187                  * devices or filesystem images.
 188                  */
 189                 memset(buf, 0, sizeof(buf));
 190                 path.mnt = mnt;
 191                 path.dentry = mnt->mnt_root;
 192                 cp = d_path(&path, buf, sizeof(buf));
 193                 if (!IS_ERR(cp)) {
 194                         memcpy(sbi->s_es->s_last_mounted, cp,
 195                                sizeof(sbi->s_es->s_last_mounted));
 196                         ext4_mark_super_dirty(sb);
 197                 }
 198         }
 199         /*
 200          * Set up the jbd2_inode if we are opening the inode for
 201          * writing and the journal is present
 202          */
 203         if (sbi->s_journal && !ei->jinode && (filp->f_mode & FMODE_WRITE)) {
 204                 struct jbd2_inode *jinode = jbd2_alloc_inode(GFP_KERNEL);
 205
 206                 spin_lock(&inode->i_lock);
 207                 if (!ei->jinode) {
 208                         if (!jinode) {
 209                                 spin_unlock(&inode->i_lock);
 210                                 return -ENOMEM;
 211                         }
 212                         ei->jinode = jinode;
 213                         jbd2_journal_init_jbd_inode(ei->jinode, inode);
 214                         jinode = NULL;
 215                 }
 216                 spin_unlock(&inode->i_lock);
 217                 if (unlikely(jinode != NULL))
 218                         jbd2_free_inode(jinode);
 219         }
 220         return dquot_file_open(inode, filp);
 221 }
 222
 223 /*
 224  * ext4_llseek() copied from generic_file_llseek() to handle both
 225  * block-mapped and extent-mapped maxbytes values. This should
 226  * otherwise be identical with generic_file_llseek().
 227  */
 228 loff_t ext4_llseek(struct file *file, loff_t offset, int origin)
 229 {
 230         struct inode *inode = file->f_mapping->host;
 231         loff_t maxbytes;
 232
 233         if (!(ext4_test_inode_flag(inode, EXT4_INODE_EXTENTS)))
 234                 maxbytes = EXT4_SB(inode->i_sb)->s_bitmap_maxbytes;
 235         else
 236                 maxbytes = inode->i_sb->s_maxbytes;
 237         mutex_lock(&inode->i_mutex);
 238         switch (origin) {
 239         case SEEK_END:
 240                 offset += inode->i_size;
 241                 break;
 242         case SEEK_CUR:
 243                 if (offset == 0) {
 244                         mutex_unlock(&inode->i_mutex);
 245                         return file->f_pos;
 246                 }
 247                 offset += file->f_pos;
 248                 break;
 249         case SEEK_DATA:
 250                 /*
 251                  * In the generic case the entire file is data, so as long as
 252                  * offset isn't at the end of the file then the offset is data.
 253                  */
 254                 if (offset >= inode->i_size) {
 255                         mutex_unlock(&inode->i_mutex);
 256                         return -ENXIO;
 257                 }
 258                 break;
 259         case SEEK_HOLE:
 260                 /*
 261                  * There is a virtual hole at the end of the file, so as long as
 262                  * offset isn't i_size or larger, return i_size.
 263                  */
 264                 if (offset >= inode->i_size) {
 265                         mutex_unlock(&inode->i_mutex);
 266                         return -ENXIO;
 267                 }
 268                 offset = inode->i_size;
 269                 break;
 270         }
 271
 272         if (offset < 0 || offset > maxbytes) {
 273                 mutex_unlock(&inode->i_mutex);
 274                 return -EINVAL;
 275         }
 276
 277         if (offset != file->f_pos) {
 278                 file->f_pos = offset;
 279                 file->f_version = 0;
 280         }
 281         mutex_unlock(&inode->i_mutex);
 282
 283         return offset;
 284 }
 285
 286 const struct file_operations ext4_file_operations = {
 287         .llseek         = ext4_llseek,
 288         .read           = do_sync_read,
 289         .write          = do_sync_write,
 290         .aio_read       = generic_file_aio_read,
 291         .aio_write      = ext4_file_write,
 292         .unlocked_ioctl = ext4_ioctl,
 293 #ifdef CONFIG_COMPAT
 294         .compat_ioctl   = ext4_compat_ioctl,
 295 #endif
 296         .mmap           = ext4_file_mmap,
 297         .open           = ext4_file_open,
 298         .release        = ext4_release_file,
 299         .fsync          = ext4_sync_file,
 300         .splice_read    = generic_file_splice_read,
 301         .splice_write   = generic_file_splice_write,
 302         .fallocate      = ext4_fallocate,
 303 };
 304
 305 const struct inode_operations ext4_file_inode_operations = {
 306         .setattr        = ext4_setattr,
 307         .getattr        = ext4_getattr,
 308 #ifdef CONFIG_EXT4_FS_XATTR
 309         .setxattr       = generic_setxattr,
 310         .getxattr       = generic_getxattr,
 311         .listxattr      = ext4_listxattr,
 312         .removexattr    = generic_removexattr,
 313 #endif
 314         .get_acl        = ext4_get_acl,
 315         .fiemap         = ext4_fiemap,
 316 };
 317