fs: Give dentry to inode_change_ok() instead of inode

[pandora-kernel.git] / fs / ext4 / inode.c
diff --git a/fs/ext4/inode.c b/fs/ext4/inode.c

index 92655fd..ff2e369 100644 (file)
--- a/fs/ext4/inode.c
+++ b/fs/ext4/inode.c
@@ -38,6 +38,7 @@
  #include <linux/printk.h>
  #include <linux/slab.h>
  #include <linux/ratelimit.h>
+#include <linux/bitops.h>
  
  #include "ext4_jbd2.h"
  #include "xattr.h"
@@ -141,28 +142,27 @@ void ext4_evict_inode(struct inode *inode)
                  * Note that directories do not have this problem because they
                  * don't use page cache.
                  */
-               if (ext4_should_journal_data(inode) &&
+               if (inode->i_ino != EXT4_JOURNAL_INO &&
+                   ext4_should_journal_data(inode) &&
                     (S_ISLNK(inode->i_mode) || S_ISREG(inode->i_mode))) {
                         journal_t *journal = EXT4_SB(inode->i_sb)->s_journal;
                         tid_t commit_tid = EXT4_I(inode)->i_datasync_tid;
  
-                       jbd2_log_start_commit(journal, commit_tid);
-                       jbd2_log_wait_commit(journal, commit_tid);
+                       jbd2_complete_transaction(journal, commit_tid);
                         filemap_write_and_wait(&inode->i_data);
                 }
                 truncate_inode_pages(&inode->i_data, 0);
                 goto no_delete;
         }
  
-       if (!is_bad_inode(inode))
-               dquot_initialize(inode);
+       if (is_bad_inode(inode))
+               goto no_delete;
+       dquot_initialize(inode);
  
         if (ext4_should_order_data(inode))
                 ext4_begin_ordered_truncate(inode, 0);
         truncate_inode_pages(&inode->i_data, 0);
  
-       if (is_bad_inode(inode))
-               goto no_delete;
  
         handle = ext4_journal_start(inode, ext4_blocks_for_truncate(inode)+3);
         if (IS_ERR(handle)) {
@@ -277,6 +277,15 @@ void ext4_da_update_reserve_space(struct inode *inode,
                 used = ei->i_reserved_data_blocks;
         }
  
+       if (unlikely(ei->i_allocated_meta_blocks > ei->i_reserved_meta_blocks)) {
+               ext4_msg(inode->i_sb, KERN_NOTICE, "%s: ino %lu, allocated %d "
+                        "with only %d reserved metadata blocks\n", __func__,
+                        inode->i_ino, ei->i_allocated_meta_blocks,
+                        ei->i_reserved_meta_blocks);
+               WARN_ON(1);
+               ei->i_allocated_meta_blocks = ei->i_reserved_meta_blocks;
+       }
+
         /* Update per-inode reservations */
         ei->i_reserved_data_blocks -= used;
         ei->i_reserved_meta_blocks -= ei->i_allocated_meta_blocks;
@@ -471,6 +480,11 @@ int ext4_map_blocks(handle_t *handle, struct inode *inode,
         ext_debug("ext4_map_blocks(): inode %lu, flag %d, max_blocks %u,"
                   "logical block %lu\n", inode->i_ino, flags, map->m_len,
                   (unsigned long) map->m_lblk);
+
+       /* We can handle the block number less than EXT_MAX_BLOCKS */
+       if (unlikely(map->m_lblk >= EXT_MAX_BLOCKS))
+               return -EIO;
+
         /*
          * Try to see if we can get the block without requesting a new
          * file system block.
@@ -581,6 +595,34 @@ int ext4_map_blocks(handle_t *handle, struct inode *inode,
         return retval;
  }
  
+/*
+ * Update EXT4_MAP_FLAGS in bh->b_state. For buffer heads attached to pages
+ * we have to be careful as someone else may be manipulating b_state as well.
+ */
+static void ext4_update_bh_state(struct buffer_head *bh, unsigned long flags)
+{
+       unsigned long old_state;
+       unsigned long new_state;
+
+       flags &= EXT4_MAP_FLAGS;
+
+       /* Dummy buffer_head? Set non-atomically. */
+       if (!bh->b_page) {
+               bh->b_state = (bh->b_state & ~EXT4_MAP_FLAGS) | flags;
+               return;
+       }
+       /*
+        * Someone else may be modifying b_state. Be careful! This is ugly but
+        * once we get rid of using bh as a container for mapping information
+        * to pass to / from get_block functions, this can go away.
+        */
+       do {
+               old_state = ACCESS_ONCE(bh->b_state);
+               new_state = (old_state & ~EXT4_MAP_FLAGS) | flags;
+       } while (unlikely(
+                cmpxchg(&bh->b_state, old_state, new_state) != old_state));
+}
+
  /* Maximum number of blocks we map for direct IO at once. */
  #define DIO_MAX_BLOCKS 4096
  
@@ -611,7 +653,7 @@ static int _ext4_get_block(struct inode *inode, sector_t iblock,
         ret = ext4_map_blocks(handle, inode, &map, flags);
         if (ret > 0) {
                 map_bh(bh, inode->i_sb, map.m_pblk);
-               bh->b_state = (bh->b_state & ~EXT4_MAP_FLAGS) | map.m_flags;
+               ext4_update_bh_state(bh, map.m_flags);
                 bh->b_size = inode->i_sb->s_blocksize * map.m_len;
                 ret = 0;
         }
@@ -652,7 +694,7 @@ struct buffer_head *ext4_getblk(handle_t *handle, struct inode *inode,
  
         bh = sb_getblk(inode->i_sb, map.m_pblk);
         if (!bh) {
-               *errp = -EIO;
+               *errp = -ENOMEM;
                 return NULL;
         }
         if (map.m_flags & EXT4_MAP_NEW) {
@@ -1102,6 +1144,17 @@ static int ext4_da_reserve_space(struct inode *inode, ext4_lblk_t lblock)
         struct ext4_inode_info *ei = EXT4_I(inode);
         unsigned int md_needed;
         int ret;
+       ext4_lblk_t save_last_lblock;
+       int save_len;
+
+       /*
+        * We will charge metadata quota at writeout time; this saves
+        * us from metadata over-estimation, though we may go over by
+        * a small amount in the end.  Here we just reserve for data.
+        */
+       ret = dquot_reserve_block(inode, EXT4_C2B(sbi, 1));
+       if (ret)
+               return ret;
  
         /*
          * recalculate the amount of metadata blocks to reserve
@@ -1110,32 +1163,31 @@ static int ext4_da_reserve_space(struct inode *inode, ext4_lblk_t lblock)
          */
  repeat:
         spin_lock(&ei->i_block_reservation_lock);
+       /*
+        * ext4_calc_metadata_amount() has side effects, which we have
+        * to be prepared undo if we fail to claim space.
+        */
+       save_len = ei->i_da_metadata_calc_len;
+       save_last_lblock = ei->i_da_metadata_calc_last_lblock;
         md_needed = EXT4_NUM_B2C(sbi,
                                  ext4_calc_metadata_amount(inode, lblock));
         trace_ext4_da_reserve_space(inode, md_needed);
-       spin_unlock(&ei->i_block_reservation_lock);
  
-       /*
-        * We will charge metadata quota at writeout time; this saves
-        * us from metadata over-estimation, though we may go over by
-        * a small amount in the end.  Here we just reserve for data.
-        */
-       ret = dquot_reserve_block(inode, EXT4_C2B(sbi, 1));
-       if (ret)
-               return ret;
         /*
          * We do still charge estimated metadata to the sb though;
          * we cannot afford to run out of free blocks.
          */
         if (ext4_claim_free_clusters(sbi, md_needed + 1, 0)) {
-               dquot_release_reservation_block(inode, EXT4_C2B(sbi, 1));
+               ei->i_da_metadata_calc_len = save_len;
+               ei->i_da_metadata_calc_last_lblock = save_last_lblock;
+               spin_unlock(&ei->i_block_reservation_lock);
                 if (ext4_should_retry_alloc(inode->i_sb, &retries)) {
                         yield();
                         goto repeat;
                 }
+               dquot_release_reservation_block(inode, EXT4_C2B(sbi, 1));
                 return -ENOSPC;
         }
-       spin_lock(&ei->i_block_reservation_lock);
         ei->i_reserved_data_blocks++;
         ei->i_reserved_meta_blocks += md_needed;
         spin_unlock(&ei->i_block_reservation_lock);
@@ -1403,6 +1455,7 @@ static void ext4_da_block_invalidatepages(struct mpage_da_data *mpd)
  
         index = mpd->first_page;
         end   = mpd->next_page - 1;
+       pagevec_init(&pvec, 0);
         while (index <= end) {
                 nr_pages = pagevec_lookup(&pvec, mapping, index, PAGEVEC_SIZE);
                 if (nr_pages == 0)
@@ -1762,7 +1815,7 @@ static int ext4_da_get_block_prep(struct inode *inode, sector_t iblock,
                 return ret;
  
         map_bh(bh, inode->i_sb, map.m_pblk);
-       bh->b_state = (bh->b_state & ~EXT4_MAP_FLAGS) | map.m_flags;
+       ext4_update_bh_state(bh, map.m_flags);
  
         if (buffer_unwritten(bh)) {
                 /* A delayed write to unwritten bh should be marked
@@ -1824,18 +1877,31 @@ static int __ext4_journalled_writepage(struct page *page,
         page_bufs = page_buffers(page);
         BUG_ON(!page_bufs);
         walk_page_buffers(handle, page_bufs, 0, len, NULL, bget_one);
-       /* As soon as we unlock the page, it can go away, but we have
-        * references to buffers so we are safe */
+       /*
+        * We need to release the page lock before we start the
+        * journal, so grab a reference so the page won't disappear
+        * out from under us.
+        */
+       get_page(page);
         unlock_page(page);
  
         handle = ext4_journal_start(inode, ext4_writepage_trans_blocks(inode));
         if (IS_ERR(handle)) {
                 ret = PTR_ERR(handle);
-               goto out;
+               put_page(page);
+               goto out_no_pagelock;
         }
-
         BUG_ON(!ext4_handle_valid(handle));
  
+       lock_page(page);
+       put_page(page);
+       if (page->mapping != mapping) {
+               /* The page got truncated from under us */
+               ext4_journal_stop(handle);
+               ret = 0;
+               goto out;
+       }
+
         ret = walk_page_buffers(handle, page_bufs, 0, len, NULL,
                                 do_journal_get_write_access);
  
@@ -1851,6 +1917,8 @@ static int __ext4_journalled_writepage(struct page *page,
         walk_page_buffers(handle, page_bufs, 0, len, NULL, bput_one);
         ext4_set_inode_state(inode, EXT4_STATE_JDATA);
  out:
+       unlock_page(page);
+out_no_pagelock:
         return ret;
  }
  
@@ -2363,6 +2431,16 @@ static int ext4_nonda_switch(struct super_block *sb)
         free_blocks  = EXT4_C2B(sbi,
                 percpu_counter_read_positive(&sbi->s_freeclusters_counter));
         dirty_blocks = percpu_counter_read_positive(&sbi->s_dirtyclusters_counter);
+       /*
+        * Start pushing delalloc when 1/2 of free blocks are dirty.
+        */
+       if (dirty_blocks && (free_blocks < 2 * dirty_blocks) &&
+           !writeback_in_progress(sb->s_bdi) &&
+           down_read_trylock(&sb->s_umount)) {
+               writeback_inodes_sb(sb, WB_REASON_FS_FREE_SPACE);
+               up_read(&sb->s_umount);
+       }
+
         if (2 * free_blocks < 3 * dirty_blocks ||
                 free_blocks < (dirty_blocks + EXT4_FREECLUSTERS_WATERMARK)) {
                 /*
@@ -2371,16 +2449,23 @@ static int ext4_nonda_switch(struct super_block *sb)
                  */
                 return 1;
         }
-       /*
-        * Even if we don't switch but are nearing capacity,
-        * start pushing delalloc when 1/2 of free blocks are dirty.
-        */
-       if (free_blocks < 2 * dirty_blocks)
-               writeback_inodes_sb_if_idle(sb, WB_REASON_FS_FREE_SPACE);
-
         return 0;
  }
  
+/* We always reserve for an inode update; the superblock could be there too */
+static int ext4_da_write_credits(struct inode *inode, loff_t pos, unsigned len)
+{
+       if (likely(EXT4_HAS_RO_COMPAT_FEATURE(inode->i_sb,
+                               EXT4_FEATURE_RO_COMPAT_LARGE_FILE)))
+               return 1;
+
+       if (pos + len <= 0x7fffffffULL)
+               return 1;
+
+       /* We might need to update the superblock to set LARGE_FILE */
+       return 2;
+}
+
  static int ext4_da_write_begin(struct file *file, struct address_space *mapping,
                                loff_t pos, unsigned len, unsigned flags,
                                struct page **pagep, void **fsdata)
@@ -2407,7 +2492,8 @@ retry:
          * to journalling the i_disksize update if writes to the end
          * of file which has an already mapped buffer.
          */
-       handle = ext4_journal_start(inode, 1);
+       handle = ext4_journal_start(inode,
+                               ext4_da_write_credits(inode, pos, len));
         if (IS_ERR(handle)) {
                 ret = PTR_ERR(handle);
                 goto out;
@@ -2480,13 +2566,14 @@ static int ext4_da_write_end(struct file *file,
         int write_mode = (int)(unsigned long)fsdata;
  
         if (write_mode == FALL_BACK_TO_NONDELALLOC) {
-               if (ext4_should_order_data(inode)) {
+               switch (ext4_inode_journal_mode(inode)) {
+               case EXT4_INODE_ORDERED_DATA_MODE:
                         return ext4_ordered_write_end(file, mapping, pos,
                                         len, copied, page, fsdata);
-               } else if (ext4_should_writeback_data(inode)) {
+               case EXT4_INODE_WRITEBACK_DATA_MODE:
                         return ext4_writeback_write_end(file, mapping, pos,
                                         len, copied, page, fsdata);
-               } else {
+               default:
                         BUG();
                 }
         }
@@ -2771,9 +2858,9 @@ static void ext4_end_io_dio(struct kiocb *iocb, loff_t offset,
         if (!(io_end->flag & EXT4_IO_END_UNWRITTEN)) {
                 ext4_free_io_end(io_end);
  out:
+               inode_dio_done(inode);
                 if (is_async)
                         aio_complete(iocb, ret, 0);
-               inode_dio_done(inode);
                 return;
         }
  
@@ -2793,9 +2880,6 @@ out:
  
         /* queue the work to convert unwritten extents to written */
         queue_work(wq, &io_end->work);
-
-       /* XXX: probably should move into the real I/O completion handler */
-       inode_dio_done(inode);
  }
  
  static void ext4_end_io_buffer_write(struct buffer_head *bh, int uptodate)
@@ -2919,9 +3003,12 @@ static ssize_t ext4_ext_direct_IO(int rw, struct kiocb *iocb,
                 iocb->private = NULL;
                 EXT4_I(inode)->cur_aio_dio = NULL;
                 if (!is_sync_kiocb(iocb)) {
-                       iocb->private = ext4_init_io_end(inode, GFP_NOFS);
-                       if (!iocb->private)
+                       ext4_io_end_t *io_end =
+                               ext4_init_io_end(inode, GFP_NOFS);
+                       if (!io_end)
                                 return -ENOMEM;
+                       io_end->flag |= EXT4_IO_END_DIRECT;
+                       iocb->private = io_end;
                         /*
                          * we save the io structure for current async
                          * direct IO, so that later ext4_map_blocks()
@@ -3084,18 +3171,25 @@ static const struct address_space_operations ext4_da_aops = {
  
  void ext4_set_aops(struct inode *inode)
  {
-       if (ext4_should_order_data(inode) &&
-               test_opt(inode->i_sb, DELALLOC))
-               inode->i_mapping->a_ops = &ext4_da_aops;
-       else if (ext4_should_order_data(inode))
-               inode->i_mapping->a_ops = &ext4_ordered_aops;
-       else if (ext4_should_writeback_data(inode) &&
-                test_opt(inode->i_sb, DELALLOC))
-               inode->i_mapping->a_ops = &ext4_da_aops;
-       else if (ext4_should_writeback_data(inode))
-               inode->i_mapping->a_ops = &ext4_writeback_aops;
-       else
+       switch (ext4_inode_journal_mode(inode)) {
+       case EXT4_INODE_ORDERED_DATA_MODE:
+               if (test_opt(inode->i_sb, DELALLOC))
+                       inode->i_mapping->a_ops = &ext4_da_aops;
+               else
+                       inode->i_mapping->a_ops = &ext4_ordered_aops;
+               break;
+       case EXT4_INODE_WRITEBACK_DATA_MODE:
+               if (test_opt(inode->i_sb, DELALLOC))
+                       inode->i_mapping->a_ops = &ext4_da_aops;
+               else
+                       inode->i_mapping->a_ops = &ext4_writeback_aops;
+               break;
+       case EXT4_INODE_JOURNAL_DATA_MODE:
                 inode->i_mapping->a_ops = &ext4_journalled_aops;
+               break;
+       default:
+               BUG();
+       }
  }
  
  
@@ -3544,11 +3638,8 @@ static int __ext4_get_inode_loc(struct inode *inode,
         iloc->offset = (inode_offset % inodes_per_block) * EXT4_INODE_SIZE(sb);
  
         bh = sb_getblk(sb, block);
-       if (!bh) {
-               EXT4_ERROR_INODE_BLOCK(inode, block,
-                                      "unable to read itable block");
-               return -EIO;
-       }
+       if (!bh)
+               return -ENOMEM;
         if (!buffer_uptodate(bh)) {
                 lock_buffer(bh);
  
@@ -3666,18 +3757,20 @@ int ext4_get_inode_loc(struct inode *inode, struct ext4_iloc *iloc)
  void ext4_set_inode_flags(struct inode *inode)
  {
         unsigned int flags = EXT4_I(inode)->i_flags;
+       unsigned int new_fl = 0;
  
-       inode->i_flags &= ~(S_SYNC|S_APPEND|S_IMMUTABLE|S_NOATIME|S_DIRSYNC);
         if (flags & EXT4_SYNC_FL)
-               inode->i_flags |= S_SYNC;
+               new_fl |= S_SYNC;
         if (flags & EXT4_APPEND_FL)
-               inode->i_flags |= S_APPEND;
+               new_fl |= S_APPEND;
         if (flags & EXT4_IMMUTABLE_FL)
-               inode->i_flags |= S_IMMUTABLE;
+               new_fl |= S_IMMUTABLE;
         if (flags & EXT4_NOATIME_FL)
-               inode->i_flags |= S_NOATIME;
+               new_fl |= S_NOATIME;
         if (flags & EXT4_DIRSYNC_FL)
-               inode->i_flags |= S_DIRSYNC;
+               new_fl |= S_DIRSYNC;
+       set_mask_bits(&inode->i_flags,
+                     S_SYNC|S_APPEND|S_IMMUTABLE|S_NOATIME|S_DIRSYNC, new_fl);
  }
  
  /* Propagate flags from i_flags to EXT4_I(inode)->i_flags */
@@ -3923,6 +4016,13 @@ bad_inode:
         return ERR_PTR(ret);
  }
  
+struct inode *ext4_iget_normal(struct super_block *sb, unsigned long ino)
+{
+       if (ino < EXT4_FIRST_INO(sb) && ino != EXT4_ROOT_INO)
+               return ERR_PTR(-EIO);
+       return ext4_iget(sb, ino);
+}
+
  static int ext4_inode_blocks_set(handle_t *handle,
                                 struct ext4_inode *raw_inode,
                                 struct ext4_inode_info *ei)
@@ -3977,6 +4077,7 @@ static int ext4_do_update_inode(handle_t *handle,
         struct ext4_inode_info *ei = EXT4_I(inode);
         struct buffer_head *bh = iloc->bh;
         int err = 0, rc, block;
+       int need_datasync = 0;
  
         /* For fields not not tracking in the in-memory inode,
          * initialise them to zero for new inodes. */
@@ -4025,7 +4126,10 @@ static int ext4_do_update_inode(handle_t *handle,
                 raw_inode->i_file_acl_high =
                         cpu_to_le16(ei->i_file_acl >> 32);
         raw_inode->i_file_acl_lo = cpu_to_le32(ei->i_file_acl);
-       ext4_isize_set(raw_inode, ei->i_disksize);
+       if (ei->i_disksize != ext4_isize(raw_inode)) {
+               ext4_isize_set(raw_inode, ei->i_disksize);
+               need_datasync = 1;
+       }
         if (ei->i_disksize > 0x7fffffffULL) {
                 struct super_block *sb = inode->i_sb;
                 if (!EXT4_HAS_RO_COMPAT_FEATURE(sb,
@@ -4078,7 +4182,7 @@ static int ext4_do_update_inode(handle_t *handle,
                 err = rc;
         ext4_clear_inode_state(inode, EXT4_STATE_NEW);
  
-       ext4_update_inode_fsync_trans(handle, inode, 0);
+       ext4_update_inode_fsync_trans(handle, inode, need_datasync);
  out_brelse:
         brelse(bh);
         ext4_std_error(inode->i_sb, err);
@@ -4187,7 +4291,7 @@ int ext4_setattr(struct dentry *dentry, struct iattr *attr)
         int orphan = 0;
         const unsigned int ia_valid = attr->ia_valid;
  
-       error = inode_change_ok(inode, attr);
+       error = setattr_prepare(dentry, attr);
         if (error)
                 return error;
  
@@ -4303,7 +4407,7 @@ int ext4_getattr(struct vfsmount *mnt, struct dentry *dentry,
                  struct kstat *stat)
  {
         struct inode *inode;
-       unsigned long delalloc_blocks;
+       unsigned long long delalloc_blocks;
  
         inode = dentry->d_inode;
         generic_fillattr(inode, stat);
@@ -4320,7 +4424,7 @@ int ext4_getattr(struct vfsmount *mnt, struct dentry *dentry,
          */
         delalloc_blocks = EXT4_I(inode)->i_reserved_data_blocks;
  
-       stat->blocks += (delalloc_blocks << inode->i_sb->s_blocksize_bits)>>9;
+       stat->blocks += delalloc_blocks << (inode->i_sb->s_blocksize_bits-9);
         return 0;
  }
  
@@ -4532,6 +4636,8 @@ int ext4_mark_inode_dirty(handle_t *handle, struct inode *inode)
         might_sleep();
         trace_ext4_mark_inode_dirty(inode, _RET_IP_);
         err = ext4_reserve_inode_write(handle, inode, &iloc);
+       if (err)
+               return err;
         if (ext4_handle_valid(handle) &&
             EXT4_I(inode)->i_extra_isize < sbi->s_want_extra_isize &&
             !ext4_test_inode_state(inode, EXT4_STATE_NO_EXPAND)) {
@@ -4562,9 +4668,7 @@ int ext4_mark_inode_dirty(handle_t *handle, struct inode *inode)
                         }
                 }
         }
-       if (!err)
-               err = ext4_mark_iloc_dirty(handle, inode, &iloc);
-       return err;
+       return ext4_mark_iloc_dirty(handle, inode, &iloc);
  }
  
  /*