Merge branch 'upstream' of git://git.kernel.org/pub/scm/linux/kernel/git/vitb/linux...

[powerpc.git] / fs / ocfs2 / cluster / heartbeat.c
diff --git a/fs/ocfs2/cluster/heartbeat.c b/fs/ocfs2/cluster/heartbeat.c

index bff0f0d..305cba3 100644 (file)
--- a/fs/ocfs2/cluster/heartbeat.c
+++ b/fs/ocfs2/cluster/heartbeat.c
@@ -54,7 +54,7 @@ static DECLARE_RWSEM(o2hb_callback_sem);
   * multiple hb threads are watching multiple regions.  A node is live
   * whenever any of the threads sees activity from the node in its region.
   */
-static spinlock_t o2hb_live_lock = SPIN_LOCK_UNLOCKED;
+static DEFINE_SPINLOCK(o2hb_live_lock);
  static struct list_head o2hb_live_slots[O2NM_MAX_NODES];
  static unsigned long o2hb_live_node_bitmap[BITS_TO_LONGS(O2NM_MAX_NODES)];
  static LIST_HEAD(o2hb_node_events);
@@ -153,6 +153,7 @@ struct o2hb_region {
  struct o2hb_bio_wait_ctxt {
         atomic_t          wc_num_reqs;
         struct completion wc_io_complete;
+       int               wc_error;
  };
  
  static void o2hb_write_timeout(void *arg)
@@ -186,6 +187,7 @@ static inline void o2hb_bio_wait_init(struct o2hb_bio_wait_ctxt *wc,
  {
         atomic_set(&wc->wc_num_reqs, num_ios);
         init_completion(&wc->wc_io_complete);
+       wc->wc_error = 0;
  }
  
  /* Used in error paths too */
@@ -218,8 +220,10 @@ static int o2hb_bio_end_io(struct bio *bio,
  {
         struct o2hb_bio_wait_ctxt *wc = bio->bi_private;
  
-       if (error)
+       if (error) {
                 mlog(ML_ERROR, "IO Error %d\n", error);
+               wc->wc_error = error;
+       }
  
         if (bio->bi_size)
                 return 1;
@@ -316,8 +320,12 @@ static int compute_max_sectors(struct block_device *bdev)
                 max_pages = q->max_hw_segments;
         max_pages--; /* Handle I/Os that straddle a page */
  
-       max_sectors = max_pages << (PAGE_SHIFT - 9);
-
+       if (max_pages) {
+               max_sectors = max_pages << (PAGE_SHIFT - 9);
+       } else {
+               /* If BIO contains 1 or less than 1 page. */
+               max_sectors = q->max_sectors;
+       }
         /* Why is fls() 1-based???? */
         pow_two_sectors = 1 << (fls(max_sectors) - 1);
  
@@ -390,6 +398,8 @@ static int o2hb_read_slots(struct o2hb_region *reg,
  
  bail_and_wait:
         o2hb_wait_on_io(reg, &wc);
+       if (wc.wc_error && !status)
+               status = wc.wc_error;
  
         if (bios) {
                 for(i = 0; i < num_bios; i++)
@@ -511,6 +521,7 @@ static inline void o2hb_prepare_block(struct o2hb_region *reg,
         hb_block->hb_seq = cpu_to_le64(cputime);
         hb_block->hb_node = node_num;
         hb_block->hb_generation = cpu_to_le64(generation);
+       hb_block->hb_dead_ms = cpu_to_le32(o2hb_dead_threshold * O2HB_REGION_TIMEOUT_MS);
  
         /* This step must always happen last! */
         hb_block->hb_cksum = cpu_to_le32(o2hb_compute_block_crc_le(reg,
@@ -639,6 +650,8 @@ static int o2hb_check_slot(struct o2hb_region *reg,
         struct o2nm_node *node;
         struct o2hb_disk_heartbeat_block *hb_block = reg->hr_tmp_block;
         u64 cputime;
+       unsigned int dead_ms = o2hb_dead_threshold * O2HB_REGION_TIMEOUT_MS;
+       unsigned int slot_dead_ms;
  
         memcpy(hb_block, slot->ds_raw_block, reg->hr_block_bytes);
  
@@ -727,6 +740,23 @@ fire_callbacks:
                               &o2hb_live_slots[slot->ds_node_num]);
  
                 slot->ds_equal_samples = 0;
+
+               /* We want to be sure that all nodes agree on the
+                * number of milliseconds before a node will be
+                * considered dead. The self-fencing timeout is
+                * computed from this value, and a discrepancy might
+                * result in heartbeat calling a node dead when it
+                * hasn't self-fenced yet. */
+               slot_dead_ms = le32_to_cpu(hb_block->hb_dead_ms);
+               if (slot_dead_ms && slot_dead_ms != dead_ms) {
+                       /* TODO: Perhaps we can fail the region here. */
+                       mlog(ML_ERROR, "Node %d on device %s has a dead count "
+                            "of %u ms, but our count is %u ms.\n"
+                            "Please double check your configuration values "
+                            "for 'O2CB_HEARTBEAT_THRESHOLD'\n",
+                            slot->ds_node_num, reg->hr_dev_name, slot_dead_ms,
+                            dead_ms);
+               }
                 goto out;
         }
  
@@ -790,20 +820,24 @@ static int o2hb_highest_node(unsigned long *nodes,
         return highest;
  }
  
-static void o2hb_do_disk_heartbeat(struct o2hb_region *reg)
+static int o2hb_do_disk_heartbeat(struct o2hb_region *reg)
  {
         int i, ret, highest_node, change = 0;
         unsigned long configured_nodes[BITS_TO_LONGS(O2NM_MAX_NODES)];
         struct bio *write_bio;
         struct o2hb_bio_wait_ctxt write_wc;
  
-       if (o2nm_configured_node_map(configured_nodes, sizeof(configured_nodes)))
-               return;
+       ret = o2nm_configured_node_map(configured_nodes,
+                                      sizeof(configured_nodes));
+       if (ret) {
+               mlog_errno(ret);
+               return ret;
+       }
  
         highest_node = o2hb_highest_node(configured_nodes, O2NM_MAX_NODES);
         if (highest_node >= O2NM_MAX_NODES) {
                 mlog(ML_NOTICE, "ocfs2_heartbeat: no configured nodes found!\n");
-               return;
+               return -EINVAL;
         }
  
         /* No sense in reading the slots of nodes that don't exist
@@ -813,7 +847,7 @@ static void o2hb_do_disk_heartbeat(struct o2hb_region *reg)
         ret = o2hb_read_slots(reg, highest_node + 1);
         if (ret < 0) {
                 mlog_errno(ret);
-               return;
+               return ret;
         }
  
         /* With an up to date view of the slots, we can check that no
@@ -831,7 +865,7 @@ static void o2hb_do_disk_heartbeat(struct o2hb_region *reg)
         ret = o2hb_issue_node_write(reg, &write_bio, &write_wc);
         if (ret < 0) {
                 mlog_errno(ret);
-               return;
+               return ret;
         }
  
         i = -1;
@@ -847,6 +881,15 @@ static void o2hb_do_disk_heartbeat(struct o2hb_region *reg)
          */
         o2hb_wait_on_io(reg, &write_wc);
         bio_put(write_bio);
+       if (write_wc.wc_error) {
+               /* Do not re-arm the write timeout on I/O error - we
+                * can't be sure that the new block ever made it to
+                * disk */
+               mlog(ML_ERROR, "Write error %d on device \"%s\"\n",
+                    write_wc.wc_error, reg->hr_dev_name);
+               return write_wc.wc_error;
+       }
+
         o2hb_arm_write_timeout(reg);
  
         /* let the person who launched us know when things are steady */
@@ -854,6 +897,8 @@ static void o2hb_do_disk_heartbeat(struct o2hb_region *reg)
                 if (atomic_dec_and_test(&reg->hr_steady_iterations))
                         wake_up(&o2hb_steady_queue);
         }
+
+       return 0;
  }
  
  /* Subtract b from a, storing the result in a. a *must* have a larger
@@ -913,7 +958,10 @@ static int o2hb_thread(void *data)
                  * likely to time itself out. */
                 do_gettimeofday(&before_hb);
  
-               o2hb_do_disk_heartbeat(reg);
+               i = 0;
+               do {
+                       ret = o2hb_do_disk_heartbeat(reg);
+               } while (ret && ++i < 2);
  
                 do_gettimeofday(&after_hb);
                 elapsed_msec = o2hb_elapsed_msecs(&before_hb, &after_hb);