spapr_iommu: Make H_PUT_TCE_INDIRECT endian-safe

[mirror_qemu.git] / arch_init.c
diff --git a/arch_init.c b/arch_init.c

index 28ece769d80af793c7aa760cb98d8561411f7652..23d3feba44ab960680b216a39afe53bd5a813b14 100644 (file)
--- a/arch_init.c
+++ b/arch_init.c
@@ -24,6 +24,7 @@
  #include <stdint.h>
  #include <stdarg.h>
  #include <stdlib.h>
+#include <zlib.h>
  #ifndef _WIN32
  #include <sys/types.h>
  #include <sys/mman.h>
@@ -52,6 +53,7 @@
  #include "exec/ram_addr.h"
  #include "hw/acpi/acpi.h"
  #include "qemu/host-utils.h"
+#include "qemu/rcu_queue.h"
  
  #ifdef DEBUG_ARCH_INIT
  #define DPRINTF(fmt, ...) \
@@ -104,6 +106,8 @@ int graphic_depth = 32;
  #define QEMU_ARCH QEMU_ARCH_XTENSA
  #elif defined(TARGET_UNICORE32)
  #define QEMU_ARCH QEMU_ARCH_UNICORE32
+#elif defined(TARGET_TRICORE)
+#define QEMU_ARCH QEMU_ARCH_TRICORE
  #endif
  
  const uint32_t arch_type = QEMU_ARCH;
@@ -124,6 +128,7 @@ static uint64_t bitmap_sync_count;
  #define RAM_SAVE_FLAG_CONTINUE 0x20
  #define RAM_SAVE_FLAG_XBZRLE   0x40
  /* 0x80 is reserved in migration.h start with 0x100 next */
+#define RAM_SAVE_FLAG_COMPRESS_PAGE    0x100
  
  static struct defconfig_file {
      const char *filename;
@@ -302,15 +307,178 @@ uint64_t xbzrle_mig_pages_overflow(void)
      return acct_info.xbzrle_overflows;
  }
  
-static size_t save_block_hdr(QEMUFile *f, RAMBlock *block, ram_addr_t offset,
-                             int cont, int flag)
+/* This is the last block that we have visited serching for dirty pages
+ */
+static RAMBlock *last_seen_block;
+/* This is the last block from where we have sent data */
+static RAMBlock *last_sent_block;
+static ram_addr_t last_offset;
+static unsigned long *migration_bitmap;
+static uint64_t migration_dirty_pages;
+static uint32_t last_version;
+static bool ram_bulk_stage;
+
+struct CompressParam {
+    bool start;
+    bool done;
+    QEMUFile *file;
+    QemuMutex mutex;
+    QemuCond cond;
+    RAMBlock *block;
+    ram_addr_t offset;
+};
+typedef struct CompressParam CompressParam;
+
+struct DecompressParam {
+    bool start;
+    QemuMutex mutex;
+    QemuCond cond;
+    void *des;
+    uint8 *compbuf;
+    int len;
+};
+typedef struct DecompressParam DecompressParam;
+
+static CompressParam *comp_param;
+static QemuThread *compress_threads;
+/* comp_done_cond is used to wake up the migration thread when
+ * one of the compression threads has finished the compression.
+ * comp_done_lock is used to co-work with comp_done_cond.
+ */
+static QemuMutex *comp_done_lock;
+static QemuCond *comp_done_cond;
+/* The empty QEMUFileOps will be used by file in CompressParam */
+static const QEMUFileOps empty_ops = { };
+
+static bool compression_switch;
+static bool quit_comp_thread;
+static bool quit_decomp_thread;
+static DecompressParam *decomp_param;
+static QemuThread *decompress_threads;
+static uint8_t *compressed_data_buf;
+
+static int do_compress_ram_page(CompressParam *param);
+
+static void *do_data_compress(void *opaque)
+{
+    CompressParam *param = opaque;
+
+    while (!quit_comp_thread) {
+        qemu_mutex_lock(&param->mutex);
+        /* Re-check the quit_comp_thread in case of
+         * terminate_compression_threads is called just before
+         * qemu_mutex_lock(&param->mutex) and after
+         * while(!quit_comp_thread), re-check it here can make
+         * sure the compression thread terminate as expected.
+         */
+        while (!param->start && !quit_comp_thread) {
+            qemu_cond_wait(&param->cond, &param->mutex);
+        }
+        if (!quit_comp_thread) {
+            do_compress_ram_page(param);
+        }
+        param->start = false;
+        qemu_mutex_unlock(&param->mutex);
+
+        qemu_mutex_lock(comp_done_lock);
+        param->done = true;
+        qemu_cond_signal(comp_done_cond);
+        qemu_mutex_unlock(comp_done_lock);
+    }
+
+    return NULL;
+}
+
+static inline void terminate_compression_threads(void)
+{
+    int idx, thread_count;
+
+    thread_count = migrate_compress_threads();
+    quit_comp_thread = true;
+    for (idx = 0; idx < thread_count; idx++) {
+        qemu_mutex_lock(&comp_param[idx].mutex);
+        qemu_cond_signal(&comp_param[idx].cond);
+        qemu_mutex_unlock(&comp_param[idx].mutex);
+    }
+}
+
+void migrate_compress_threads_join(void)
+{
+    int i, thread_count;
+
+    if (!migrate_use_compression()) {
+        return;
+    }
+    terminate_compression_threads();
+    thread_count = migrate_compress_threads();
+    for (i = 0; i < thread_count; i++) {
+        qemu_thread_join(compress_threads + i);
+        qemu_fclose(comp_param[i].file);
+        qemu_mutex_destroy(&comp_param[i].mutex);
+        qemu_cond_destroy(&comp_param[i].cond);
+    }
+    qemu_mutex_destroy(comp_done_lock);
+    qemu_cond_destroy(comp_done_cond);
+    g_free(compress_threads);
+    g_free(comp_param);
+    g_free(comp_done_cond);
+    g_free(comp_done_lock);
+    compress_threads = NULL;
+    comp_param = NULL;
+    comp_done_cond = NULL;
+    comp_done_lock = NULL;
+}
+
+void migrate_compress_threads_create(void)
+{
+    int i, thread_count;
+
+    if (!migrate_use_compression()) {
+        return;
+    }
+    quit_comp_thread = false;
+    compression_switch = true;
+    thread_count = migrate_compress_threads();
+    compress_threads = g_new0(QemuThread, thread_count);
+    comp_param = g_new0(CompressParam, thread_count);
+    comp_done_cond = g_new0(QemuCond, 1);
+    comp_done_lock = g_new0(QemuMutex, 1);
+    qemu_cond_init(comp_done_cond);
+    qemu_mutex_init(comp_done_lock);
+    for (i = 0; i < thread_count; i++) {
+        /* com_param[i].file is just used as a dummy buffer to save data, set
+         * it's ops to empty.
+         */
+        comp_param[i].file = qemu_fopen_ops(NULL, &empty_ops);
+        comp_param[i].done = true;
+        qemu_mutex_init(&comp_param[i].mutex);
+        qemu_cond_init(&comp_param[i].cond);
+        qemu_thread_create(compress_threads + i, "compress",
+                           do_data_compress, comp_param + i,
+                           QEMU_THREAD_JOINABLE);
+    }
+}
+
+/**
+ * save_page_header: Write page header to wire
+ *
+ * If this is the 1st block, it also writes the block identification
+ *
+ * Returns: Number of bytes written
+ *
+ * @f: QEMUFile where to send the data
+ * @block: block that contains the page we want to send
+ * @offset: offset inside the block for the page
+ *          in the lower bits, it contains flags
+ */
+static size_t save_page_header(QEMUFile *f, RAMBlock *block, ram_addr_t offset)
  {
      size_t size;
  
-    qemu_put_be64(f, offset | cont | flag);
+    qemu_put_be64(f, offset);
      size = 8;
  
-    if (!cont) {
+    if (!(offset & RAM_SAVE_FLAG_CONTINUE)) {
          qemu_put_byte(f, strlen(block->idstr));
          qemu_put_buffer(f, (uint8_t *)block->idstr,
                          strlen(block->idstr));
@@ -319,17 +487,6 @@ static size_t save_block_hdr(QEMUFile *f, RAMBlock *block, ram_addr_t offset,
      return size;
  }
  
-/* This is the last block that we have visited serching for dirty pages
- */
-static RAMBlock *last_seen_block;
-/* This is the last block from where we have sent data */
-static RAMBlock *last_sent_block;
-static ram_addr_t last_offset;
-static unsigned long *migration_bitmap;
-static uint64_t migration_dirty_pages;
-static uint32_t last_version;
-static bool ram_bulk_stage;
-
  /* Update the xbzrle cache to reflect a page that's been sent as all 0.
   * The important thing is that a stale (not-yet-0'd) page be replaced
   * by the new data.
@@ -344,22 +501,40 @@ static void xbzrle_cache_zero_page(ram_addr_t current_addr)
  
      /* We don't care if this fails to allocate a new cache page
       * as long as it updated an old one */
-    cache_insert(XBZRLE.cache, current_addr, ZERO_TARGET_PAGE);
+    cache_insert(XBZRLE.cache, current_addr, ZERO_TARGET_PAGE,
+                 bitmap_sync_count);
  }
  
  #define ENCODING_FLAG_XBZRLE 0x1
  
+/**
+ * save_xbzrle_page: compress and send current page
+ *
+ * Returns: 1 means that we wrote the page
+ *          0 means that page is identical to the one already sent
+ *          -1 means that xbzrle would be longer than normal
+ *
+ * @f: QEMUFile where to send the data
+ * @current_data:
+ * @current_addr:
+ * @block: block that contains the page we want to send
+ * @offset: offset inside the block for the page
+ * @last_stage: if we are at the completion stage
+ * @bytes_transferred: increase it with the number of transferred bytes
+ */
  static int save_xbzrle_page(QEMUFile *f, uint8_t **current_data,
                              ram_addr_t current_addr, RAMBlock *block,
-                            ram_addr_t offset, int cont, bool last_stage)
+                            ram_addr_t offset, bool last_stage,
+                            uint64_t *bytes_transferred)
  {
-    int encoded_len = 0, bytes_sent = -1;
+    int encoded_len = 0, bytes_xbzrle;
      uint8_t *prev_cached_page;
  
-    if (!cache_is_cached(XBZRLE.cache, current_addr)) {
+    if (!cache_is_cached(XBZRLE.cache, current_addr, bitmap_sync_count)) {
          acct_info.xbzrle_cache_miss++;
          if (!last_stage) {
-            if (cache_insert(XBZRLE.cache, current_addr, *current_data) == -1) {
+            if (cache_insert(XBZRLE.cache, current_addr, *current_data,
+                             bitmap_sync_count) == -1) {
                  return -1;
              } else {
                  /* update *current_data when the page has been
@@ -399,15 +574,16 @@ static int save_xbzrle_page(QEMUFile *f, uint8_t **current_data,
      }
  
      /* Send XBZRLE based compressed page */
-    bytes_sent = save_block_hdr(f, block, offset, cont, RAM_SAVE_FLAG_XBZRLE);
+    bytes_xbzrle = save_page_header(f, block, offset | RAM_SAVE_FLAG_XBZRLE);
      qemu_put_byte(f, ENCODING_FLAG_XBZRLE);
      qemu_put_be16(f, encoded_len);
      qemu_put_buffer(f, XBZRLE.encoded_buf, encoded_len);
-    bytes_sent += encoded_len + 1 + 2;
+    bytes_xbzrle += encoded_len + 1 + 2;
      acct_info.xbzrle_pages++;
-    acct_info.xbzrle_bytes += bytes_sent;
+    acct_info.xbzrle_bytes += bytes_xbzrle;
+    *bytes_transferred += bytes_xbzrle;
  
-    return bytes_sent;
+    return 1;
  }
  
  static inline
@@ -483,20 +659,30 @@ static void migration_bitmap_sync_range(ram_addr_t start, ram_addr_t length)
  }
  
  
-/* Needs iothread lock! */
+/* Fix me: there are too many global variables used in migration process. */
+static int64_t start_time;
+static int64_t bytes_xfer_prev;
+static int64_t num_dirty_pages_period;
+static uint64_t xbzrle_cache_miss_prev;
+static uint64_t iterations_prev;
  
+static void migration_bitmap_sync_init(void)
+{
+    start_time = 0;
+    bytes_xfer_prev = 0;
+    num_dirty_pages_period = 0;
+    xbzrle_cache_miss_prev = 0;
+    iterations_prev = 0;
+}
+
+/* Called with iothread lock held, to protect ram_list.dirty_memory[] */
  static void migration_bitmap_sync(void)
  {
      RAMBlock *block;
      uint64_t num_dirty_pages_init = migration_dirty_pages;
      MigrationState *s = migrate_get_current();
-    static int64_t start_time;
-    static int64_t bytes_xfer_prev;
-    static int64_t num_dirty_pages_period;
      int64_t end_time;
      int64_t bytes_xfer_now;
-    static uint64_t xbzrle_cache_miss_prev;
-    static uint64_t iterations_prev;
  
      bitmap_sync_count++;
  
@@ -511,9 +697,12 @@ static void migration_bitmap_sync(void)
      trace_migration_bitmap_sync_start();
      address_space_sync_dirty_bitmap(&address_space_memory);
  
-    QTAILQ_FOREACH(block, &ram_list.blocks, next) {
-        migration_bitmap_sync_range(block->mr->ram_addr, block->length);
+    rcu_read_lock();
+    QLIST_FOREACH_RCU(block, &ram_list.blocks, next) {
+        migration_bitmap_sync_range(block->mr->ram_addr, block->used_length);
      }
+    rcu_read_unlock();
+
      trace_migration_bitmap_sync_end(migration_dirty_pages
                                      - num_dirty_pages_init);
      num_dirty_pages_period += migration_dirty_pages - num_dirty_pages_init;
@@ -541,7 +730,7 @@ static void migration_bitmap_sync(void)
               mig_throttle_on = false;
          }
          if (migrate_use_xbzrle()) {
-            if (iterations_prev != 0) {
+            if (iterations_prev != acct_info.iterations) {
                  acct_info.xbzrle_cache_miss_rate =
                     (double)(acct_info.xbzrle_cache_miss -
                              xbzrle_cache_miss_prev) /
@@ -555,101 +744,320 @@ static void migration_bitmap_sync(void)
          s->dirty_bytes_rate = s->dirty_pages_rate * TARGET_PAGE_SIZE;
          start_time = end_time;
          num_dirty_pages_period = 0;
-        s->dirty_sync_count = bitmap_sync_count;
      }
+    s->dirty_sync_count = bitmap_sync_count;
  }
  
-/*
+/**
+ * save_zero_page: Send the zero page to the stream
+ *
+ * Returns: Number of pages written.
+ *
+ * @f: QEMUFile where to send the data
+ * @block: block that contains the page we want to send
+ * @offset: offset inside the block for the page
+ * @p: pointer to the page
+ * @bytes_transferred: increase it with the number of transferred bytes
+ */
+static int save_zero_page(QEMUFile *f, RAMBlock *block, ram_addr_t offset,
+                          uint8_t *p, uint64_t *bytes_transferred)
+{
+    int pages = -1;
+
+    if (is_zero_range(p, TARGET_PAGE_SIZE)) {
+        acct_info.dup_pages++;
+        *bytes_transferred += save_page_header(f, block,
+                                               offset | RAM_SAVE_FLAG_COMPRESS);
+        qemu_put_byte(f, 0);
+        *bytes_transferred += 1;
+        pages = 1;
+    }
+
+    return pages;
+}
+
+/**
   * ram_save_page: Send the given page to the stream
   *
- * Returns: Number of bytes written.
+ * Returns: Number of pages written.
+ *
+ * @f: QEMUFile where to send the data
+ * @block: block that contains the page we want to send
+ * @offset: offset inside the block for the page
+ * @last_stage: if we are at the completion stage
+ * @bytes_transferred: increase it with the number of transferred bytes
   */
  static int ram_save_page(QEMUFile *f, RAMBlock* block, ram_addr_t offset,
-                         bool last_stage)
+                         bool last_stage, uint64_t *bytes_transferred)
  {
-    int bytes_sent;
-    int cont;
+    int pages = -1;
+    uint64_t bytes_xmit;
      ram_addr_t current_addr;
      MemoryRegion *mr = block->mr;
      uint8_t *p;
      int ret;
      bool send_async = true;
  
-    cont = (block == last_sent_block) ? RAM_SAVE_FLAG_CONTINUE : 0;
-
      p = memory_region_get_ram_ptr(mr) + offset;
  
      /* In doubt sent page as normal */
-    bytes_sent = -1;
+    bytes_xmit = 0;
      ret = ram_control_save_page(f, block->offset,
-                           offset, TARGET_PAGE_SIZE, &bytes_sent);
+                           offset, TARGET_PAGE_SIZE, &bytes_xmit);
+    if (bytes_xmit) {
+        *bytes_transferred += bytes_xmit;
+        pages = 1;
+    }
  
      XBZRLE_cache_lock();
  
      current_addr = block->offset + offset;
+
+    if (block == last_sent_block) {
+        offset |= RAM_SAVE_FLAG_CONTINUE;
+    }
      if (ret != RAM_SAVE_CONTROL_NOT_SUPP) {
          if (ret != RAM_SAVE_CONTROL_DELAYED) {
-            if (bytes_sent > 0) {
+            if (bytes_xmit > 0) {
                  acct_info.norm_pages++;
-            } else if (bytes_sent == 0) {
+            } else if (bytes_xmit == 0) {
                  acct_info.dup_pages++;
              }
          }
-    } else if (is_zero_range(p, TARGET_PAGE_SIZE)) {
-        acct_info.dup_pages++;
-        bytes_sent = save_block_hdr(f, block, offset, cont,
-                                    RAM_SAVE_FLAG_COMPRESS);
-        qemu_put_byte(f, 0);
-        bytes_sent++;
-        /* Must let xbzrle know, otherwise a previous (now 0'd) cached
-         * page would be stale
-         */
-        xbzrle_cache_zero_page(current_addr);
-    } else if (!ram_bulk_stage && migrate_use_xbzrle()) {
-        bytes_sent = save_xbzrle_page(f, &p, current_addr, block,
-                                      offset, cont, last_stage);
-        if (!last_stage) {
-            /* Can't send this cached data async, since the cache page
-             * might get updated before it gets to the wire
+    } else {
+        pages = save_zero_page(f, block, offset, p, bytes_transferred);
+        if (pages > 0) {
+            /* Must let xbzrle know, otherwise a previous (now 0'd) cached
+             * page would be stale
               */
-            send_async = false;
+            xbzrle_cache_zero_page(current_addr);
+        } else if (!ram_bulk_stage && migrate_use_xbzrle()) {
+            pages = save_xbzrle_page(f, &p, current_addr, block,
+                                     offset, last_stage, bytes_transferred);
+            if (!last_stage) {
+                /* Can't send this cached data async, since the cache page
+                 * might get updated before it gets to the wire
+                 */
+                send_async = false;
+            }
          }
      }
  
      /* XBZRLE overflow or normal page */
-    if (bytes_sent == -1) {
-        bytes_sent = save_block_hdr(f, block, offset, cont, RAM_SAVE_FLAG_PAGE);
+    if (pages == -1) {
+        *bytes_transferred += save_page_header(f, block,
+                                               offset | RAM_SAVE_FLAG_PAGE);
          if (send_async) {
              qemu_put_buffer_async(f, p, TARGET_PAGE_SIZE);
          } else {
              qemu_put_buffer(f, p, TARGET_PAGE_SIZE);
          }
-        bytes_sent += TARGET_PAGE_SIZE;
+        *bytes_transferred += TARGET_PAGE_SIZE;
+        pages = 1;
          acct_info.norm_pages++;
      }
  
      XBZRLE_cache_unlock();
  
+    return pages;
+}
+
+static int do_compress_ram_page(CompressParam *param)
+{
+    int bytes_sent, blen;
+    uint8_t *p;
+    RAMBlock *block = param->block;
+    ram_addr_t offset = param->offset;
+
+    p = memory_region_get_ram_ptr(block->mr) + (offset & TARGET_PAGE_MASK);
+
+    bytes_sent = save_page_header(param->file, block, offset |
+                                  RAM_SAVE_FLAG_COMPRESS_PAGE);
+    blen = qemu_put_compression_data(param->file, p, TARGET_PAGE_SIZE,
+                                     migrate_compress_level());
+    bytes_sent += blen;
+
      return bytes_sent;
  }
  
-/*
- * ram_find_and_save_block: Finds a page to send and sends it to f
+static inline void start_compression(CompressParam *param)
+{
+    param->done = false;
+    qemu_mutex_lock(&param->mutex);
+    param->start = true;
+    qemu_cond_signal(&param->cond);
+    qemu_mutex_unlock(&param->mutex);
+}
+
+static inline void start_decompression(DecompressParam *param)
+{
+    qemu_mutex_lock(&param->mutex);
+    param->start = true;
+    qemu_cond_signal(&param->cond);
+    qemu_mutex_unlock(&param->mutex);
+}
+
+static uint64_t bytes_transferred;
+
+static void flush_compressed_data(QEMUFile *f)
+{
+    int idx, len, thread_count;
+
+    if (!migrate_use_compression()) {
+        return;
+    }
+    thread_count = migrate_compress_threads();
+    for (idx = 0; idx < thread_count; idx++) {
+        if (!comp_param[idx].done) {
+            qemu_mutex_lock(comp_done_lock);
+            while (!comp_param[idx].done && !quit_comp_thread) {
+                qemu_cond_wait(comp_done_cond, comp_done_lock);
+            }
+            qemu_mutex_unlock(comp_done_lock);
+        }
+        if (!quit_comp_thread) {
+            len = qemu_put_qemu_file(f, comp_param[idx].file);
+            bytes_transferred += len;
+        }
+    }
+}
+
+static inline void set_compress_params(CompressParam *param, RAMBlock *block,
+                                       ram_addr_t offset)
+{
+    param->block = block;
+    param->offset = offset;
+}
+
+static int compress_page_with_multi_thread(QEMUFile *f, RAMBlock *block,
+                                           ram_addr_t offset,
+                                           uint64_t *bytes_transferred)
+{
+    int idx, thread_count, bytes_xmit = -1, pages = -1;
+
+    thread_count = migrate_compress_threads();
+    qemu_mutex_lock(comp_done_lock);
+    while (true) {
+        for (idx = 0; idx < thread_count; idx++) {
+            if (comp_param[idx].done) {
+                bytes_xmit = qemu_put_qemu_file(f, comp_param[idx].file);
+                set_compress_params(&comp_param[idx], block, offset);
+                start_compression(&comp_param[idx]);
+                pages = 1;
+                acct_info.norm_pages++;
+                *bytes_transferred += bytes_xmit;
+                break;
+            }
+        }
+        if (pages > 0) {
+            break;
+        } else {
+            qemu_cond_wait(comp_done_cond, comp_done_lock);
+        }
+    }
+    qemu_mutex_unlock(comp_done_lock);
+
+    return pages;
+}
+
+/**
+ * ram_save_compressed_page: compress the given page and send it to the stream
+ *
+ * Returns: Number of pages written.
+ *
+ * @f: QEMUFile where to send the data
+ * @block: block that contains the page we want to send
+ * @offset: offset inside the block for the page
+ * @last_stage: if we are at the completion stage
+ * @bytes_transferred: increase it with the number of transferred bytes
+ */
+static int ram_save_compressed_page(QEMUFile *f, RAMBlock *block,
+                                    ram_addr_t offset, bool last_stage,
+                                    uint64_t *bytes_transferred)
+{
+    int pages = -1;
+    uint64_t bytes_xmit;
+    MemoryRegion *mr = block->mr;
+    uint8_t *p;
+    int ret;
+
+    p = memory_region_get_ram_ptr(mr) + offset;
+
+    bytes_xmit = 0;
+    ret = ram_control_save_page(f, block->offset,
+                                offset, TARGET_PAGE_SIZE, &bytes_xmit);
+    if (bytes_xmit) {
+        *bytes_transferred += bytes_xmit;
+        pages = 1;
+    }
+    if (block == last_sent_block) {
+        offset |= RAM_SAVE_FLAG_CONTINUE;
+    }
+    if (ret != RAM_SAVE_CONTROL_NOT_SUPP) {
+        if (ret != RAM_SAVE_CONTROL_DELAYED) {
+            if (bytes_xmit > 0) {
+                acct_info.norm_pages++;
+            } else if (bytes_xmit == 0) {
+                acct_info.dup_pages++;
+            }
+        }
+    } else {
+        /* When starting the process of a new block, the first page of
+         * the block should be sent out before other pages in the same
+         * block, and all the pages in last block should have been sent
+         * out, keeping this order is important, because the 'cont' flag
+         * is used to avoid resending the block name.
+         */
+        if (block != last_sent_block) {
+            flush_compressed_data(f);
+            pages = save_zero_page(f, block, offset, p, bytes_transferred);
+            if (pages == -1) {
+                set_compress_params(&comp_param[0], block, offset);
+                /* Use the qemu thread to compress the data to make sure the
+                 * first page is sent out before other pages
+                 */
+                bytes_xmit = do_compress_ram_page(&comp_param[0]);
+                acct_info.norm_pages++;
+                qemu_put_qemu_file(f, comp_param[0].file);
+                *bytes_transferred += bytes_xmit;
+                pages = 1;
+            }
+        } else {
+            pages = save_zero_page(f, block, offset, p, bytes_transferred);
+            if (pages == -1) {
+                pages = compress_page_with_multi_thread(f, block, offset,
+                                                        bytes_transferred);
+            }
+        }
+    }
+
+    return pages;
+}
+
+/**
+ * ram_find_and_save_block: Finds a dirty page and sends it to f
+ *
+ * Called within an RCU critical section.
   *
- * Returns:  The number of bytes written.
+ * Returns:  The number of pages written
   *           0 means no dirty pages
+ *
+ * @f: QEMUFile where to send the data
+ * @last_stage: if we are at the completion stage
+ * @bytes_transferred: increase it with the number of transferred bytes
   */
  
-static int ram_find_and_save_block(QEMUFile *f, bool last_stage)
+static int ram_find_and_save_block(QEMUFile *f, bool last_stage,
+                                   uint64_t *bytes_transferred)
  {
      RAMBlock *block = last_seen_block;
      ram_addr_t offset = last_offset;
      bool complete_round = false;
-    int bytes_sent = 0;
+    int pages = 0;
      MemoryRegion *mr;
  
      if (!block)
-        block = QTAILQ_FIRST(&ram_list.blocks);
+        block = QLIST_FIRST_RCU(&ram_list.blocks);
  
      while (true) {
          mr = block->mr;
@@ -658,32 +1066,44 @@ static int ram_find_and_save_block(QEMUFile *f, bool last_stage)
              offset >= last_offset) {
              break;
          }
-        if (offset >= block->length) {
+        if (offset >= block->used_length) {
              offset = 0;
-            block = QTAILQ_NEXT(block, next);
+            block = QLIST_NEXT_RCU(block, next);
              if (!block) {
-                block = QTAILQ_FIRST(&ram_list.blocks);
+                block = QLIST_FIRST_RCU(&ram_list.blocks);
                  complete_round = true;
                  ram_bulk_stage = false;
+                if (migrate_use_xbzrle()) {
+                    /* If xbzrle is on, stop using the data compression at this
+                     * point. In theory, xbzrle can do better than compression.
+                     */
+                    flush_compressed_data(f);
+                    compression_switch = false;
+                }
              }
          } else {
-            bytes_sent = ram_save_page(f, block, offset, last_stage);
+            if (compression_switch && migrate_use_compression()) {
+                pages = ram_save_compressed_page(f, block, offset, last_stage,
+                                                 bytes_transferred);
+            } else {
+                pages = ram_save_page(f, block, offset, last_stage,
+                                      bytes_transferred);
+            }
  
              /* if page is unmodified, continue to the next */
-            if (bytes_sent > 0) {
+            if (pages > 0) {
                  last_sent_block = block;
                  break;
              }
          }
      }
+
      last_seen_block = block;
      last_offset = offset;
  
-    return bytes_sent;
+    return pages;
  }
  
-static uint64_t bytes_transferred;
-
  void acct_update_position(QEMUFile *f, size_t size, bool zero)
  {
      uint64_t pages = size / TARGET_PAGE_SIZE;
@@ -716,9 +1136,10 @@ uint64_t ram_bytes_total(void)
      RAMBlock *block;
      uint64_t total = 0;
  
-    QTAILQ_FOREACH(block, &ram_list.blocks, next)
-        total += block->length;
-
+    rcu_read_lock();
+    QLIST_FOREACH_RCU(block, &ram_list.blocks, next)
+        total += block->used_length;
+    rcu_read_unlock();
      return total;
  }
  
@@ -764,6 +1185,13 @@ static void reset_ram_globals(void)
  
  #define MAX_WAIT 50 /* ms, half buffered_file limit */
  
+
+/* Each of ram_save_setup, ram_save_iterate and ram_save_complete has
+ * long-running RCU critical section.  When rcu-reclaims in the code
+ * start to become numerous it will be necessary to reduce the
+ * granularity of these critical sections.
+ */
+
  static int ram_save_setup(QEMUFile *f, void *opaque)
  {
      RAMBlock *block;
@@ -772,6 +1200,7 @@ static int ram_save_setup(QEMUFile *f, void *opaque)
      mig_throttle_on = false;
      dirty_rate_high_cnt = 0;
      bitmap_sync_count = 0;
+    migration_bitmap_sync_init();
  
      if (migrate_use_xbzrle()) {
          XBZRLE_cache_lock();
@@ -803,8 +1232,10 @@ static int ram_save_setup(QEMUFile *f, void *opaque)
          acct_clear();
      }
  
+    /* iothread lock needed for ram_list.dirty_memory[] */
      qemu_mutex_lock_iothread();
      qemu_mutex_lock_ramlist();
+    rcu_read_lock();
      bytes_transferred = 0;
      reset_ram_globals();
  
@@ -816,27 +1247,22 @@ static int ram_save_setup(QEMUFile *f, void *opaque)
       * Count the total number of pages used by ram blocks not including any
       * gaps due to alignment or unplugs.
       */
-    migration_dirty_pages = 0;
-    QTAILQ_FOREACH(block, &ram_list.blocks, next) {
-        uint64_t block_pages;
-
-        block_pages = block->length >> TARGET_PAGE_BITS;
-        migration_dirty_pages += block_pages;
-    }
+    migration_dirty_pages = ram_bytes_total() >> TARGET_PAGE_BITS;
  
      memory_global_dirty_log_start();
      migration_bitmap_sync();
+    qemu_mutex_unlock_ramlist();
      qemu_mutex_unlock_iothread();
  
      qemu_put_be64(f, ram_bytes_total() | RAM_SAVE_FLAG_MEM_SIZE);
  
-    QTAILQ_FOREACH(block, &ram_list.blocks, next) {
+    QLIST_FOREACH_RCU(block, &ram_list.blocks, next) {
          qemu_put_byte(f, strlen(block->idstr));
          qemu_put_buffer(f, (uint8_t *)block->idstr, strlen(block->idstr));
-        qemu_put_be64(f, block->length);
+        qemu_put_be64(f, block->used_length);
      }
  
-    qemu_mutex_unlock_ramlist();
+    rcu_read_unlock();
  
      ram_control_before_iterate(f, RAM_CONTROL_SETUP);
      ram_control_after_iterate(f, RAM_CONTROL_SETUP);
@@ -851,27 +1277,29 @@ static int ram_save_iterate(QEMUFile *f, void *opaque)
      int ret;
      int i;
      int64_t t0;
-    int total_sent = 0;
-
-    qemu_mutex_lock_ramlist();
+    int pages_sent = 0;
  
+    rcu_read_lock();
      if (ram_list.version != last_version) {
          reset_ram_globals();
      }
  
+    /* Read version before ram_list.blocks */
+    smp_rmb();
+
      ram_control_before_iterate(f, RAM_CONTROL_ROUND);
  
      t0 = qemu_clock_get_ns(QEMU_CLOCK_REALTIME);
      i = 0;
      while ((ret = qemu_file_rate_limit(f)) == 0) {
-        int bytes_sent;
+        int pages;
  
-        bytes_sent = ram_find_and_save_block(f, false);
-        /* no more blocks to sent */
-        if (bytes_sent == 0) {
+        pages = ram_find_and_save_block(f, false, &bytes_transferred);
+        /* no more pages to sent */
+        if (pages == 0) {
              break;
          }
-        total_sent += bytes_sent;
+        pages_sent += pages;
          acct_info.iterations++;
          check_guest_throttling();
          /* we want to check in the 1st loop, just in case it was the 1st time
@@ -889,8 +1317,8 @@ static int ram_save_iterate(QEMUFile *f, void *opaque)
          }
          i++;
      }
-
-    qemu_mutex_unlock_ramlist();
+    flush_compressed_data(f);
+    rcu_read_unlock();
  
      /*
       * Must occur before EOS (or any QEMUFile operation)
@@ -898,12 +1326,6 @@ static int ram_save_iterate(QEMUFile *f, void *opaque)
       */
      ram_control_after_iterate(f, RAM_CONTROL_ROUND);
  
-    bytes_transferred += total_sent;
-
-    /*
-     * Do not count these 8 bytes into total_sent, so that we can
-     * return 0 if no page had been dirtied.
-     */
      qemu_put_be64(f, RAM_SAVE_FLAG_EOS);
      bytes_transferred += 8;
  
@@ -912,12 +1334,14 @@ static int ram_save_iterate(QEMUFile *f, void *opaque)
          return ret;
      }
  
-    return total_sent;
+    return pages_sent;
  }
  
+/* Called with iothread lock */
  static int ram_save_complete(QEMUFile *f, void *opaque)
  {
-    qemu_mutex_lock_ramlist();
+    rcu_read_lock();
+
      migration_bitmap_sync();
  
      ram_control_before_iterate(f, RAM_CONTROL_FINISH);
@@ -926,20 +1350,20 @@ static int ram_save_complete(QEMUFile *f, void *opaque)
  
      /* flush all remaining blocks regardless of rate limiting */
      while (true) {
-        int bytes_sent;
+        int pages;
  
-        bytes_sent = ram_find_and_save_block(f, true);
+        pages = ram_find_and_save_block(f, true, &bytes_transferred);
          /* no more blocks to sent */
-        if (bytes_sent == 0) {
+        if (pages == 0) {
              break;
          }
-        bytes_transferred += bytes_sent;
      }
  
+    flush_compressed_data(f);
      ram_control_after_iterate(f, RAM_CONTROL_FINISH);
      migration_end();
  
-    qemu_mutex_unlock_ramlist();
+    rcu_read_unlock();
      qemu_put_be64(f, RAM_SAVE_FLAG_EOS);
  
      return 0;
@@ -953,7 +1377,9 @@ static uint64_t ram_save_pending(QEMUFile *f, void *opaque, uint64_t max_size)
  
      if (remaining_size < max_size) {
          qemu_mutex_lock_iothread();
+        rcu_read_lock();
          migration_bitmap_sync();
+        rcu_read_unlock();
          qemu_mutex_unlock_iothread();
          remaining_size = ram_save_remaining() * TARGET_PAGE_SIZE;
      }
@@ -995,6 +1421,9 @@ static int load_xbzrle(QEMUFile *f, ram_addr_t addr, void *host)
      return 0;
  }
  
+/* Must be called from within a rcu critical section.
+ * Returns a pointer from within the RCU-protected ram_list.
+ */
  static inline void *host_from_stream_offset(QEMUFile *f,
                                              ram_addr_t offset,
                                              int flags)
@@ -1004,7 +1433,7 @@ static inline void *host_from_stream_offset(QEMUFile *f,
      uint8_t len;
  
      if (flags & RAM_SAVE_FLAG_CONTINUE) {
-        if (!block) {
+        if (!block || block->max_length <= offset) {
              error_report("Ack, bad migration stream!");
              return NULL;
          }
@@ -1016,9 +1445,11 @@ static inline void *host_from_stream_offset(QEMUFile *f,
      qemu_get_buffer(f, (uint8_t *)id, len);
      id[len] = 0;
  
-    QTAILQ_FOREACH(block, &ram_list.blocks, next) {
-        if (!strncmp(id, block->idstr, sizeof(id)))
+    QLIST_FOREACH_RCU(block, &ram_list.blocks, next) {
+        if (!strncmp(id, block->idstr, sizeof(id)) &&
+            block->max_length > offset) {
              return memory_region_get_ram_ptr(block->mr) + offset;
+        }
      }
  
      error_report("Can't find block %s!", id);
@@ -1036,11 +1467,104 @@ void ram_handle_compressed(void *host, uint8_t ch, uint64_t size)
      }
  }
  
+static void *do_data_decompress(void *opaque)
+{
+    DecompressParam *param = opaque;
+    unsigned long pagesize;
+
+    while (!quit_decomp_thread) {
+        qemu_mutex_lock(&param->mutex);
+        while (!param->start && !quit_decomp_thread) {
+            qemu_cond_wait(&param->cond, &param->mutex);
+            pagesize = TARGET_PAGE_SIZE;
+            if (!quit_decomp_thread) {
+                /* uncompress() will return failed in some case, especially
+                 * when the page is dirted when doing the compression, it's
+                 * not a problem because the dirty page will be retransferred
+                 * and uncompress() won't break the data in other pages.
+                 */
+                uncompress((Bytef *)param->des, &pagesize,
+                           (const Bytef *)param->compbuf, param->len);
+            }
+            param->start = false;
+        }
+        qemu_mutex_unlock(&param->mutex);
+    }
+
+    return NULL;
+}
+
+void migrate_decompress_threads_create(void)
+{
+    int i, thread_count;
+
+    thread_count = migrate_decompress_threads();
+    decompress_threads = g_new0(QemuThread, thread_count);
+    decomp_param = g_new0(DecompressParam, thread_count);
+    compressed_data_buf = g_malloc0(compressBound(TARGET_PAGE_SIZE));
+    quit_decomp_thread = false;
+    for (i = 0; i < thread_count; i++) {
+        qemu_mutex_init(&decomp_param[i].mutex);
+        qemu_cond_init(&decomp_param[i].cond);
+        decomp_param[i].compbuf = g_malloc0(compressBound(TARGET_PAGE_SIZE));
+        qemu_thread_create(decompress_threads + i, "decompress",
+                           do_data_decompress, decomp_param + i,
+                           QEMU_THREAD_JOINABLE);
+    }
+}
+
+void migrate_decompress_threads_join(void)
+{
+    int i, thread_count;
+
+    quit_decomp_thread = true;
+    thread_count = migrate_decompress_threads();
+    for (i = 0; i < thread_count; i++) {
+        qemu_mutex_lock(&decomp_param[i].mutex);
+        qemu_cond_signal(&decomp_param[i].cond);
+        qemu_mutex_unlock(&decomp_param[i].mutex);
+    }
+    for (i = 0; i < thread_count; i++) {
+        qemu_thread_join(decompress_threads + i);
+        qemu_mutex_destroy(&decomp_param[i].mutex);
+        qemu_cond_destroy(&decomp_param[i].cond);
+        g_free(decomp_param[i].compbuf);
+    }
+    g_free(decompress_threads);
+    g_free(decomp_param);
+    g_free(compressed_data_buf);
+    decompress_threads = NULL;
+    decomp_param = NULL;
+    compressed_data_buf = NULL;
+}
+
+static void decompress_data_with_multi_threads(uint8_t *compbuf,
+                                               void *host, int len)
+{
+    int idx, thread_count;
+
+    thread_count = migrate_decompress_threads();
+    while (true) {
+        for (idx = 0; idx < thread_count; idx++) {
+            if (!decomp_param[idx].start) {
+                memcpy(decomp_param[idx].compbuf, compbuf, len);
+                decomp_param[idx].des = host;
+                decomp_param[idx].len = len;
+                start_decompression(&decomp_param[idx]);
+                break;
+            }
+        }
+        if (idx < thread_count) {
+            break;
+        }
+    }
+}
+
  static int ram_load(QEMUFile *f, void *opaque, int version_id)
  {
-    ram_addr_t addr;
-    int flags, ret = 0;
+    int flags = 0, ret = 0;
      static uint64_t seq_iter;
+    int len = 0;
  
      seq_iter++;
  
@@ -1048,34 +1572,45 @@ static int ram_load(QEMUFile *f, void *opaque, int version_id)
          ret = -EINVAL;
      }
  
-    while (!ret) {
-        addr = qemu_get_be64(f);
+    /* This RCU critical section can be very long running.
+     * When RCU reclaims in the code start to become numerous,
+     * it will be necessary to reduce the granularity of this
+     * critical section.
+     */
+    rcu_read_lock();
+    while (!ret && !(flags & RAM_SAVE_FLAG_EOS)) {
+        ram_addr_t addr, total_ram_bytes;
+        void *host;
+        uint8_t ch;
  
+        addr = qemu_get_be64(f);
          flags = addr & ~TARGET_PAGE_MASK;
          addr &= TARGET_PAGE_MASK;
  
-        if (flags & RAM_SAVE_FLAG_MEM_SIZE) {
+        switch (flags & ~RAM_SAVE_FLAG_CONTINUE) {
+        case RAM_SAVE_FLAG_MEM_SIZE:
              /* Synchronize RAM block list */
-            char id[256];
-            ram_addr_t length;
-            ram_addr_t total_ram_bytes = addr;
-
-            while (total_ram_bytes) {
+            total_ram_bytes = addr;
+            while (!ret && total_ram_bytes) {
                  RAMBlock *block;
                  uint8_t len;
+                char id[256];
+                ram_addr_t length;
  
                  len = qemu_get_byte(f);
                  qemu_get_buffer(f, (uint8_t *)id, len);
                  id[len] = 0;
                  length = qemu_get_be64(f);
  
-                QTAILQ_FOREACH(block, &ram_list.blocks, next) {
+                QLIST_FOREACH_RCU(block, &ram_list.blocks, next) {
                      if (!strncmp(id, block->idstr, sizeof(id))) {
-                        if (block->length != length) {
-                            error_report("Length mismatch: %s: 0x" RAM_ADDR_FMT
-                                         " in != 0x" RAM_ADDR_FMT, id, length,
-                                         block->length);
-                            ret =  -EINVAL;
+                        if (length != block->used_length) {
+                            Error *local_err = NULL;
+
+                            ret = qemu_ram_resize(block->offset, length, &local_err);
+                            if (local_err) {
+                                error_report_err(local_err);
+                            }
                          }
                          break;
                      }
@@ -1086,63 +1621,78 @@ static int ram_load(QEMUFile *f, void *opaque, int version_id)
                                   "accept migration", id);
                      ret = -EINVAL;
                  }
-                if (ret) {
-                    break;
-                }
  
                  total_ram_bytes -= length;
              }
-        } else if (flags & RAM_SAVE_FLAG_COMPRESS) {
-            void *host;
-            uint8_t ch;
-
+            break;
+        case RAM_SAVE_FLAG_COMPRESS:
              host = host_from_stream_offset(f, addr, flags);
              if (!host) {
                  error_report("Illegal RAM offset " RAM_ADDR_FMT, addr);
                  ret = -EINVAL;
                  break;
              }
-
              ch = qemu_get_byte(f);
              ram_handle_compressed(host, ch, TARGET_PAGE_SIZE);
-        } else if (flags & RAM_SAVE_FLAG_PAGE) {
-            void *host;
-
+            break;
+        case RAM_SAVE_FLAG_PAGE:
              host = host_from_stream_offset(f, addr, flags);
              if (!host) {
                  error_report("Illegal RAM offset " RAM_ADDR_FMT, addr);
                  ret = -EINVAL;
                  break;
              }
-
              qemu_get_buffer(f, host, TARGET_PAGE_SIZE);
-        } else if (flags & RAM_SAVE_FLAG_XBZRLE) {
-            void *host = host_from_stream_offset(f, addr, flags);
+            break;
+        case RAM_SAVE_FLAG_COMPRESS_PAGE:
+            host = host_from_stream_offset(f, addr, flags);
              if (!host) {
-                error_report("Illegal RAM offset " RAM_ADDR_FMT, addr);
+                error_report("Invalid RAM offset " RAM_ADDR_FMT, addr);
                  ret = -EINVAL;
                  break;
              }
  
+            len = qemu_get_be32(f);
+            if (len < 0 || len > compressBound(TARGET_PAGE_SIZE)) {
+                error_report("Invalid compressed data length: %d", len);
+                ret = -EINVAL;
+                break;
+            }
+            qemu_get_buffer(f, compressed_data_buf, len);
+            decompress_data_with_multi_threads(compressed_data_buf, host, len);
+            break;
+        case RAM_SAVE_FLAG_XBZRLE:
+            host = host_from_stream_offset(f, addr, flags);
+            if (!host) {
+                error_report("Illegal RAM offset " RAM_ADDR_FMT, addr);
+                ret = -EINVAL;
+                break;
+            }
              if (load_xbzrle(f, addr, host) < 0) {
                  error_report("Failed to decompress XBZRLE page at "
                               RAM_ADDR_FMT, addr);
                  ret = -EINVAL;
                  break;
              }
-        } else if (flags & RAM_SAVE_FLAG_HOOK) {
-            ram_control_load_hook(f, flags);
-        } else if (flags & RAM_SAVE_FLAG_EOS) {
-            /* normal exit */
              break;
-        } else {
-            error_report("Unknown migration flags: %#x", flags);
-            ret = -EINVAL;
+        case RAM_SAVE_FLAG_EOS:
+            /* normal exit */
              break;
+        default:
+            if (flags & RAM_SAVE_FLAG_HOOK) {
+                ram_control_load_hook(f, flags);
+            } else {
+                error_report("Unknown combination of migration flags: %#x",
+                             flags);
+                ret = -EINVAL;
+            }
+        }
+        if (!ret) {
+            ret = qemu_file_get_error(f);
          }
-        ret = qemu_file_get_error(f);
      }
  
+    rcu_read_unlock();
      DPRINTF("Completed load of VM with exit code %d seq iteration "
              "%" PRIu64 "\n", ret, seq_iter);
      return ret;
@@ -1335,11 +1885,6 @@ void cpudef_init(void)
  #endif
  }
  
-int tcg_available(void)
-{
-    return 1;
-}
-
  int kvm_available(void)
  {
  #ifdef CONFIG_KVM