qobject: Replace qobject_incref/QINCREF qobject_decref/QDECREF

[mirror_qemu.git] / block / qed.c
diff --git a/block/qed.c b/block/qed.c

index 86cad2188c0f40e213951bb7ed19d95fdda96420..1db8eaf241687970ceb2035d2050adae15669608 100644 (file)
--- a/block/qed.c
+++ b/block/qed.c
@@ -16,10 +16,15 @@
  #include "qapi/error.h"
  #include "qemu/timer.h"
  #include "qemu/bswap.h"
+#include "qemu/option.h"
  #include "trace.h"
  #include "qed.h"
-#include "qapi/qmp/qerror.h"
  #include "sysemu/block-backend.h"
+#include "qapi/qmp/qdict.h"
+#include "qapi/qobject-input-visitor.h"
+#include "qapi/qapi-visit-block-core.h"
+
+static QemuOptsList qed_create_opts;
  
  static int bdrv_qed_probe(const uint8_t *buf, int buf_size,
                            const char *filename)
@@ -93,6 +98,8 @@ int qed_write_header_sync(BDRVQEDState *s)
   *
   * This function only updates known header fields in-place and does not affect
   * extra data after the QED header.
+ *
+ * No new allocating reqs can start while this function runs.
   */
  static int coroutine_fn qed_write_header(BDRVQEDState *s)
  {
@@ -109,6 +116,8 @@ static int coroutine_fn qed_write_header(BDRVQEDState *s)
      QEMUIOVector qiov;
      int ret;
  
+    assert(s->allocating_acb || s->allocating_write_reqs_plugged);
+
      buf = qemu_blockalign(s->bs, len);
      iov = (struct iovec) {
          .iov_base = buf,
@@ -219,6 +228,8 @@ static int qed_read_string(BdrvChild *file, uint64_t offset, size_t n,
   * This function only produces the offset where the new clusters should be
   * written.  It updates BDRVQEDState but does not make any changes to the image
   * file.
+ *
+ * Called with table_lock held.
   */
  static uint64_t qed_alloc_clusters(BDRVQEDState *s, unsigned int n)
  {
@@ -236,6 +247,8 @@ QEDTable *qed_alloc_table(BDRVQEDState *s)
  
  /**
   * Allocate a new zeroed L2 table
+ *
+ * Called with table_lock held.
   */
  static CachedL2Table *qed_new_l2_table(BDRVQEDState *s)
  {
@@ -249,19 +262,32 @@ static CachedL2Table *qed_new_l2_table(BDRVQEDState *s)
      return l2_table;
  }
  
-static void qed_plug_allocating_write_reqs(BDRVQEDState *s)
+static bool qed_plug_allocating_write_reqs(BDRVQEDState *s)
  {
+    qemu_co_mutex_lock(&s->table_lock);
+
+    /* No reentrancy is allowed.  */
      assert(!s->allocating_write_reqs_plugged);
+    if (s->allocating_acb != NULL) {
+        /* Another allocating write came concurrently.  This cannot happen
+         * from bdrv_qed_co_drain_begin, but it can happen when the timer runs.
+         */
+        qemu_co_mutex_unlock(&s->table_lock);
+        return false;
+    }
  
      s->allocating_write_reqs_plugged = true;
+    qemu_co_mutex_unlock(&s->table_lock);
+    return true;
  }
  
  static void qed_unplug_allocating_write_reqs(BDRVQEDState *s)
  {
+    qemu_co_mutex_lock(&s->table_lock);
      assert(s->allocating_write_reqs_plugged);
-
      s->allocating_write_reqs_plugged = false;
-    qemu_co_enter_next(&s->allocating_write_reqs);
+    qemu_co_queue_next(&s->allocating_write_reqs);
+    qemu_co_mutex_unlock(&s->table_lock);
  }
  
  static void coroutine_fn qed_need_check_timer_entry(void *opaque)
@@ -269,17 +295,14 @@ static void coroutine_fn qed_need_check_timer_entry(void *opaque)
      BDRVQEDState *s = opaque;
      int ret;
  
-    /* The timer should only fire when allocating writes have drained */
-    assert(!s->allocating_acb);
-
      trace_qed_need_check_timer_cb(s);
  
-    qed_acquire(s);
-    qed_plug_allocating_write_reqs(s);
+    if (!qed_plug_allocating_write_reqs(s)) {
+        return;
+    }
  
      /* Ensure writes are on disk before clearing flag */
      ret = bdrv_co_flush(s->bs->file->bs);
-    qed_release(s);
      if (ret < 0) {
          qed_unplug_allocating_write_reqs(s);
          return;
@@ -301,16 +324,6 @@ static void qed_need_check_timer_cb(void *opaque)
      qemu_coroutine_enter(co);
  }
  
-void qed_acquire(BDRVQEDState *s)
-{
-    aio_context_acquire(bdrv_get_aio_context(s->bs));
-}
-
-void qed_release(BDRVQEDState *s)
-{
-    aio_context_release(bdrv_get_aio_context(s->bs));
-}
-
  static void qed_start_need_check_timer(BDRVQEDState *s)
  {
      trace_qed_start_need_check_timer(s);
@@ -350,7 +363,7 @@ static void bdrv_qed_attach_aio_context(BlockDriverState *bs,
      }
  }
  
-static void bdrv_qed_drain(BlockDriverState *bs)
+static void coroutine_fn bdrv_qed_co_drain_begin(BlockDriverState *bs)
  {
      BDRVQEDState *s = bs->opaque;
  
@@ -359,20 +372,28 @@ static void bdrv_qed_drain(BlockDriverState *bs)
       */
      if (s->need_check_timer && timer_pending(s->need_check_timer)) {
          qed_cancel_need_check_timer(s);
-        qed_need_check_timer_cb(s);
+        qed_need_check_timer_entry(s);
      }
  }
  
-static int bdrv_qed_do_open(BlockDriverState *bs, QDict *options, int flags,
-                            Error **errp)
+static void bdrv_qed_init_state(BlockDriverState *bs)
  {
      BDRVQEDState *s = bs->opaque;
-    QEDHeader le_header;
-    int64_t file_size;
-    int ret;
  
+    memset(s, 0, sizeof(BDRVQEDState));
      s->bs = bs;
+    qemu_co_mutex_init(&s->table_lock);
      qemu_co_queue_init(&s->allocating_write_reqs);
+}
+
+/* Called with table_lock held.  */
+static int coroutine_fn bdrv_qed_do_open(BlockDriverState *bs, QDict *options,
+                                         int flags, Error **errp)
+{
+    BDRVQEDState *s = bs->opaque;
+    QEDHeader le_header;
+    int64_t file_size;
+    int ret;
  
      ret = bdrv_pread(bs->file, 0, &le_header, sizeof(le_header));
      if (ret < 0) {
@@ -498,16 +519,50 @@ out:
      return ret;
  }
  
+typedef struct QEDOpenCo {
+    BlockDriverState *bs;
+    QDict *options;
+    int flags;
+    Error **errp;
+    int ret;
+} QEDOpenCo;
+
+static void coroutine_fn bdrv_qed_open_entry(void *opaque)
+{
+    QEDOpenCo *qoc = opaque;
+    BDRVQEDState *s = qoc->bs->opaque;
+
+    qemu_co_mutex_lock(&s->table_lock);
+    qoc->ret = bdrv_qed_do_open(qoc->bs, qoc->options, qoc->flags, qoc->errp);
+    qemu_co_mutex_unlock(&s->table_lock);
+}
+
  static int bdrv_qed_open(BlockDriverState *bs, QDict *options, int flags,
                           Error **errp)
  {
+    QEDOpenCo qoc = {
+        .bs = bs,
+        .options = options,
+        .flags = flags,
+        .errp = errp,
+        .ret = -EINPROGRESS
+    };
+
      bs->file = bdrv_open_child(NULL, options, "file", bs, &child_file,
                                 false, errp);
      if (!bs->file) {
          return -EINVAL;
      }
  
-    return bdrv_qed_do_open(bs, options, flags, errp);
+    bdrv_qed_init_state(bs);
+    if (qemu_in_coroutine()) {
+        bdrv_qed_open_entry(&qoc);
+    } else {
+        qemu_coroutine_enter(qemu_coroutine_create(bdrv_qed_open_entry, &qoc));
+        BDRV_POLL_WHILE(bs, qoc.ret == -EINPROGRESS);
+    }
+    BDRV_POLL_WHILE(bs, qoc.ret == -EINPROGRESS);
+    return qoc.ret;
  }
  
  static void bdrv_qed_refresh_limits(BlockDriverState *bs, Error **errp)
@@ -544,57 +599,95 @@ static void bdrv_qed_close(BlockDriverState *bs)
      qemu_vfree(s->l1_table);
  }
  
-static int qed_create(const char *filename, uint32_t cluster_size,
-                      uint64_t image_size, uint32_t table_size,
-                      const char *backing_file, const char *backing_fmt,
-                      QemuOpts *opts, Error **errp)
+static int coroutine_fn bdrv_qed_co_create(BlockdevCreateOptions *opts,
+                                           Error **errp)
  {
-    QEDHeader header = {
-        .magic = QED_MAGIC,
-        .cluster_size = cluster_size,
-        .table_size = table_size,
-        .header_size = 1,
-        .features = 0,
-        .compat_features = 0,
-        .l1_table_offset = cluster_size,
-        .image_size = image_size,
-    };
+    BlockdevCreateOptionsQed *qed_opts;
+    BlockBackend *blk = NULL;
+    BlockDriverState *bs = NULL;
+
+    QEDHeader header;
      QEDHeader le_header;
      uint8_t *l1_table = NULL;
-    size_t l1_size = header.cluster_size * header.table_size;
-    Error *local_err = NULL;
+    size_t l1_size;
      int ret = 0;
-    BlockBackend *blk;
  
-    ret = bdrv_create_file(filename, opts, &local_err);
-    if (ret < 0) {
-        error_propagate(errp, local_err);
-        return ret;
+    assert(opts->driver == BLOCKDEV_DRIVER_QED);
+    qed_opts = &opts->u.qed;
+
+    /* Validate options and set default values */
+    if (!qed_opts->has_cluster_size) {
+        qed_opts->cluster_size = QED_DEFAULT_CLUSTER_SIZE;
+    }
+    if (!qed_opts->has_table_size) {
+        qed_opts->table_size = QED_DEFAULT_TABLE_SIZE;
      }
  
-    blk = blk_new_open(filename, NULL, NULL,
-                       BDRV_O_RDWR | BDRV_O_RESIZE | BDRV_O_PROTOCOL,
-                       &local_err);
-    if (blk == NULL) {
-        error_propagate(errp, local_err);
+    if (!qed_is_cluster_size_valid(qed_opts->cluster_size)) {
+        error_setg(errp, "QED cluster size must be within range [%u, %u] "
+                         "and power of 2",
+                   QED_MIN_CLUSTER_SIZE, QED_MAX_CLUSTER_SIZE);
+        return -EINVAL;
+    }
+    if (!qed_is_table_size_valid(qed_opts->table_size)) {
+        error_setg(errp, "QED table size must be within range [%u, %u] "
+                         "and power of 2",
+                   QED_MIN_TABLE_SIZE, QED_MAX_TABLE_SIZE);
+        return -EINVAL;
+    }
+    if (!qed_is_image_size_valid(qed_opts->size, qed_opts->cluster_size,
+                                 qed_opts->table_size))
+    {
+        error_setg(errp, "QED image size must be a non-zero multiple of "
+                         "cluster size and less than %" PRIu64 " bytes",
+                   qed_max_image_size(qed_opts->cluster_size,
+                                      qed_opts->table_size));
+        return -EINVAL;
+    }
+
+    /* Create BlockBackend to write to the image */
+    bs = bdrv_open_blockdev_ref(qed_opts->file, errp);
+    if (bs == NULL) {
          return -EIO;
      }
  
+    blk = blk_new(BLK_PERM_WRITE | BLK_PERM_RESIZE, BLK_PERM_ALL);
+    ret = blk_insert_bs(blk, bs, errp);
+    if (ret < 0) {
+        goto out;
+    }
      blk_set_allow_write_beyond_eof(blk, true);
  
+    /* Prepare image format */
+    header = (QEDHeader) {
+        .magic = QED_MAGIC,
+        .cluster_size = qed_opts->cluster_size,
+        .table_size = qed_opts->table_size,
+        .header_size = 1,
+        .features = 0,
+        .compat_features = 0,
+        .l1_table_offset = qed_opts->cluster_size,
+        .image_size = qed_opts->size,
+    };
+
+    l1_size = header.cluster_size * header.table_size;
+
      /* File must start empty and grow, check truncate is supported */
      ret = blk_truncate(blk, 0, PREALLOC_MODE_OFF, errp);
      if (ret < 0) {
          goto out;
      }
  
-    if (backing_file) {
+    if (qed_opts->has_backing_file) {
          header.features |= QED_F_BACKING_FILE;
          header.backing_filename_offset = sizeof(le_header);
-        header.backing_filename_size = strlen(backing_file);
+        header.backing_filename_size = strlen(qed_opts->backing_file);
  
-        if (qed_fmt_is_raw(backing_fmt)) {
-            header.features |= QED_F_BACKING_FORMAT_NO_PROBE;
+        if (qed_opts->has_backing_fmt) {
+            const char *backing_fmt = BlockdevDriver_str(qed_opts->backing_fmt);
+            if (qed_fmt_is_raw(backing_fmt)) {
+                header.features |= QED_F_BACKING_FORMAT_NO_PROBE;
+            }
          }
      }
  
@@ -603,7 +696,7 @@ static int qed_create(const char *filename, uint32_t cluster_size,
      if (ret < 0) {
          goto out;
      }
-    ret = blk_pwrite(blk, sizeof(le_header), backing_file,
+    ret = blk_pwrite(blk, sizeof(le_header), qed_opts->backing_file,
                       header.backing_filename_size, 0);
      if (ret < 0) {
          goto out;
@@ -619,124 +712,129 @@ static int qed_create(const char *filename, uint32_t cluster_size,
  out:
      g_free(l1_table);
      blk_unref(blk);
+    bdrv_unref(bs);
      return ret;
  }
  
-static int bdrv_qed_create(const char *filename, QemuOpts *opts, Error **errp)
+static int coroutine_fn bdrv_qed_co_create_opts(const char *filename,
+                                                QemuOpts *opts,
+                                                Error **errp)
  {
-    uint64_t image_size = 0;
-    uint32_t cluster_size = QED_DEFAULT_CLUSTER_SIZE;
-    uint32_t table_size = QED_DEFAULT_TABLE_SIZE;
-    char *backing_file = NULL;
-    char *backing_fmt = NULL;
+    BlockdevCreateOptions *create_options = NULL;
+    QDict *qdict = NULL;
+    QObject *qobj;
+    Visitor *v;
+    BlockDriverState *bs = NULL;
+    Error *local_err = NULL;
      int ret;
  
-    image_size = ROUND_UP(qemu_opt_get_size_del(opts, BLOCK_OPT_SIZE, 0),
-                          BDRV_SECTOR_SIZE);
-    backing_file = qemu_opt_get_del(opts, BLOCK_OPT_BACKING_FILE);
-    backing_fmt = qemu_opt_get_del(opts, BLOCK_OPT_BACKING_FMT);
-    cluster_size = qemu_opt_get_size_del(opts,
-                                         BLOCK_OPT_CLUSTER_SIZE,
-                                         QED_DEFAULT_CLUSTER_SIZE);
-    table_size = qemu_opt_get_size_del(opts, BLOCK_OPT_TABLE_SIZE,
-                                       QED_DEFAULT_TABLE_SIZE);
-
-    if (!qed_is_cluster_size_valid(cluster_size)) {
-        error_setg(errp, "QED cluster size must be within range [%u, %u] "
-                         "and power of 2",
-                   QED_MIN_CLUSTER_SIZE, QED_MAX_CLUSTER_SIZE);
+    static const QDictRenames opt_renames[] = {
+        { BLOCK_OPT_BACKING_FILE,       "backing-file" },
+        { BLOCK_OPT_BACKING_FMT,        "backing-fmt" },
+        { BLOCK_OPT_CLUSTER_SIZE,       "cluster-size" },
+        { BLOCK_OPT_TABLE_SIZE,         "table-size" },
+        { NULL, NULL },
+    };
+
+    /* Parse options and convert legacy syntax */
+    qdict = qemu_opts_to_qdict_filtered(opts, NULL, &qed_create_opts, true);
+
+    if (!qdict_rename_keys(qdict, opt_renames, errp)) {
          ret = -EINVAL;
-        goto finish;
+        goto fail;
      }
-    if (!qed_is_table_size_valid(table_size)) {
-        error_setg(errp, "QED table size must be within range [%u, %u] "
-                         "and power of 2",
-                   QED_MIN_TABLE_SIZE, QED_MAX_TABLE_SIZE);
+
+    /* Create and open the file (protocol layer) */
+    ret = bdrv_create_file(filename, opts, &local_err);
+    if (ret < 0) {
+        error_propagate(errp, local_err);
+        goto fail;
+    }
+
+    bs = bdrv_open(filename, NULL, NULL,
+                   BDRV_O_RDWR | BDRV_O_RESIZE | BDRV_O_PROTOCOL, errp);
+    if (bs == NULL) {
+        ret = -EIO;
+        goto fail;
+    }
+
+    /* Now get the QAPI type BlockdevCreateOptions */
+    qdict_put_str(qdict, "driver", "qed");
+    qdict_put_str(qdict, "file", bs->node_name);
+
+    qobj = qdict_crumple(qdict, errp);
+    qobject_unref(qdict);
+    qdict = qobject_to(QDict, qobj);
+    if (qdict == NULL) {
          ret = -EINVAL;
-        goto finish;
+        goto fail;
      }
-    if (!qed_is_image_size_valid(image_size, cluster_size, table_size)) {
-        error_setg(errp, "QED image size must be a non-zero multiple of "
-                         "cluster size and less than %" PRIu64 " bytes",
-                   qed_max_image_size(cluster_size, table_size));
+
+    v = qobject_input_visitor_new_keyval(QOBJECT(qdict));
+    visit_type_BlockdevCreateOptions(v, NULL, &create_options, &local_err);
+    visit_free(v);
+
+    if (local_err) {
+        error_propagate(errp, local_err);
          ret = -EINVAL;
-        goto finish;
+        goto fail;
      }
  
-    ret = qed_create(filename, cluster_size, image_size, table_size,
-                     backing_file, backing_fmt, opts, errp);
+    /* Silently round up size */
+    assert(create_options->driver == BLOCKDEV_DRIVER_QED);
+    create_options->u.qed.size =
+        ROUND_UP(create_options->u.qed.size, BDRV_SECTOR_SIZE);
+
+    /* Create the qed image (format layer) */
+    ret = bdrv_qed_co_create(create_options, errp);
  
-finish:
-    g_free(backing_file);
-    g_free(backing_fmt);
+fail:
+    qobject_unref(qdict);
+    bdrv_unref(bs);
+    qapi_free_BlockdevCreateOptions(create_options);
      return ret;
  }
  
-typedef struct {
-    BlockDriverState *bs;
-    Coroutine *co;
-    uint64_t pos;
-    int64_t status;
-    int *pnum;
-    BlockDriverState **file;
-} QEDIsAllocatedCB;
-
-static void qed_is_allocated_cb(void *opaque, int ret, uint64_t offset, size_t len)
+static int coroutine_fn bdrv_qed_co_block_status(BlockDriverState *bs,
+                                                 bool want_zero,
+                                                 int64_t pos, int64_t bytes,
+                                                 int64_t *pnum, int64_t *map,
+                                                 BlockDriverState **file)
  {
-    QEDIsAllocatedCB *cb = opaque;
-    BDRVQEDState *s = cb->bs->opaque;
-    *cb->pnum = len / BDRV_SECTOR_SIZE;
+    BDRVQEDState *s = bs->opaque;
+    size_t len = MIN(bytes, SIZE_MAX);
+    int status;
+    QEDRequest request = { .l2_table = NULL };
+    uint64_t offset;
+    int ret;
+
+    qemu_co_mutex_lock(&s->table_lock);
+    ret = qed_find_cluster(s, &request, pos, &len, &offset);
+
+    *pnum = len;
      switch (ret) {
      case QED_CLUSTER_FOUND:
-        offset |= qed_offset_into_cluster(s, cb->pos);
-        cb->status = BDRV_BLOCK_DATA | BDRV_BLOCK_OFFSET_VALID | offset;
-        *cb->file = cb->bs->file->bs;
+        *map = offset | qed_offset_into_cluster(s, pos);
+        status = BDRV_BLOCK_DATA | BDRV_BLOCK_OFFSET_VALID;
+        *file = bs->file->bs;
          break;
      case QED_CLUSTER_ZERO:
-        cb->status = BDRV_BLOCK_ZERO;
+        status = BDRV_BLOCK_ZERO;
          break;
      case QED_CLUSTER_L2:
      case QED_CLUSTER_L1:
-        cb->status = 0;
+        status = 0;
          break;
      default:
          assert(ret < 0);
-        cb->status = ret;
+        status = ret;
          break;
      }
  
-    if (cb->co) {
-        aio_co_wake(cb->co);
-    }
-}
-
-static int64_t coroutine_fn bdrv_qed_co_get_block_status(BlockDriverState *bs,
-                                                 int64_t sector_num,
-                                                 int nb_sectors, int *pnum,
-                                                 BlockDriverState **file)
-{
-    BDRVQEDState *s = bs->opaque;
-    size_t len = (size_t)nb_sectors * BDRV_SECTOR_SIZE;
-    QEDIsAllocatedCB cb = {
-        .bs = bs,
-        .pos = (uint64_t)sector_num * BDRV_SECTOR_SIZE,
-        .status = BDRV_BLOCK_OFFSET_MASK,
-        .pnum = pnum,
-        .file = file,
-    };
-    QEDRequest request = { .l2_table = NULL };
-    uint64_t offset;
-    int ret;
-
-    ret = qed_find_cluster(s, &request, cb.pos, &len, &offset);
-    qed_is_allocated_cb(&cb, ret, offset, len);
-
-    /* The callback was invoked immediately */
-    assert(cb.status != BDRV_BLOCK_OFFSET_MASK);
-
      qed_unref_l2_cache_entry(request.l2_table);
+    qemu_co_mutex_unlock(&s->table_lock);
  
-    return cb.status;
+    return status;
  }
  
  static BDRVQEDState *acb_to_s(QEDAIOCB *acb)
@@ -865,6 +963,8 @@ out:
   *
   * The cluster offset may be an allocated byte offset in the image file, the
   * zero cluster marker, or the unallocated cluster marker.
+ *
+ * Called with table_lock held.
   */
  static void coroutine_fn qed_update_l2_table(BDRVQEDState *s, QEDTable *table,
                                               int index, unsigned int n,
@@ -880,6 +980,7 @@ static void coroutine_fn qed_update_l2_table(BDRVQEDState *s, QEDTable *table,
      }
  }
  
+/* Called with table_lock held.  */
  static void coroutine_fn qed_aio_complete(QEDAIOCB *acb)
  {
      BDRVQEDState *s = acb_to_s(acb);
@@ -903,7 +1004,7 @@ static void coroutine_fn qed_aio_complete(QEDAIOCB *acb)
      if (acb == s->allocating_acb) {
          s->allocating_acb = NULL;
          if (!qemu_co_queue_empty(&s->allocating_write_reqs)) {
-            qemu_co_enter_next(&s->allocating_write_reqs);
+            qemu_co_queue_next(&s->allocating_write_reqs);
          } else if (s->header.features & QED_F_NEED_CHECK) {
              qed_start_need_check_timer(s);
          }
@@ -912,6 +1013,8 @@ static void coroutine_fn qed_aio_complete(QEDAIOCB *acb)
  
  /**
   * Update L1 table with new L2 table offset and write it out
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_l1_update(QEDAIOCB *acb)
  {
@@ -940,6 +1043,8 @@ static int coroutine_fn qed_aio_write_l1_update(QEDAIOCB *acb)
  
  /**
   * Update L2 table with new cluster offsets and write them out
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_l2_update(QEDAIOCB *acb, uint64_t offset)
  {
@@ -976,50 +1081,26 @@ static int coroutine_fn qed_aio_write_l2_update(QEDAIOCB *acb, uint64_t offset)
  
  /**
   * Write data to the image file
+ *
+ * Called with table_lock *not* held.
   */
  static int coroutine_fn qed_aio_write_main(QEDAIOCB *acb)
  {
      BDRVQEDState *s = acb_to_s(acb);
      uint64_t offset = acb->cur_cluster +
                        qed_offset_into_cluster(s, acb->cur_pos);
-    int ret;
  
      trace_qed_aio_write_main(s, acb, 0, offset, acb->cur_qiov.size);
  
      BLKDBG_EVENT(s->bs->file, BLKDBG_WRITE_AIO);
-    ret = bdrv_co_pwritev(s->bs->file, offset, acb->cur_qiov.size,
-                          &acb->cur_qiov, 0);
-    if (ret < 0) {
-        return ret;
-    }
-
-    if (acb->find_cluster_ret != QED_CLUSTER_FOUND) {
-        if (s->bs->backing) {
-            /*
-             * Flush new data clusters before updating the L2 table
-             *
-             * This flush is necessary when a backing file is in use.  A crash
-             * during an allocating write could result in empty clusters in the
-             * image.  If the write only touched a subregion of the cluster,
-             * then backing image sectors have been lost in the untouched
-             * region.  The solution is to flush after writing a new data
-             * cluster and before updating the L2 table.
-             */
-            ret = bdrv_co_flush(s->bs->file->bs);
-            if (ret < 0) {
-                return ret;
-            }
-        }
-        ret = qed_aio_write_l2_update(acb, acb->cur_cluster);
-        if (ret < 0) {
-            return ret;
-        }
-    }
-    return 0;
+    return bdrv_co_pwritev(s->bs->file, offset, acb->cur_qiov.size,
+                           &acb->cur_qiov, 0);
  }
  
  /**
   * Populate untouched regions of new data cluster
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_cow(QEDAIOCB *acb)
  {
@@ -1027,6 +1108,8 @@ static int coroutine_fn qed_aio_write_cow(QEDAIOCB *acb)
      uint64_t start, len, offset;
      int ret;
  
+    qemu_co_mutex_unlock(&s->table_lock);
+
      /* Populate front untouched region of new data cluster */
      start = qed_start_of_cluster(s, acb->cur_pos);
      len = qed_offset_into_cluster(s, acb->cur_pos);
@@ -1034,7 +1117,7 @@ static int coroutine_fn qed_aio_write_cow(QEDAIOCB *acb)
      trace_qed_aio_write_prefill(s, acb, start, len, acb->cur_cluster);
      ret = qed_copy_from_backing_file(s, start, len, acb->cur_cluster);
      if (ret < 0) {
-        return ret;
+        goto out;
      }
  
      /* Populate back untouched region of new data cluster */
@@ -1047,10 +1130,31 @@ static int coroutine_fn qed_aio_write_cow(QEDAIOCB *acb)
      trace_qed_aio_write_postfill(s, acb, start, len, offset);
      ret = qed_copy_from_backing_file(s, start, len, offset);
      if (ret < 0) {
-        return ret;
+        goto out;
+    }
+
+    ret = qed_aio_write_main(acb);
+    if (ret < 0) {
+        goto out;
      }
  
-    return qed_aio_write_main(acb);
+    if (s->bs->backing) {
+        /*
+         * Flush new data clusters before updating the L2 table
+         *
+         * This flush is necessary when a backing file is in use.  A crash
+         * during an allocating write could result in empty clusters in the
+         * image.  If the write only touched a subregion of the cluster,
+         * then backing image sectors have been lost in the untouched
+         * region.  The solution is to flush after writing a new data
+         * cluster and before updating the L2 table.
+         */
+        ret = bdrv_co_flush(s->bs->file->bs);
+    }
+
+out:
+    qemu_co_mutex_lock(&s->table_lock);
+    return ret;
  }
  
  /**
@@ -1073,6 +1177,8 @@ static bool qed_should_set_need_check(BDRVQEDState *s)
   * @len:        Length in bytes
   *
   * This path is taken when writing to previously unallocated clusters.
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_alloc(QEDAIOCB *acb, size_t len)
  {
@@ -1087,7 +1193,7 @@ static int coroutine_fn qed_aio_write_alloc(QEDAIOCB *acb, size_t len)
      /* Freeze this request if another allocating write is in progress */
      if (s->allocating_acb != acb || s->allocating_write_reqs_plugged) {
          if (s->allocating_acb != NULL) {
-            qemu_co_queue_wait(&s->allocating_write_reqs, NULL);
+            qemu_co_queue_wait(&s->allocating_write_reqs, &s->table_lock);
              assert(s->allocating_acb == NULL);
          }
          s->allocating_acb = acb;
@@ -1103,6 +1209,7 @@ static int coroutine_fn qed_aio_write_alloc(QEDAIOCB *acb, size_t len)
          if (acb->find_cluster_ret == QED_CLUSTER_ZERO) {
              return 0;
          }
+        acb->cur_cluster = 1;
      } else {
          acb->cur_cluster = qed_alloc_clusters(s, acb->cur_nclusters);
      }
@@ -1115,15 +1222,14 @@ static int coroutine_fn qed_aio_write_alloc(QEDAIOCB *acb, size_t len)
          }
      }
  
-    if (acb->flags & QED_AIOCB_ZERO) {
-        ret = qed_aio_write_l2_update(acb, 1);
-    } else {
+    if (!(acb->flags & QED_AIOCB_ZERO)) {
          ret = qed_aio_write_cow(acb);
+        if (ret < 0) {
+            return ret;
+        }
      }
-    if (ret < 0) {
-        return ret;
-    }
-    return 0;
+
+    return qed_aio_write_l2_update(acb, acb->cur_cluster);
  }
  
  /**
@@ -1134,10 +1240,17 @@ static int coroutine_fn qed_aio_write_alloc(QEDAIOCB *acb, size_t len)
   * @len:        Length in bytes
   *
   * This path is taken when writing to already allocated clusters.
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_inplace(QEDAIOCB *acb, uint64_t offset,
                                                size_t len)
  {
+    BDRVQEDState *s = acb_to_s(acb);
+    int r;
+
+    qemu_co_mutex_unlock(&s->table_lock);
+
      /* Allocate buffer for zero writes */
      if (acb->flags & QED_AIOCB_ZERO) {
          struct iovec *iov = acb->qiov->iov;
@@ -1145,7 +1258,8 @@ static int coroutine_fn qed_aio_write_inplace(QEDAIOCB *acb, uint64_t offset,
          if (!iov->iov_base) {
              iov->iov_base = qemu_try_blockalign(acb->bs, iov->iov_len);
              if (iov->iov_base == NULL) {
-                return -ENOMEM;
+                r = -ENOMEM;
+                goto out;
              }
              memset(iov->iov_base, 0, iov->iov_len);
          }
@@ -1155,8 +1269,11 @@ static int coroutine_fn qed_aio_write_inplace(QEDAIOCB *acb, uint64_t offset,
      acb->cur_cluster = offset;
      qemu_iovec_concat(&acb->cur_qiov, acb->qiov, acb->qiov_offset, len);
  
-    /* Do the actual write */
-    return qed_aio_write_main(acb);
+    /* Do the actual write.  */
+    r = qed_aio_write_main(acb);
+out:
+    qemu_co_mutex_lock(&s->table_lock);
+    return r;
  }
  
  /**
@@ -1166,6 +1283,8 @@ static int coroutine_fn qed_aio_write_inplace(QEDAIOCB *acb, uint64_t offset,
   * @ret:        QED_CLUSTER_FOUND, QED_CLUSTER_L2 or QED_CLUSTER_L1
   * @offset:     Cluster offset in bytes
   * @len:        Length in bytes
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_write_data(void *opaque, int ret,
                                             uint64_t offset, size_t len)
@@ -1197,6 +1316,8 @@ static int coroutine_fn qed_aio_write_data(void *opaque, int ret,
   * @ret:        QED_CLUSTER_FOUND, QED_CLUSTER_L2 or QED_CLUSTER_L1
   * @offset:     Cluster offset in bytes
   * @len:        Length in bytes
+ *
+ * Called with table_lock held.
   */
  static int coroutine_fn qed_aio_read_data(void *opaque, int ret,
                                            uint64_t offset, size_t len)
@@ -1204,6 +1325,9 @@ static int coroutine_fn qed_aio_read_data(void *opaque, int ret,
      QEDAIOCB *acb = opaque;
      BDRVQEDState *s = acb_to_s(acb);
      BlockDriverState *bs = acb->bs;
+    int r;
+
+    qemu_co_mutex_unlock(&s->table_lock);
  
      /* Adjust offset into cluster */
      offset += qed_offset_into_cluster(s, acb->cur_pos);
@@ -1212,22 +1336,23 @@ static int coroutine_fn qed_aio_read_data(void *opaque, int ret,
  
      qemu_iovec_concat(&acb->cur_qiov, acb->qiov, acb->qiov_offset, len);
  
-    /* Handle zero cluster and backing file reads */
+    /* Handle zero cluster and backing file reads, otherwise read
+     * data cluster directly.
+     */
      if (ret == QED_CLUSTER_ZERO) {
          qemu_iovec_memset(&acb->cur_qiov, 0, 0, acb->cur_qiov.size);
-        return 0;
+        r = 0;
      } else if (ret != QED_CLUSTER_FOUND) {
-        return qed_read_backing_file(s, acb->cur_pos, &acb->cur_qiov,
-                                     &acb->backing_qiov);
+        r = qed_read_backing_file(s, acb->cur_pos, &acb->cur_qiov,
+                                  &acb->backing_qiov);
+    } else {
+        BLKDBG_EVENT(bs->file, BLKDBG_READ_AIO);
+        r = bdrv_co_preadv(bs->file, offset, acb->cur_qiov.size,
+                           &acb->cur_qiov, 0);
      }
  
-    BLKDBG_EVENT(bs->file, BLKDBG_READ_AIO);
-    ret = bdrv_co_preadv(bs->file, offset, acb->cur_qiov.size,
-                         &acb->cur_qiov, 0);
-    if (ret < 0) {
-        return ret;
-    }
-    return 0;
+    qemu_co_mutex_lock(&s->table_lock);
+    return r;
  }
  
  /**
@@ -1240,6 +1365,7 @@ static int coroutine_fn qed_aio_next_io(QEDAIOCB *acb)
      size_t len;
      int ret;
  
+    qemu_co_mutex_lock(&s->table_lock);
      while (1) {
          trace_qed_aio_next_io(s, acb, 0, acb->cur_pos + acb->cur_qiov.size);
  
@@ -1279,6 +1405,7 @@ static int coroutine_fn qed_aio_next_io(QEDAIOCB *acb)
  
      trace_qed_aio_complete(s, acb, ret);
      qed_aio_complete(acb);
+    qemu_co_mutex_unlock(&s->table_lock);
      return ret;
  }
  
@@ -1351,7 +1478,7 @@ static int bdrv_qed_truncate(BlockDriverState *bs, int64_t offset,
  
      if (prealloc != PREALLOC_MODE_OFF) {
          error_setg(errp, "Unsupported preallocation mode '%s'",
-                   PreallocMode_lookup[prealloc]);
+                   PreallocMode_str(prealloc));
          return -ENOTSUP;
      }
  
@@ -1390,7 +1517,6 @@ static int bdrv_qed_get_info(BlockDriverState *bs, BlockDriverInfo *bdi)
      bdi->cluster_size = s->header.cluster_size;
      bdi->is_dirty = s->header.features & QED_F_NEED_CHECK;
      bdi->unallocated_blocks_are_zero = true;
-    bdi->can_write_zeroes_with_unmap = true;
      return 0;
  }
  
@@ -1466,7 +1592,8 @@ static int bdrv_qed_change_backing_file(BlockDriverState *bs,
      return ret;
  }
  
-static void bdrv_qed_invalidate_cache(BlockDriverState *bs, Error **errp)
+static void coroutine_fn bdrv_qed_co_invalidate_cache(BlockDriverState *bs,
+                                                      Error **errp)
  {
      BDRVQEDState *s = bs->opaque;
      Error *local_err = NULL;
@@ -1474,8 +1601,10 @@ static void bdrv_qed_invalidate_cache(BlockDriverState *bs, Error **errp)
  
      bdrv_qed_close(bs);
  
-    memset(s, 0, sizeof(BDRVQEDState));
+    bdrv_qed_init_state(bs);
+    qemu_co_mutex_lock(&s->table_lock);
      ret = bdrv_qed_do_open(bs, NULL, bs->open_flags, &local_err);
+    qemu_co_mutex_unlock(&s->table_lock);
      if (local_err) {
          error_propagate(errp, local_err);
          error_prepend(errp, "Could not reopen qed layer: ");
@@ -1486,12 +1615,17 @@ static void bdrv_qed_invalidate_cache(BlockDriverState *bs, Error **errp)
      }
  }
  
-static int bdrv_qed_check(BlockDriverState *bs, BdrvCheckResult *result,
-                          BdrvCheckMode fix)
+static int bdrv_qed_co_check(BlockDriverState *bs, BdrvCheckResult *result,
+                             BdrvCheckMode fix)
  {
      BDRVQEDState *s = bs->opaque;
+    int ret;
+
+    qemu_co_mutex_lock(&s->table_lock);
+    ret = qed_check(s, result, !!fix);
+    qemu_co_mutex_unlock(&s->table_lock);
  
-    return qed_check(s, result, !!fix);
+    return ret;
  }
  
  static QemuOptsList qed_create_opts = {
@@ -1539,9 +1673,10 @@ static BlockDriver bdrv_qed = {
      .bdrv_close               = bdrv_qed_close,
      .bdrv_reopen_prepare      = bdrv_qed_reopen_prepare,
      .bdrv_child_perm          = bdrv_format_default_perms,
-    .bdrv_create              = bdrv_qed_create,
+    .bdrv_co_create           = bdrv_qed_co_create,
+    .bdrv_co_create_opts      = bdrv_qed_co_create_opts,
      .bdrv_has_zero_init       = bdrv_has_zero_init_1,
-    .bdrv_co_get_block_status = bdrv_qed_co_get_block_status,
+    .bdrv_co_block_status     = bdrv_qed_co_block_status,
      .bdrv_co_readv            = bdrv_qed_co_readv,
      .bdrv_co_writev           = bdrv_qed_co_writev,
      .bdrv_co_pwrite_zeroes    = bdrv_qed_co_pwrite_zeroes,
@@ -1550,11 +1685,11 @@ static BlockDriver bdrv_qed = {
      .bdrv_get_info            = bdrv_qed_get_info,
      .bdrv_refresh_limits      = bdrv_qed_refresh_limits,
      .bdrv_change_backing_file = bdrv_qed_change_backing_file,
-    .bdrv_invalidate_cache    = bdrv_qed_invalidate_cache,
-    .bdrv_check               = bdrv_qed_check,
+    .bdrv_co_invalidate_cache = bdrv_qed_co_invalidate_cache,
+    .bdrv_co_check            = bdrv_qed_co_check,
      .bdrv_detach_aio_context  = bdrv_qed_detach_aio_context,
      .bdrv_attach_aio_context  = bdrv_qed_attach_aio_context,
-    .bdrv_drain               = bdrv_qed_drain,
+    .bdrv_co_drain_begin      = bdrv_qed_co_drain_begin,
  };
  
  static void bdrv_qed_init(void)