1 From 0000000000000000000000000000000000000000 Mon Sep 17 00:00:00 2001
2 From: Fabian Ebner <f.ebner@proxmox.com>
3 Date: Wed, 25 May 2022 13:59:38 +0200
4 Subject: [PATCH] PVE-Backup: ensure jobs in di_list are referenced
6 Ensures that qmp_backup_cancel doesn't pick a job that's already been
7 freed. With unlucky timings it seems possible that:
8 1. job_exit -> job_completed -> job_finalize_single starts
9 2. pvebackup_co_complete_stream gets spawned in completion callback
10 3. job finalize_single finishes -> job's refcount hits zero -> job is
12 4. qmp_backup_cancel comes in and locks backup_state.backup_mutex
13 before pvebackup_co_complete_stream can remove the job from the
15 5. qmp_backup_cancel will pick a job that's already been freed
17 Signed-off-by: Fiona Ebner <f.ebner@proxmox.com>
18 Signed-off-by: Wolfgang Bumiller <w.bumiller@proxmox.com>
19 [FE: adapt for new job lock mechanism replacing AioContext locks]
20 Signed-off-by: Fiona Ebner <f.ebner@proxmox.com>
22 pve-backup.c | 22 +++++++++++++++++++---
23 1 file changed, 19 insertions(+), 3 deletions(-)
25 diff --git a/pve-backup.c b/pve-backup.c
26 index fde3554133..0cf30e1ced 100644
29 @@ -316,6 +316,13 @@ static void coroutine_fn pvebackup_co_complete_stream(void *opaque)
34 + WITH_JOB_LOCK_GUARD() {
35 + job_unref_locked(&di->job->job);
40 // remove self from job list
41 backup_state.di_list = g_list_remove(backup_state.di_list, di);
43 @@ -491,6 +498,11 @@ static void create_backup_jobs_bh(void *opaque) {
44 aio_context_release(aio_context);
48 + WITH_JOB_LOCK_GUARD() {
49 + job_ref_locked(&job->job);
53 if (!job || local_err) {
54 error_setg(errp, "backup_job_create failed: %s",
55 @@ -518,11 +530,15 @@ static void create_backup_jobs_bh(void *opaque) {
59 - if (!canceled && di->job) {
61 WITH_JOB_LOCK_GUARD() {
62 - job_cancel_sync_locked(&di->job->job, true);
64 + job_cancel_sync_locked(&di->job->job, true);
67 + job_unref_locked(&di->job->job);