Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -523,7 +523,7 @@ public String obfuscatePassword(String result, boolean hidePassword) {
return StringUtils.obfuscatePasswordInJsonLikeString(result);
}

private void scheduleExecution(final AsyncJobVO job) {
protected void scheduleExecution(final AsyncJobVO job) {
scheduleExecution(job, false);
}

Expand Down Expand Up @@ -701,58 +701,58 @@ private int getAndResetPendingSignals(AsyncJob job) {
return signals;
}

private void executeQueueItem(SyncQueueItemVO item, boolean fromPreviousSession) {
protected void executeQueueItem(SyncQueueItemVO item, boolean fromPreviousSession) {
AsyncJobVO job = _jobDao.findById(item.getContentId());
if (job != null) {
if (job == null) {
if (logger.isDebugEnabled()) {
logger.debug("Schedule queued job-" + job.getId());
}

job.setSyncSource(item);

//
// TODO: a temporary solution to work-around DB deadlock situation
//
// to live with DB deadlocks, we will give a chance for job to be rescheduled
// in case of exceptions (most-likely DB deadlock exceptions)
try {
job.setExecutingMsid(getMsid());
_jobDao.update(job.getId(), job);
} catch (Exception e) {
logger.warn("Unexpected exception while dispatching job-" + item.getContentId(), e);

try {
_queueMgr.returnItem(item.getId());
} catch (Throwable thr) {
logger.error("Unexpected exception while returning job-" + item.getContentId() + " to queue", thr);
}
logger.debug("Unable to find related job for queue item: " + item.toString());
}
_queueMgr.purgeItem(item.getId());
return;
}

try {
scheduleExecution(job);
} catch (RejectedExecutionException e) {
logger.warn("Execution for job-" + job.getId() + " is rejected, return it to the queue for next turn");
if (logger.isDebugEnabled()) {
logger.debug("Schedule queued job-" + job.getId());
}
job.setSyncSource(item);

try {
_queueMgr.returnItem(item.getId());
} catch (Exception e2) {
logger.error("Unexpected exception while returning job-" + item.getContentId() + " to queue", e2);
}
//
// TODO: a temporary solution to work-around DB deadlock situation
//
// to live with DB deadlocks, we will give a chance for job to be rescheduled
// in case of exceptions (most-likely DB deadlock exceptions)
Comment on lines +719 to +723

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

how temporary? what is the work forwards?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Thanks @DaanHoogland, good question.

For context, that TODO predates this PR (it is on main today) and the restructure only re-indented it, so I kept its intent as is rather than change it here.

On why it is called temporary: stamping the executing management-server id via _jobDao.update can hit a DB deadlock, and the workaround gives the job another chance by returning the queue item to the sync queue to be retried on a later turn instead of failing it. This PR keeps that behaviour and only fixes the case where a failed dispatch also fell through and executed the job immediately, so it ran twice.

For the work forwards, I think the cleaner fix is to address the deadlock at its source, around the locking and transaction for the sync_queue and async_job update on dispatch, so the retry is no longer needed. That felt larger than this bug fix, so I kept it out of scope here, but I would be glad to look into it as a follow up if you agree that is the right direction.

try {
job.setExecutingMsid(getMsid());
_jobDao.update(job.getId(), job);
} catch (Exception e) {
logger.warn("Unexpected exception while dispatching job-" + item.getContentId(), e);
returnItemToQueue(item);
return;
}

try {
job.setExecutingMsid(null);
_jobDao.update(job.getId(), job);
} catch (Exception e3) {
logger.warn("Unexpected exception while update job-" + item.getContentId() + " msid for bookkeeping");
}
}
try {
scheduleExecution(job);
} catch (RejectedExecutionException e) {
logger.warn("Execution for job-" + job.getId() + " is rejected, return it to the queue for next turn");
returnItemToQueue(item);
clearExecutingMsid(job, item);
}
}

} else {
if (logger.isDebugEnabled()) {
logger.debug("Unable to find related job for queue item: " + item.toString());
}
private void returnItemToQueue(SyncQueueItemVO item) {
try {
_queueMgr.returnItem(item.getId());
} catch (Throwable thr) {
logger.error("Unexpected exception while returning job-" + item.getContentId() + " to queue", thr);
}
}

_queueMgr.purgeItem(item.getId());
private void clearExecutingMsid(AsyncJobVO job, SyncQueueItemVO item) {
try {
job.setExecutingMsid(null);
_jobDao.update(job.getId(), job);
} catch (Exception e) {
logger.warn("Unexpected exception while update job-" + item.getContentId() + " msid for bookkeeping");
}
}

Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,70 @@
// Licensed to the Apache Software Foundation (ASF) under one
// or more contributor license agreements. See the NOTICE file
// distributed with this work for additional information
// regarding copyright ownership. The ASF licenses this file
// to you under the Apache License, Version 2.0 (the
// "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
package org.apache.cloudstack.framework.jobs.impl;

import org.apache.cloudstack.framework.jobs.dao.AsyncJobDao;
import org.junit.Test;
import org.junit.runner.RunWith;
import org.mockito.InjectMocks;
import org.mockito.Mock;
import org.mockito.Mockito;
import org.mockito.Spy;
import org.mockito.junit.MockitoJUnitRunner;

import com.cloud.utils.exception.CloudRuntimeException;

@RunWith(MockitoJUnitRunner.Silent.class)
public class AsyncJobManagerImplExecuteQueueItemTest {

@Mock
AsyncJobDao _jobDao;
@Mock
SyncQueueManager _queueMgr;

@Spy
@InjectMocks
AsyncJobManagerImpl asyncJobManager = new AsyncJobManagerImpl();

@Test
public void executeQueueItemDoesNotScheduleWhenTheJobUpdateFailsAndItemIsReturned() {
long contentId = 10L;
long itemId = 20L;
long jobId = 1L;

SyncQueueItemVO item = Mockito.mock(SyncQueueItemVO.class);
Mockito.when(item.getContentId()).thenReturn(contentId);
Mockito.when(item.getId()).thenReturn(itemId);

AsyncJobVO job = Mockito.mock(AsyncJobVO.class);
Mockito.when(job.getId()).thenReturn(jobId);
Mockito.when(_jobDao.findById(contentId)).thenReturn(job);

// Simulate the DB deadlock the catch block was written to survive.
Mockito.doThrow(new CloudRuntimeException("simulated DB deadlock"))
.when(_jobDao).update(Mockito.anyLong(), Mockito.any(AsyncJobVO.class));

// Stub the executor path so we can assert whether it is reached (and avoid the real submit).
Mockito.doNothing().when(asyncJobManager).scheduleExecution(Mockito.any(AsyncJobVO.class));

asyncJobManager.executeQueueItem(item, false);

// The queue item was returned for a later retry; the job must NOT also be scheduled now, or it
// would run twice (once here and once when the heartbeat re-dequeues the returned item).
Mockito.verify(_queueMgr).returnItem(itemId);
Mockito.verify(asyncJobManager, Mockito.never()).scheduleExecution(Mockito.any(AsyncJobVO.class));
}
}
Loading