Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -32,7 +32,6 @@
import org.apache.cloudstack.storage.datastore.db.ImageStoreDao;
import org.apache.commons.lang3.StringUtils;

import com.cloud.alert.AlertManager;
import com.cloud.api.ApiDBUtils;
import com.cloud.api.query.dao.DomainJoinDao;
import com.cloud.api.query.dao.StoragePoolJoinDao;
Expand Down Expand Up @@ -126,8 +125,6 @@ public String toString() {
@Inject
private DomainJoinDao domainDao;
@Inject
private AlertManager alertManager;
@Inject
DedicatedResourceDao _dedicatedDao;
@Inject
private AccountDao _accountDao;
Expand Down Expand Up @@ -494,10 +491,17 @@ private void addVMsBySizeMetrics(final List<Item> metricsList, final long dcId,
public void updateMetrics() {
final List<Item> latestMetricsItems = new ArrayList<Item>();
try {
// NOTE: capacity data is refreshed independently by AlertManagerImpl's own

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Nit: this inline comment is quite long (7 lines), reads more like a commit message / issue writeup than a code comment. Could trim to a line or two referencing the issue.

// periodic CapacityChecker timer (see AlertManagerImpl#start()). Do NOT force a
// synchronous recalculateCapacity() here: it spins up a fresh thread pool per host
// and per storage pool across ALL zones on every single scrape, so with Z zones a
// single Prometheus scrape triggered Z redundant full recalculations. That extra,
// uncoordinated load compounds over time (thread churn + overlapping runs with the
// timer) and was the cause of https://github.com/apache/cloudstack/issues/13586
// (scrape_duration_seconds climbing until a management-server restart).
for (final DataCenterVO dc : dcDao.listAll()) {
final String zoneName = dc.getName();
final String zoneUuid = dc.getUuid();
alertManager.recalculateCapacity();
addHostMetrics(latestMetricsItems, dc.getId(), zoneName, zoneUuid);
addVMMetrics(latestMetricsItems, dc.getId(), zoneName, zoneUuid);
addVolumeMetrics(latestMetricsItems, dc.getId(), zoneName, zoneUuid);
Expand Down
33 changes: 25 additions & 8 deletions server/src/main/java/com/cloud/alert/AlertManagerImpl.java
Original file line number Diff line number Diff line change
Expand Up @@ -161,6 +161,8 @@

private final ExecutorService _executor;

private ExecutorService _capacityExecutorService;

Check warning on line 164 in server/src/main/java/com/cloud/alert/AlertManagerImpl.java

View check run for this annotation

SonarQubeCloud / SonarCloud Code Analysis

Rename this field "_capacityExecutorService" to match the regular expression '^[a-z][a-zA-Z0-9]*$'.

See more on https://sonarcloud.io/project/issues?id=apache_cloudstack&issues=AZ9_U4wrQxgFAdfhAtKw&open=AZ9_U4wrQxgFAdfhAtKw&pullRequest=13650

protected SMTPMailSender mailSender;
protected String[] recipients = null;
protected String senderAddress = null;
Expand Down Expand Up @@ -249,6 +251,9 @@
@Override
public boolean stop() {
_timer.cancel();
if (_capacityExecutorService != null) {
_capacityExecutorService.shutdown();
}
return true;
}

Expand Down Expand Up @@ -281,6 +286,24 @@
}
}

/**
* Shared, long-lived pool for capacity recalculation, reused across every
* recalculateHostCapacities()/recalculateStorageCapacities() call instead of creating and
* tearing down a new thread pool per invocation. Repeatedly creating/shutting down pools was
* unnecessary overhead under frequent callers (e.g. the Prometheus exporter used to trigger a
* full recalculation on every scrape, see https://github.com/apache/cloudstack/issues/13586).
* Lazily created so this remains safe for callers that invoke the recalculate methods directly
* without going through configure()/start() (e.g. unit tests).
*/
private synchronized ExecutorService getCapacityExecutorService() {
if (_capacityExecutorService == null || _capacityExecutorService.isShutdown()) {
_capacityExecutorService = Executors.newFixedThreadPool(
Math.max(1, CapacityManager.CapacityCalculateWorkers.value()),
new NamedThreadFactory("Capacity-Calculator"));
}
return _capacityExecutorService;
}

/**
* Recalculates the capacities of hosts, including CPU and RAM.
*/
Expand All @@ -290,10 +313,8 @@
return;
}
ConcurrentHashMap<Long, Future<Void>> futures = new ConcurrentHashMap<>();
ExecutorService executorService = Executors.newFixedThreadPool(Math.max(1,
Math.min(CapacityManager.CapacityCalculateWorkers.value(), hostIds.size())));
for (Long hostId : hostIds) {
futures.put(hostId, executorService.submit(() -> {
futures.put(hostId, getCapacityExecutorService().submit(() -> {
final HostVO host = hostDao.findById(hostId);
_capacityMgr.updateCapacityForHost(host);
return null;
Expand All @@ -307,7 +328,6 @@
entry.getKey(), e.getMessage()), e);
}
}
executorService.shutdown();
}

protected void recalculateStorageCapacities() {
Expand All @@ -316,10 +336,8 @@
return;
}
ConcurrentHashMap<Long, Future<Void>> futures = new ConcurrentHashMap<>();
ExecutorService executorService = Executors.newFixedThreadPool(Math.max(1,
Math.min(CapacityManager.CapacityCalculateWorkers.value(), storagePoolIds.size())));
for (Long poolId: storagePoolIds) {
futures.put(poolId, executorService.submit(() -> {
futures.put(poolId, getCapacityExecutorService().submit(() -> {
Transaction.execute(new TransactionCallbackNoReturn() {
@Override
public void doInTransactionWithoutResult(TransactionStatus status) {
Expand All @@ -343,7 +361,6 @@
entry.getKey(), e.getMessage()), e);
}
}
executorService.shutdown();
}

@Override
Expand Down
Loading