Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions documentation/sphinx/source/mr-status-json-schemas.rst.inc
Original file line number Diff line number Diff line change
Expand Up @@ -650,6 +650,7 @@
},
"active_tss_count":0,
"degraded_processes":0,
"degraded_multi_region":true,
"database_available":true,
"database_lock_state":{
"locked":true,
Expand Down
47 changes: 34 additions & 13 deletions fdbcli/StatusCommand.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -710,9 +710,14 @@ void printStatus(StatusObjectReader statusObj,
outputString += format(" (%d without data loss)", dataLoss);
}

// Both verdicts are evaluated before the branching below so that region availability can
// be reported independently of how the log state is classified.
const bool possiblyLosingData = logEpochsMayBeLosingData(statusObjCluster);
bool degradedMultiRegion = false;
statusObjCluster.get("degraded_multi_region", degradedMultiRegion);

if (dataLoss == -1) {
ASSERT_WE_THINK(availLoss == -1);
const bool possiblyLosingData = logEpochsMayBeLosingData(statusObjCluster);
if (possiblyLosingData) {
outputString += format(
"\n\n Warning: the database may have data loss and availability loss. Please "
Expand All @@ -725,18 +730,14 @@ void printStatus(StatusObjectReader statusObj,
}
if (statusObjCluster.has("logs")) {
for (StatusObjectReader logEpoch : statusObjCluster.last().get_array()) {
bool logEpochPossiblyLosingData;
if (logEpoch.get("possibly_losing_data", logEpochPossiblyLosingData) &&
!logEpochPossiblyLosingData) {
continue;
}
// Current epoch doesn't have an end version.
int64_t epoch, beginVersion, endVersion = invalidVersion;
bool current;
logEpoch.get("epoch", epoch);
logEpoch.get("begin_version", beginVersion);
logEpoch.get("end_version", endVersion);
logEpoch.get("current", current);
// Unknown means "assume at risk": the field is absent on servers that predate it.
bool logEpochPossiblyLosingData = true;
const bool dataAtRisk =
!logEpoch.get("possibly_losing_data", logEpochPossiblyLosingData) ||
logEpochPossiblyLosingData;
// Unavailable log interfaces are the ones that must come back for the cluster to
// finish recovering, so an epoch is reported when either its data is at risk or any
// of its log interfaces is unavailable.
std::string missing_log_interfaces;
if (logEpoch.has("log_interfaces")) {
for (StatusObjectReader logInterface : logEpoch.last().get_array()) {
Expand All @@ -749,6 +750,16 @@ void printStatus(StatusObjectReader statusObj,
}
}
}
if (!dataAtRisk && missing_log_interfaces.empty()) {
continue;
}
// Current epoch doesn't have an end version.
int64_t epoch, beginVersion, endVersion = invalidVersion;
bool current;
logEpoch.get("epoch", epoch);
logEpoch.get("begin_version", beginVersion);
logEpoch.get("end_version", endVersion);
logEpoch.get("current", current);
outputString += format(
" %s log epoch: %lld begin: %lld end: %s, missing "
"log interfaces(id,address): %s\n",
Expand All @@ -760,6 +771,16 @@ void printStatus(StatusObjectReader statusObj,
}
}
}
// Region availability is orthogonal to the data-loss verdict above: report it whenever the
// cluster says a region is unavailable, whichever way the log state was classified. The
// statement about committed data is only made when the log state does not indicate data loss.
if (degradedMultiRegion) {
outputString += possiblyLosingData
? "\n One region is unavailable; data in the surviving region may be "
"incomplete.\n"
: "\n One region is unavailable; committed data is expected to remain "
"safe in the surviving region.\n";
}
}
}

Expand Down
88 changes: 88 additions & 0 deletions fdbcli/tests/fdbcli_tests.py
Original file line number Diff line number Diff line change
Expand Up @@ -458,6 +458,93 @@ def status_json_file_region_failover_message():
assert "may have data loss" not in stdout


def status_json_degraded_multi_region_message():
# A multi-region cluster whose primary region is down while its satellite survives recovers without data
# loss: the log generation that lost its primary set still has an intact synchronous satellite copy, so
# the server reports possibly_losing_data=false for it. Status must warn about availability loss, explain
# that the surviving region is expected to hold all committed data, and must not claim data loss.
status_json = {
"client": {
"cluster_file": {"path": "fdb.cluster", "up_to_date": True},
"coordinators": {"coordinators": [], "quorum_reachable": True},
"database_status": {"available": True, "healthy": False},
"messages": [],
"timestamp": 1417807090,
},
"cluster": {
"configuration": {
"redundancy_mode": "double",
"storage_engine": "ssd-2",
"coordinators_count": 3,
"excluded_servers": [],
},
"data": {"state": {"name": "healthy", "healthy": True}},
"degraded_multi_region": True,
"fault_tolerance": {
"max_zone_failures_without_losing_availability": -1,
"max_zone_failures_without_losing_data": -1,
},
"logs": [
{
"epoch": 2,
"current": True,
"begin_version": 100,
"possibly_losing_data": False,
"log_interfaces": [],
},
{
"epoch": 1,
"current": False,
"begin_version": 1,
"end_version": 100,
"possibly_losing_data": False,
"log_fault_tolerance": -1,
"satellite_log_replication_factor": 1,
"satellite_log_fault_tolerance": 0,
"log_interfaces": [
{
"id": "aaaaaaaaaaaaaaaa",
"healthy": False,
"address": "1.1.1.1:4500",
}
],
},
],
"machines": {},
"processes": {},
},
}

def render(status):
with tempfile.NamedTemporaryFile(mode="w", suffix=".json") as status_file:
json.dump(status, status_file)
status_file.flush()
result = subprocess.run(
[command_template[0], "--status-from-json", status_file.name],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
env=fdbcli_env,
)
assert result.returncode == 0, result.stderr.decode("utf-8")
return result.stdout.decode("utf-8")

stdout = render(status_json)
assert "Warning: the database may have availability loss." in stdout
assert "may have data loss" not in stdout
assert (
"One region is unavailable; committed data is expected to remain safe in the surviving region."
in stdout
)
# The unavailable interfaces of the generation whose data is safe are still listed, so that the operator
# knows what has to come back for the cluster to finish recovering.
assert "missing log interfaces(id,address): aaaaaaaaaaaaaaaa,1.1.1.1:4500" in stdout

# The region note is only printed when the cluster explicitly reports an unavailable region.
del status_json["cluster"]["degraded_multi_region"]
stdout = render(status_json)
assert "One region is unavailable" not in stdout


@enable_logging()
def consistencycheck(logger):
consistency_check_on_output = "ConsistencyCheck is on"
Expand Down Expand Up @@ -1013,6 +1100,7 @@ def tls_address_suffix():
integer_options()
tls_address_suffix()
status_json_file_region_failover_message()
status_json_degraded_multi_region_message()
idempotency_ids()
client_threads_per_version_env_ignored()
else:
Expand Down
2 changes: 2 additions & 0 deletions fdbclient/Schemas.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -632,6 +632,7 @@ const KeyRef JSONSchemas::statusSchema = R"statusSchema(
},
"active_tss_count":0,
"degraded_processes":0,
"degraded_multi_region":true,
"database_available":true,
"database_lock_state": {
"locked": true,
Expand Down Expand Up @@ -1550,6 +1551,7 @@ file is writable and has not been overwritten externally."
},
"maintenance_zone":"0ccb4e0fdbdb5583010f6b77d9d10ece",
"maintenance_seconds_remaining":1.0,
"degraded_multi_region":true,
"data":{
"least_operating_space_bytes_log_server":0,
"average_partition_size_bytes":0,
Expand Down
28 changes: 17 additions & 11 deletions fdbserver/SimulatedCluster.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -481,6 +481,8 @@ class TestConfig : public BasicTestConfig {
Optional<bool> generateFearless, buggify, faultInjection;
Optional<std::string> config;
Optional<std::string> remoteConfig;
// Satellite redundancy mode override applied to all regions (e.g. "one_satellite_single")
Optional<std::string> satelliteRedundancyMode;
bool randomlyRenameZoneId = false;
bool simHTTPServerEnabled = true;

Expand Down Expand Up @@ -548,6 +550,7 @@ class TestConfig : public BasicTestConfig {
.add("storageEngineType", &storageEngineType)
.add("config", &config)
.add("remoteConfig", &remoteConfig)
.add("satelliteRedundancyMode", &satelliteRedundancyMode)
.add("buggify", &buggify)
.add("faultInjection", &faultInjection)
.add("StderrSeverity", &stderrSeverity)
Expand Down Expand Up @@ -1896,7 +1899,10 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) {

bool needsRemote = generateFearless;
if (generateFearless) {
if (datacenters > 4) {
std::string satelliteRedundancyModeStr;
if (testConfig.satelliteRedundancyMode.present()) {
satelliteRedundancyModeStr = testConfig.satelliteRedundancyMode.get();
} else if (datacenters > 4) {
// FIXME: we cannot use one satellite replication with more than one satellite per region because
// canKillProcesses does not respect usable_dcs
int satellite_replication_type = deterministicRandom()->randomInt(0, 3);
Expand All @@ -1912,14 +1918,12 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) {
}
case 1: {
CODE_PROBE(true, "Simulated cluster using two satellite fast redundancy mode");
primaryObj["satellite_redundancy_mode"] = "two_satellite_fast";
remoteObj["satellite_redundancy_mode"] = "two_satellite_fast";
satelliteRedundancyModeStr = "two_satellite_fast";
break;
}
case 2: {
CODE_PROBE(true, "Simulated cluster using two satellite safe redundancy mode");
primaryObj["satellite_redundancy_mode"] = "two_satellite_safe";
remoteObj["satellite_redundancy_mode"] = "two_satellite_safe";
satelliteRedundancyModeStr = "two_satellite_safe";
break;
}
default:
Expand All @@ -1939,27 +1943,29 @@ void SimulationConfig::setRegions(const TestConfig& testConfig) {
}
case 2: {
CODE_PROBE(true, "Simulated cluster using single satellite redundancy mode");
primaryObj["satellite_redundancy_mode"] = "one_satellite_single";
remoteObj["satellite_redundancy_mode"] = "one_satellite_single";
satelliteRedundancyModeStr = "one_satellite_single";
break;
}
case 3: {
CODE_PROBE(true, "Simulated cluster using double satellite redundancy mode");
primaryObj["satellite_redundancy_mode"] = "one_satellite_double";
remoteObj["satellite_redundancy_mode"] = "one_satellite_double";
satelliteRedundancyModeStr = "one_satellite_double";
break;
}
case 4: {
CODE_PROBE(true, "Simulated cluster using triple satellite redundancy mode");
primaryObj["satellite_redundancy_mode"] = "one_satellite_triple";
remoteObj["satellite_redundancy_mode"] = "one_satellite_triple";
satelliteRedundancyModeStr = "one_satellite_triple";
break;
}
default:
ASSERT(false); // Programmer forgot to adjust cases.
}
}

if (!satelliteRedundancyModeStr.empty()) {
primaryObj["satellite_redundancy_mode"] = satelliteRedundancyModeStr;
remoteObj["satellite_redundancy_mode"] = satelliteRedundancyModeStr;
}

// Calculate the maximum satellite_logs we can support based on available machines
bool useNormalDCsAsSatellites =
datacenters > 4 && testConfig.minimumRegions < 2 && deterministicRandom()->random01() < 0.3;
Expand Down
37 changes: 37 additions & 0 deletions fdbserver/clustercontroller/ClusterRecovery.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -514,6 +514,10 @@ Future<Void> trackTlogRecovery(Reference<ClusterRecoveryData> self,
DBRecoveryCount recoverCount = self->cstate.myDBState.recoveryCount + 1;
DatabaseConfiguration configuration =
self->configuration; // self-configuration can be changed by configurationMonitor so we need a copy
// Start of the current remote-region stall, if any. Kept across loop iterations so that
// re-emissions of the event on unrelated core-state changes do not reset the reported stall
// start; status derives the elapsed duration from this timestamp.
Optional<double> remoteLogsMissingSince;
while (true) {
DBCoreState newState;
self->logSystem->toCoreState(newState);
Expand Down Expand Up @@ -581,6 +585,37 @@ Future<Void> trackTlogRecovery(Reference<ClusterRecoveryData> self,
.trackLatest(self->clusterRecoveryStateEventHolder->trackingKey);
}

// "Degraded multi-region": with usableRegions > 1 the remote region's log set has not been
// recruited (allLogs == false). oldTLogData is deliberately not a discriminator: when the
// remote region is down, old generations cannot be purged (finalUpdate requires allLogs),
// so oldTLogData stays non-empty precisely in this stalled state.
//
// Gate on ACCEPTING_COMMITS: a missing remote log set is a transient recruiting artifact
// during normal recovery, so the stall must not start accumulating before the cluster
// accepts commits, or a slow-but-healthy recovery trips degraded_multi_region.
bool remoteRegionLogsMissing =
configuration.usableRegions > 1 && !allLogs && self->recoveryState >= RecoveryState::ACCEPTING_COMMITS;
if (remoteRegionLogsMissing && !remoteLogsMissingSince.present()) {
remoteLogsMissingSince = now();
} else if (!remoteRegionLogsMissing) {
remoteLogsMissingSince = Optional<double>();
}
// Carry the absolute time the stall began rather than a pre-computed duration: status derives
// how long the stall has lasted when it reads this event, so the event does not have to be
// re-emitted periodically just to keep a duration field fresh. Format as a string because
// numeric trace fields are emitted with "%g" (six significant digits), which is far too coarse
// for an absolute timestamp.
double remoteRegionStallStartSeconds = remoteLogsMissingSince.present() ? remoteLogsMissingSince.get() : 0.0;
TraceEvent(
getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME).c_str(),
self->dbgid)
.detail("RemoteRegionLogsMissing", remoteRegionLogsMissing)
.detail("AllLogs", allLogs)
.detail("OldTLogDataSize", newState.oldTLogData.size())
.detail("UsableRegions", configuration.usableRegions)
.detail("RemoteRegionStallStartSeconds", format("%.6f", remoteRegionStallStartSeconds))
.trackLatest(self->clusterRecoveryRemoteRegionStallEventHolder->trackingKey);

self->registrationTrigger.trigger();

if (finalUpdate) {
Expand Down Expand Up @@ -2100,6 +2135,8 @@ const std::string& getRecoveryEventName(ClusterRecoveryEventType type) {
SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryAvailable" });
recoveryEventNameMap.insert({ ClusterRecoveryEventType::CLUSTER_RECOVERY_METRICS_EVENT_NAME,
SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryMetrics" });
recoveryEventNameMap.insert({ ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME,
SERVER_KNOBS->CLUSTER_RECOVERY_EVENT_NAME_PREFIX + "RecoveryRemoteRegionStall" });
}

auto iter = recoveryEventNameMap.find(type);
Expand Down
4 changes: 4 additions & 0 deletions fdbserver/clustercontroller/ClusterRecovery.h
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,7 @@ enum ClusterRecoveryEventType {
CLUSTER_RECOVERY_COMMIT_EVENT_NAME,
CLUSTER_RECOVERY_AVAILABLE_EVENT_NAME,
CLUSTER_RECOVERY_METRICS_EVENT_NAME,
CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME,
CLUSTER_RECOVERY_LAST // Always the last entry
};

Expand Down Expand Up @@ -254,6 +255,7 @@ struct ClusterRecoveryData : NonCopyable, ReferenceCounted<ClusterRecoveryData>
Reference<EventCacheHolder> clusterRecoveryGenerationsEventHolder;
Reference<EventCacheHolder> clusterRecoveryDurationEventHolder;
Reference<EventCacheHolder> clusterRecoveryAvailableEventHolder;
Reference<EventCacheHolder> clusterRecoveryRemoteRegionStallEventHolder;

ClusterRecoveryData(ClusterControllerData* controllerData,
Reference<AsyncVar<ServerDBInfo> const> const& dbInfo,
Expand Down Expand Up @@ -290,6 +292,8 @@ struct ClusterRecoveryData : NonCopyable, ReferenceCounted<ClusterRecoveryData>
getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_DURATION_EVENT_NAME));
clusterRecoveryAvailableEventHolder = makeReference<EventCacheHolder>(
getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_AVAILABLE_EVENT_NAME));
clusterRecoveryRemoteRegionStallEventHolder = makeReference<EventCacheHolder>(
getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_REMOTE_REGION_STALL_EVENT_NAME));
logger =
cc.traceCounters(getRecoveryEventName(ClusterRecoveryEventType::CLUSTER_RECOVERY_METRICS_EVENT_NAME),
dbgid,
Expand Down
Loading
Loading