Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
59 changes: 59 additions & 0 deletions libs/server/Databases/CheckpointStatus.cs
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
// Copyright (c) Microsoft Corporation.
// Licensed under the MIT license.

namespace Garnet.server
{
/// <summary>
/// Outcome of a checkpoint request issued against one or more logical databases.
/// </summary>
public enum CheckpointStatus
{
/// <summary>
/// The checkpoint completed and its data is durable. For a background checkpoint this means the checkpoint
/// was successfully started; its outcome is reported through the database's last save time and status.
/// </summary>
Success,

/// <summary>
/// A checkpoint is already in progress for at least one of the requested databases, so no new checkpoint
/// was started. Nothing was written and no existing checkpoint was invalidated.
/// </summary>
AlreadyInProgress,

/// <summary>
/// The checkpoint was started but did not complete, so nothing durable was written.
/// </summary>
Failed,
}

/// <summary>
/// Outcome of a single database's checkpoint attempt.
/// </summary>
internal readonly struct CheckpointResult
{
/// <summary>
/// True if the checkpoint completed and its data is durable. Only then may the database's last save time
/// be advanced; advancing it for a failed checkpoint reports data as durable that was never written.
/// </summary>
public bool IsSuccessful { get; init; }

/// <summary>
/// Store tail address covered by a full checkpoint, or null for an incremental checkpoint. Null is also
/// returned for a failed checkpoint, which is why it cannot by itself signal failure.
/// </summary>
public long? StoreTailAddress { get; init; }

/// <summary>
/// A result denoting a checkpoint that did not complete.
/// </summary>
public static CheckpointResult Failed => default;

/// <summary>
/// Creates a result denoting a completed checkpoint.
/// </summary>
/// <param name="storeTailAddress">Store tail address covered by a full checkpoint, or null for an incremental checkpoint.</param>
/// <returns>A successful result.</returns>
public static CheckpointResult Succeeded(long? storeTailAddress) =>
new() { IsSuccessful = true, StoreTailAddress = storeTailAddress };
}
}
60 changes: 51 additions & 9 deletions libs/server/Databases/DatabaseManagerBase.cs
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ internal abstract class DatabaseManagerBase : IDatabaseManager
public abstract ValueTask RecoverCheckpointAsync(bool replicaRecover = false, bool recoverFromToken = false, CheckpointMetadata metadata = null);

/// <inheritdoc/>
public abstract Task<bool> TakeCheckpointAsync(bool background, int dbId = -1, CancellationToken token = default, ILogger logger = null);
public abstract Task<CheckpointStatus> TakeCheckpointAsync(bool background, int dbId = -1, CancellationToken token = default, ILogger logger = null);

/// <inheritdoc/>
public abstract Task TakeOnDemandCheckpointAsync(DateTimeOffset entryTime, int dbId = 0);
Expand Down Expand Up @@ -246,8 +246,13 @@ private static bool AofCanReconstructFromOrigin(GarnetDatabase db, out string st
/// <param name="db">Database to checkpoint</param>
/// <param name="logger">Logger</param>
/// <param name="token">Cancellation token</param>
/// <returns>Tuple of store tail address and object store tail address</returns>
protected async Task<long?> TakeCheckpointAsync(GarnetDatabase db, ILogger logger = null, CancellationToken token = default)
/// <returns>The checkpoint outcome, including the store tail address covered by a full checkpoint</returns>
/// <remarks>
/// Failures are reported through the returned <see cref="CheckpointResult"/> rather than thrown, so that a
/// background checkpoint cannot tear down the server and so that one database's failure does not abort the
/// bookkeeping of the other databases checkpointed alongside it.
/// </remarks>
protected async Task<CheckpointResult> TakeCheckpointAsync(GarnetDatabase db, ILogger logger = null, CancellationToken token = default)
{
try
{
Expand All @@ -258,16 +263,42 @@ private static bool AofCanReconstructFromOrigin(GarnetDatabase db, out string st
lastSaveStoreTailAddress - db.LastSaveStoreTailAddress >= StoreWrapper.serverOptions.FullCheckpointLogInterval;

var checkpointType = StoreWrapper.serverOptions.UseFoldOverCheckpoints ? CheckpointType.FoldOver : CheckpointType.Snapshot;
await InitiateCheckpointAsync(db, full, checkpointType, logger).ConfigureAwait(false);
if (!await InitiateCheckpointAsync(db, full, checkpointType, logger).ConfigureAwait(false))
return CheckpointResult.Failed;

return full ? lastSaveStoreTailAddress : null;
return CheckpointResult.Succeeded(full ? lastSaveStoreTailAddress : null);
}
catch (Exception ex)
{
logger?.LogError(ex, "Checkpointing threw exception, DB ID: {id}", db.Id);
// The caller's logger is optional, so fall back to this manager's logger; otherwise a failed
// checkpoint leaves no trace at all.
(logger ?? Logger)?.LogError(ex, "Checkpointing threw exception, DB ID: {id}", db.Id);
}

return null;
return CheckpointResult.Failed;
}

/// <summary>
/// Record the outcome of a checkpoint attempt on the specified database
/// </summary>
/// <param name="db">Database that was checkpointed</param>
/// <param name="result">Outcome of the checkpoint attempt</param>
/// <remarks>
/// The last save time is advanced only for a successful checkpoint. Advancing it for a failed one reports
/// data to clients (through LASTSAVE, and to the cluster through on-demand checkpointing) as durable when
/// nothing was written.
/// </remarks>
protected static void RecordCheckpointOutcome(GarnetDatabase db, CheckpointResult result)
{
db.LastSaveSucceeded = result.IsSuccessful;

if (!result.IsSuccessful)
return;

if (result.StoreTailAddress.HasValue)
db.LastSaveStoreTailAddress = result.StoreTailAddress.Value;

db.LastSaveTime = DateTimeOffset.UtcNow;
}

/// <summary>
Expand Down Expand Up @@ -559,8 +590,8 @@ private ValueTask CompactionCommitAofAsync(GarnetDatabase db)
/// <param name="full">True if full checkpoint should be initiated</param>
/// <param name="checkpointType">Type of checkpoint</param>
/// <param name="logger">Logger</param>
/// <returns>Task</returns>
private async Task InitiateCheckpointAsync(GarnetDatabase db, bool full, CheckpointType checkpointType,
/// <returns>True if the checkpoint ran to completion</returns>
private async Task<bool> InitiateCheckpointAsync(GarnetDatabase db, bool full, CheckpointType checkpointType,
ILogger logger = null)
{
logger?.LogInformation("Initiating checkpoint; full = {full}, type = {checkpointType}, dbId = {dbId}", full, checkpointType, db.Id);
Expand Down Expand Up @@ -594,6 +625,16 @@ private async Task InitiateCheckpointAsync(GarnetDatabase db, bool full, Checkpo

checkpointResult.success = await db.StateMachineDriver.RunAsync(sm).ConfigureAwait(false);

if (!checkpointResult.success)
{
// Another state machine operation (such as an index resize) was already running, so the checkpoint
// never ran. Nothing was written, so the AOF must not be truncated and no checkpoint entry may be
// registered with the cluster - both would discard data this checkpoint does not cover.
Comment thread
TedHartMS marked this conversation as resolved.
(logger ?? Logger)?.LogWarning(
"Checkpoint did not run because another state machine operation is in progress, DB ID: {id}", db.Id);
return false;
}

// If cluster is enabled the replication manager is responsible for truncating AOF
if (StoreWrapper.serverOptions.EnableCluster && StoreWrapper.serverOptions.EnableAOF)
{
Expand All @@ -612,6 +653,7 @@ private async Task InitiateCheckpointAsync(GarnetDatabase db, bool full, Checkpo
logger ?? Logger);

logger?.LogInformation("Completed checkpoint for DB ID: {id}", db.Id);
return true;
}

internal static void RunPostCheckpointCleanup(Action cleanup, int dbId, ILogger logger)
Expand Down
8 changes: 6 additions & 2 deletions libs/server/Databases/IDatabaseManager.cs
Original file line number Diff line number Diff line change
Expand Up @@ -91,8 +91,12 @@ public interface IDatabaseManager : IDisposable
/// <param name="dbId">ID of database to checkpoint, or -1 (default) to checkpoint all active databases</param>
/// <param name="token">Cancellation token</param>
/// <param name="logger">Logger</param>
/// <returns>False if another checkpointing process is already in progress</returns>
public Task<bool> TakeCheckpointAsync(bool background, int dbId = -1, CancellationToken token = default, ILogger logger = null);
/// <returns>
/// <see cref="CheckpointStatus.AlreadyInProgress"/> if another checkpointing process is already in progress,
/// <see cref="CheckpointStatus.Failed"/> if a foreground checkpoint did not complete, otherwise
/// <see cref="CheckpointStatus.Success"/>. A background checkpoint reports success once it has started.
/// </returns>
public Task<CheckpointStatus> TakeCheckpointAsync(bool background, int dbId = -1, CancellationToken token = default, ILogger logger = null);

/// <summary>
/// Take a checkpoint if no checkpoint was taken after the provided time offset
Expand Down
Loading
Loading