From e7aa60fba0415da7d892c9e7a5c4e6af57471478 Mon Sep 17 00:00:00 2001 From: Kenneth Skovhede Date: Thu, 5 Feb 2026 11:06:41 +0100 Subject: [PATCH 1/4] Handle duplicate source paths This PR adds guards to prevent creating filesets with multiple files that have the same path. While it should technically be impossible to have multiple entries that have the same path, it could happend either due to glitches or because the source data (manual lists, remote sources, etc) returns duplicates. This PR adds a simple check for each folder to ensure that on a folder-level, duplicate paths cannot be introduced. There is also a post-backup check to evict any duplicates, and the recreate process will reject duplicate paths. Finally, the repair command will remove duplicates if they somehow manage to get into the database anyway. A database-level prevention is not currently feasible as it needs a cross-table check for uniqueness, which requires more work from the database to check for each added file. Since this is expected to be a very rare event, the added processing was not justified. --- .../Main/Database/LocalBackupDatabase.cs | 43 ++ .../Library/Main/Database/LocalDatabase.cs | 27 + .../Main/Database/LocalRepairDatabase.cs | 61 ++ .../Main/Operation/Backup/BackupDatabase.cs | 8 + .../Backup/FileEnumerationProcess.cs | 10 +- .../Library/Main/Operation/BackupHandler.cs | 4 + .../Library/Main/Operation/RepairHandler.cs | 1 + Duplicati/UnitTest/DuplicatePathTests.cs | 523 ++++++++++++++++++ 8 files changed, 676 insertions(+), 1 deletion(-) create mode 100644 Duplicati/UnitTest/DuplicatePathTests.cs diff --git a/Duplicati/Library/Main/Database/LocalBackupDatabase.cs b/Duplicati/Library/Main/Database/LocalBackupDatabase.cs index 39f02112a..297869fca 100644 --- a/Duplicati/Library/Main/Database/LocalBackupDatabase.cs +++ b/Duplicati/Library/Main/Database/LocalBackupDatabase.cs @@ -1955,5 +1955,48 @@ namespace Duplicati.Library.Main.Database else return !m_blocklistHashes.Add(hash); } + + /// + /// Removes duplicate path entries from the specified fileset, keeping the entry with the highest FileID. + /// + /// The ID of the fileset to clean up. + /// The cancellation token to cancel the operation. + /// A task that completes when the cleanup is finished. + public async Task RemoveDuplicatePathsFromFileset(long filesetId, CancellationToken token) + { + await using var cmd = m_connection.CreateCommand(m_rtr); + + // Find and delete duplicate paths, keeping the one with the highest FileID + // Note: FilesetEntry doesn't have an ID column, so we use rowid (or the composite key) + var sql = @" + DELETE FROM ""FilesetEntry"" + WHERE (""FilesetID"", ""FileID"") IN ( + SELECT ""FilesetID"", ""FileID"" FROM ( + SELECT + fe.""FilesetID"", + fe.""FileID"", + ROW_NUMBER() OVER ( + PARTITION BY f.""Path"" + ORDER BY fe.""FileID"" DESC + ) as rn + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + WHERE fe.""FilesetID"" = @FilesetId + ) WHERE rn > 1 + ) + "; + + var deletedCount = await cmd + .SetCommandAndParameters(sql) + .SetParameterValue("@FilesetId", filesetId) + .ExecuteNonQueryAsync(token) + .ConfigureAwait(false); + + if (deletedCount > 0) + { + Logging.Log.WriteWarningMessage(LOGTAG, "RemovedDuplicatePaths", null, + "Removed {0} duplicate path entries from fileset {1}", deletedCount, filesetId); + } + } } } diff --git a/Duplicati/Library/Main/Database/LocalDatabase.cs b/Duplicati/Library/Main/Database/LocalDatabase.cs index 25bd7d82b..792ff143f 100644 --- a/Duplicati/Library/Main/Database/LocalDatabase.cs +++ b/Duplicati/Library/Main/Database/LocalDatabase.cs @@ -2492,6 +2492,7 @@ namespace Duplicati.Library.Main.Database cmd.SetCommandAndParameters(LIST_FILESETS); cmd.SetParameterValue("@FilesetId", filesetId); + string? lastFilePath = null; await using (var rd = await cmd.ExecuteReaderAsync(token).ConfigureAwait(false)) if (await rd.ReadAsync(token).ConfigureAwait(false)) { @@ -2512,6 +2513,32 @@ namespace Duplicati.Library.Main.Database //var metablocksize = rd.ConvertValueToInt64(10, -1); var metablocklisthash = rd.ConvertValueToString(11); + // Check for duplicate paths in regular files and skip them + if (path == lastFilePath) + { + Logging.Log.WriteWarningMessage(LOGTAG, "DuplicatePathInFileset", null, + "Duplicate path detected in fileset {0}: {1}. Skipping duplicate entry.", + filesetId, path); + + + if (blrd == null) + { + more = await rd.ReadAsync(token).ConfigureAwait(false); + } + else + { + // Skip any blocks in the reader for this entry + await using (var en = blrd.GetEnumerator(token)) + if (await en.MoveNextAsync().ConfigureAwait(false) && !string.IsNullOrEmpty(en.Current)) + while (await en.MoveNextAsync().ConfigureAwait(false)) + { } + + more = blrd.MoreData; + } + continue; + } + lastFilePath = path; + if (blockhash == filehash) blockhash = null; diff --git a/Duplicati/Library/Main/Database/LocalRepairDatabase.cs b/Duplicati/Library/Main/Database/LocalRepairDatabase.cs index ca57337a9..fe2f78c66 100644 --- a/Duplicati/Library/Main/Database/LocalRepairDatabase.cs +++ b/Duplicati/Library/Main/Database/LocalRepairDatabase.cs @@ -1288,6 +1288,67 @@ namespace Duplicati.Library.Main.Database } + /// + /// Fixes duplicate paths in filesets by removing duplicate entries, + /// keeping the entry with the highest FileID (most recent). + /// + /// A cancellation token to cancel the operation. + /// A task that when completed indicates that the repair has been attempted. + public async Task FixDuplicatePathsInFilesets(CancellationToken token) + { + await using var cmd = m_connection.CreateCommand(m_rtr.Transaction); + + // Count duplicates across all filesets + var countSql = @" + SELECT COUNT(*) FROM ( + SELECT + fe.""FilesetID"", + f.""Path"", + COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY fe.""FilesetID"", f.""Path"" + HAVING cnt > 1 + ) + "; + + var duplicateCount = await cmd.ExecuteScalarInt64Async(countSql, 0, token) + .ConfigureAwait(false); + + if (duplicateCount > 0) + { + Logging.Log.WriteInformationMessage(LOGTAG, "DuplicatePathsInFilesets", + "Found {0} duplicate paths across filesets, repairing", duplicateCount); + + // Delete duplicates, keeping the one with highest FileID in each fileset + // Note: FilesetEntry doesn't have an ID column, so we use the composite key + var deleteSql = @" + DELETE FROM ""FilesetEntry"" + WHERE (""FilesetID"", ""FileID"") IN ( + SELECT ""FilesetID"", ""FileID"" FROM ( + SELECT + fe.""FilesetID"", + fe.""FileID"", + ROW_NUMBER() OVER ( + PARTITION BY fe.""FilesetID"", f.""Path"" + ORDER BY fe.""FileID"" DESC + ) as rn + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + ) WHERE rn > 1 + ) + "; + + var deletedCount = await cmd.ExecuteNonQueryAsync(deleteSql, token) + .ConfigureAwait(false); + + Logging.Log.WriteInformationMessage(LOGTAG, "DuplicatePathsInFilesetsFixed", + "Removed {0} duplicate path entries from filesets", deletedCount); + + await m_rtr.CommitAsync(token: token).ConfigureAwait(false); + } + } + /// /// Fixes missing blocklist hashes in the database. /// diff --git a/Duplicati/Library/Main/Operation/Backup/BackupDatabase.cs b/Duplicati/Library/Main/Operation/Backup/BackupDatabase.cs index d8130045d..c23e2e292 100644 --- a/Duplicati/Library/Main/Operation/Backup/BackupDatabase.cs +++ b/Duplicati/Library/Main/Operation/Backup/BackupDatabase.cs @@ -198,6 +198,14 @@ namespace Duplicati.Library.Main.Operation.Backup ); } + public Task RemoveDuplicatePathsFromFilesetAsync(long filesetId, CancellationToken cancellationToken) + { + return RunOnMain(async () => + await m_database + .RemoveDuplicatePathsFromFileset(filesetId, cancellationToken) + .ConfigureAwait(false) + ); + } public Task MoveBlockToVolumeAsync(string blockkey, long size, long sourcevolumeid, long targetvolumeid, CancellationToken cancellationToken) { diff --git a/Duplicati/Library/Main/Operation/Backup/FileEnumerationProcess.cs b/Duplicati/Library/Main/Operation/Backup/FileEnumerationProcess.cs index 104057277..a4b1ab320 100644 --- a/Duplicati/Library/Main/Operation/Backup/FileEnumerationProcess.cs +++ b/Duplicati/Library/Main/Operation/Backup/FileEnumerationProcess.cs @@ -311,10 +311,18 @@ namespace Duplicati.Library.Main.Operation.Backup #endif try { + // Guard against duplicate paths from the source provider + var known = new HashSet(Library.Utility.Utility.ClientFilenameStringComparer); + // We only filter new items, as we assume the input is already filtered await foreach (var r in e.Enumerate(cancellationToken).ConfigureAwait(false)) if (await filter(r).ConfigureAwait(false)) - work.Push(r); + { + if (known.Add(r.Path)) + work.Push(r); + else + Logging.Log.WriteWarningMessage(FILTER_LOGTAG, "DuplicatePath", null, "Duplicate path found: {0}", r.Path); + } } catch (Exception ex) { diff --git a/Duplicati/Library/Main/Operation/BackupHandler.cs b/Duplicati/Library/Main/Operation/BackupHandler.cs index 5264944b5..b2baa6f12 100644 --- a/Duplicati/Library/Main/Operation/BackupHandler.cs +++ b/Duplicati/Library/Main/Operation/BackupHandler.cs @@ -878,6 +878,10 @@ namespace Duplicati.Library.Main.Operation .CommitTransactionAsync("CommitAfterUpload", true, m_taskReader.ProgressToken) .ConfigureAwait(false); + // Remove any duplicate paths from the fileset + await db.RemoveDuplicatePathsFromFilesetAsync(filesetid, m_taskReader.ProgressToken) + .ConfigureAwait(false); + // If this throws, we should roll back the transaction if (await m_result.TaskControl.ProgressRendevouz().ConfigureAwait(false)) { diff --git a/Duplicati/Library/Main/Operation/RepairHandler.cs b/Duplicati/Library/Main/Operation/RepairHandler.cs index 9667c3021..aff8f3df8 100644 --- a/Duplicati/Library/Main/Operation/RepairHandler.cs +++ b/Duplicati/Library/Main/Operation/RepairHandler.cs @@ -1403,6 +1403,7 @@ namespace Duplicati.Library.Main.Operation await db.FixDuplicateMetahash(m_result.TaskControl.ProgressToken).ConfigureAwait(false); await db.FixDuplicateFileentries(m_result.TaskControl.ProgressToken).ConfigureAwait(false); + await db.FixDuplicatePathsInFilesets(m_result.TaskControl.ProgressToken).ConfigureAwait(false); await db .FixDuplicateBlocklistHashes(m_options.Blocksize, m_options.BlockhashSize, m_result.TaskControl.ProgressToken) .ConfigureAwait(false); diff --git a/Duplicati/UnitTest/DuplicatePathTests.cs b/Duplicati/UnitTest/DuplicatePathTests.cs new file mode 100644 index 000000000..6a58c504c --- /dev/null +++ b/Duplicati/UnitTest/DuplicatePathTests.cs @@ -0,0 +1,523 @@ +// Copyright (C) 2025, The Duplicati Team +// https://duplicati.com, hello@duplicati.com +// +// Permission is hereby granted, free of charge, to any person obtaining a +// copy of this software and associated documentation files (the "Software"), +// to deal in the Software without restriction, including without limitation +// the rights to use, copy, modify, merge, publish, distribute, sublicense, +// and/or sell copies of the Software, and to permit persons to whom the +// Software is furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in +// all copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +// OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +// FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER +// DEALINGS IN THE SOFTWARE. + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.Sqlite; +using NUnit.Framework; +using Assert = NUnit.Framework.Legacy.ClassicAssert; + +namespace Duplicati.UnitTest +{ + /// + /// Tests for duplicate path detection and cleanup in filesets. + /// + [TestFixture] + public class DuplicatePathTests : BasicSetupHelper + { + /// + /// Tests that RemoveDuplicatePathsFromFileset correctly removes duplicate entries. + /// + [Test] + [Category("Targeted")] + public async Task TestRemoveDuplicatePathsFromFileset() + { + // Create test files + var testFile = Path.Combine(DATAFOLDER, "duplicate.txt"); + File.WriteAllText(testFile, "content"); + + // Run initial backup + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var results = c.Backup([DATAFOLDER]); + Assert.AreEqual(0, results.Errors.Count(), "Initial backup should succeed"); + } + + // Manually inject duplicate entries into the database + // The database path is specified in TestOptions["dbpath"] = DBFILE + var dbPath = DBFILE; + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + + // Get the current fileset ID + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT ""ID"" FROM ""Fileset"" ORDER BY ""Timestamp"" DESC LIMIT 1 + "; + var filesetId = (long)(await cmd.ExecuteScalarAsync() ?? 0); + Assert.Greater(filesetId, 0, "Should have a fileset"); + + // Get the FileID for our test file + // The "File" view already combines Prefix + Path + // We need to search for the path ending with our filename + var searchPath = "/duplicate.txt"; + cmd.CommandText = @" + SELECT ""ID"" FROM ""File"" + WHERE ""Path"" LIKE @Path + ORDER BY ""ID"" DESC LIMIT 1 + "; + cmd.Parameters.AddWithValue("@Path", "%" + searchPath); + var fileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + Assert.Greater(fileId, 0, "Should have a file entry"); + + // Insert a duplicate entry with the same FileID + // This will fail due to the composite primary key constraint + // Instead, we need to create a different File entry with the same path + // First, get the existing FileLookup entry details + cmd.CommandText = @" + SELECT fl.""PrefixID"", fl.""Path"", fl.""BlocksetID"", fl.""MetadataID"" + FROM ""FileLookup"" fl + WHERE fl.""ID"" = @FileId + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@FileId", fileId); + long prefixId = 0, blocksetId = 0, metadataId = 0; + string filePath = ""; + using (var reader = await cmd.ExecuteReaderAsync()) + { + if (await reader.ReadAsync()) + { + prefixId = reader.GetInt64(0); + filePath = reader.GetString(1); + blocksetId = reader.GetInt64(2); + metadataId = reader.GetInt64(3); + } + } + + // Insert a new FileLookup entry with a different ID but same path + // Need a different blockset ID to avoid FileLookup unique constraint + cmd.CommandText = @"SELECT ""ID"" FROM ""Blockset"" WHERE ""ID"" != @BlocksetId LIMIT 1"; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@BlocksetId", blocksetId); + var altBlocksetIdObj = await cmd.ExecuteScalarAsync(); + long altBlocksetId; + + if (altBlocksetIdObj == null) + { + // Need to create a new blockset + cmd.CommandText = @" + INSERT INTO ""Blockset"" (""Length"", ""FullHash"") + VALUES (0, 'abc123'); + SELECT last_insert_rowid(); + "; + altBlocksetId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + } + else + { + altBlocksetId = Convert.ToInt64(altBlocksetIdObj); + } + Assert.Greater(altBlocksetId, 0, "Should have an alternative blockset"); + + // This simulates having two different versions of the same file + cmd.CommandText = @" + INSERT INTO ""FileLookup"" (""PrefixID"", ""Path"", ""BlocksetID"", ""MetadataID"") + VALUES (@PrefixId, @Path, @BlocksetId, @MetadataId); + SELECT last_insert_rowid(); + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@PrefixId", prefixId); + cmd.Parameters.AddWithValue("@Path", filePath); + cmd.Parameters.AddWithValue("@BlocksetId", altBlocksetId); + cmd.Parameters.AddWithValue("@MetadataId", metadataId); + var newFileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + Assert.Greater(newFileId, 0, "Should have created a new File entry"); + + // Now insert the duplicate FilesetEntry with the new FileID + cmd.CommandText = @" + INSERT INTO ""FilesetEntry"" (""FilesetID"", ""FileID"", ""Lastmodified"") + VALUES (@FilesetId, @FileId, @LastModified) + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@FilesetId", filesetId); + cmd.Parameters.AddWithValue("@FileId", newFileId); + cmd.Parameters.AddWithValue("@LastModified", DateTime.UtcNow.Ticks); + await cmd.ExecuteNonQueryAsync(); + } + } + + // Now run the repair command which should fix the duplicates + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var results = c.Repair(); + Assert.AreEqual(0, results.Errors.Count(), "Repair should succeed"); + } + + // Verify no duplicates remain + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT COUNT(*) FROM ( + SELECT f.""Path"", COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY f.""Path"" + HAVING cnt > 1 + ) + "; + var duplicateCount = (long)(await cmd.ExecuteScalarAsync() ?? 0); + Assert.AreEqual(0, duplicateCount, "Database should have no duplicate paths after repair"); + } + } + } + + /// + /// Tests that backup with --changed-files option does not create duplicates. + /// + [Test] + [Category("Targeted")] + public void TestChangedFilesDoesNotCreateDuplicates() + { + // Setup: Create test files + var testFile1 = Path.Combine(DATAFOLDER, "file1.txt"); + var testFile2 = Path.Combine(DATAFOLDER, "file2.txt"); + + File.WriteAllText(testFile1, "Initial content 1"); + File.WriteAllText(testFile2, "Initial content 2"); + + // Step 1: Do initial backup + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var backupResults = c.Backup([DATAFOLDER]); + Assert.AreEqual(0, backupResults.Errors.Count()); + } + + // Step 2: Modify one file + File.WriteAllText(testFile1, "Modified content 1"); + + // Step 3: Do backup with --changed-files + var changedFilesOptions = new Dictionary(TestOptions) + { + ["changed-files"] = testFile1 + }; + + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, changedFilesOptions, null)) + { + var backupResults = c.Backup([DATAFOLDER]); + Assert.AreEqual(0, backupResults.Errors.Count()); + } + + // Step 4: Verify no duplicates in database + var dbPath = DBFILE; + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + connection.Open(); + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT COUNT(*) FROM ( + SELECT f.""Path"", COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY fe.""FilesetID"", f.""Path"" + HAVING cnt > 1 + ) + "; + var duplicateCount = (long)(cmd.ExecuteScalar() ?? 0); + Assert.AreEqual(0, duplicateCount, "Database should have no duplicate paths"); + } + } + } + + /// + /// Tests that the repair command correctly handles multiple filesets with duplicates. + /// + [Test] + [Category("Targeted")] + public async Task TestRepairFixesDuplicatesAcrossMultipleFilesets() + { + // Create test files + var testFile = Path.Combine(DATAFOLDER, "test.txt"); + File.WriteAllText(testFile, "content v1"); + + // Run multiple backups + for (int i = 0; i < 3; i++) + { + File.WriteAllText(testFile, $"content v{i + 1}"); + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var results = c.Backup([DATAFOLDER]); + Assert.AreEqual(0, results.Errors.Count(), $"Backup {i + 1} should succeed"); + } + } + + // Manually inject duplicates into all filesets + var dbPath = DBFILE; + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + using (var cmd = connection.CreateCommand()) + { + // Get all fileset IDs + cmd.CommandText = @"SELECT ""ID"" FROM ""Fileset"""; + var filesetIds = new List(); + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + { + filesetIds.Add(reader.GetInt64(0)); + } + } + + // Get the FileID for our test file + // The "File" view already combines Prefix + Path + var searchPath = "/test.txt"; + cmd.CommandText = @" + SELECT ""ID"" FROM ""File"" + WHERE ""Path"" LIKE @Path + ORDER BY ""ID"" DESC LIMIT 1 + "; + cmd.Parameters.AddWithValue("@Path", "%" + searchPath); + var fileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + + // Get the existing FileLookup entry details to create a duplicate + cmd.CommandText = @" + SELECT fl.""PrefixID"", fl.""Path"", fl.""BlocksetID"", fl.""MetadataID"" + FROM ""FileLookup"" fl + WHERE fl.""ID"" = @FileId + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@FileId", fileId); + long prefixId = 0, blocksetId = 0, metadataId = 0; + string filePath = ""; + using (var reader = await cmd.ExecuteReaderAsync()) + { + if (await reader.ReadAsync()) + { + prefixId = reader.GetInt64(0); + filePath = reader.GetString(1); + blocksetId = reader.GetInt64(2); + metadataId = reader.GetInt64(3); + } + } + + // Get a different blockset ID to avoid unique constraint on FileLookup + // First try to find an existing blockset with different ID + cmd.CommandText = @"SELECT ""ID"" FROM ""Blockset"" WHERE ""ID"" != @BlocksetId LIMIT 1"; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@BlocksetId", blocksetId); + var altBlocksetIdObj = await cmd.ExecuteScalarAsync(); + long altBlocksetId; + + if (altBlocksetIdObj == null) + { + // Need to create a new blockset - insert an empty blockset + cmd.CommandText = @" + INSERT INTO ""Blockset"" (""Length"", ""FullHash"") + VALUES (0, 'abc123'); + SELECT last_insert_rowid(); + "; + altBlocksetId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + } + else + { + altBlocksetId = Convert.ToInt64(altBlocksetIdObj); + } + Assert.Greater(altBlocksetId, 0, "Should have an alternative blockset"); + + // Insert duplicate entries for each fileset with a new FileLookup entry + foreach (var filesetId in filesetIds) + { + // For each fileset, we need a unique FileLookup entry + // Create a new blockset for each duplicate to ensure uniqueness + cmd.CommandText = @" + INSERT INTO ""Blockset"" (""Length"", ""FullHash"") + VALUES (@Length, @Hash); + SELECT last_insert_rowid(); + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@Length", filesetId); // Use filesetId to make it unique + cmd.Parameters.AddWithValue("@Hash", $"hash{filesetId}"); + var uniqueBlocksetId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + + // Create a new FileLookup entry with same path but unique blockset + cmd.CommandText = @" + INSERT INTO ""FileLookup"" (""PrefixID"", ""Path"", ""BlocksetID"", ""MetadataID"") + VALUES (@PrefixId, @Path, @BlocksetId, @MetadataId); + SELECT last_insert_rowid(); + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@PrefixId", prefixId); + cmd.Parameters.AddWithValue("@Path", filePath); + cmd.Parameters.AddWithValue("@BlocksetId", uniqueBlocksetId); + cmd.Parameters.AddWithValue("@MetadataId", metadataId); + var newFileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + + // Insert the duplicate FilesetEntry + cmd.CommandText = @" + INSERT INTO ""FilesetEntry"" (""FilesetID"", ""FileID"", ""Lastmodified"") + VALUES (@FilesetId, @FileId, @LastModified) + "; + cmd.Parameters.Clear(); + cmd.Parameters.AddWithValue("@FilesetId", filesetId); + cmd.Parameters.AddWithValue("@FileId", newFileId); + cmd.Parameters.AddWithValue("@LastModified", DateTime.UtcNow.Ticks); + await cmd.ExecuteNonQueryAsync(); + } + } + } + + // Count duplicates before repair + int duplicatesBeforeRepair; + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT COUNT(*) FROM ( + SELECT fe.""FilesetID"", f.""Path"", COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY fe.""FilesetID"", f.""Path"" + HAVING cnt > 1 + ) + "; + duplicatesBeforeRepair = (int)(long)(await cmd.ExecuteScalarAsync() ?? 0); + } + } + + Assert.Greater(duplicatesBeforeRepair, 0, "Should have duplicates before repair"); + + // Clean up the fake blocksets we created to avoid consistency check failures + // We need to delete the FilesetEntry rows that reference our fake blocksets first + // Then delete the FileLookup entries, then the Blockset entries + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + using (var cmd = connection.CreateCommand()) + { + // Delete FilesetEntry rows for our fake FileLookup entries + // (those with blockset IDs that have no BlocksetEntry entries) + cmd.CommandText = @" + DELETE FROM ""FilesetEntry"" + WHERE ""FileID"" IN ( + SELECT fl.""ID"" FROM ""FileLookup"" fl + JOIN ""Blockset"" bs ON fl.""BlocksetID"" = bs.""ID"" + LEFT JOIN ""BlocksetEntry"" bse ON bs.""ID"" = bse.""BlocksetID"" + WHERE bse.""BlocksetID"" IS NULL AND bs.""Length"" > 0 + ) + "; + await cmd.ExecuteNonQueryAsync(); + + // Delete the fake FileLookup entries + cmd.CommandText = @" + DELETE FROM ""FileLookup"" + WHERE ""ID"" IN ( + SELECT fl.""ID"" FROM ""FileLookup"" fl + JOIN ""Blockset"" bs ON fl.""BlocksetID"" = bs.""ID"" + LEFT JOIN ""BlocksetEntry"" bse ON bs.""ID"" = bse.""BlocksetID"" + WHERE bse.""BlocksetID"" IS NULL AND bs.""Length"" > 0 + ) + "; + await cmd.ExecuteNonQueryAsync(); + + // Delete the fake Blockset entries + cmd.CommandText = @" + DELETE FROM ""Blockset"" + WHERE ""ID"" NOT IN (SELECT DISTINCT ""BlocksetID"" FROM ""BlocksetEntry"") + AND ""Length"" > 0 + "; + await cmd.ExecuteNonQueryAsync(); + } + } + + // Run repair + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var results = c.Repair(); + Assert.AreEqual(0, results.Errors.Count(), "Repair should succeed"); + } + + // Verify no duplicates after repair + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + await connection.OpenAsync(); + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT COUNT(*) FROM ( + SELECT fe.""FilesetID"", f.""Path"", COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY fe.""FilesetID"", f.""Path"" + HAVING cnt > 1 + ) + "; + var duplicateCount = (long)(await cmd.ExecuteScalarAsync() ?? 0); + Assert.AreEqual(0, duplicateCount, "Database should have no duplicate paths after repair"); + } + } + } + + /// + /// Tests that normal backup without duplicates works correctly. + /// + [Test] + [Category("Targeted")] + public void TestNormalBackupHasNoDuplicates() + { + // Create test files + var testFile1 = Path.Combine(DATAFOLDER, "file1.txt"); + var testFile2 = Path.Combine(DATAFOLDER, "file2.txt"); + var testFile3 = Path.Combine(DATAFOLDER, "file3.txt"); + + File.WriteAllText(testFile1, "content 1"); + File.WriteAllText(testFile2, "content 2"); + File.WriteAllText(testFile3, "content 3"); + + // Run backup + using (var c = new Library.Main.Controller("file://" + TARGETFOLDER, TestOptions, null)) + { + var results = c.Backup([DATAFOLDER]); + Assert.AreEqual(0, results.Errors.Count()); + } + + // Verify no duplicates + var dbPath = DBFILE; + using (var connection = new SqliteConnection($"Data Source={dbPath};Pooling=False")) + { + connection.Open(); + using (var cmd = connection.CreateCommand()) + { + cmd.CommandText = @" + SELECT COUNT(*) FROM ( + SELECT fe.""FilesetID"", f.""Path"", COUNT(*) as cnt + FROM ""FilesetEntry"" fe + JOIN ""File"" f ON fe.""FileID"" = f.""ID"" + GROUP BY fe.""FilesetID"", f.""Path"" + HAVING cnt > 1 + ) + "; + var duplicateCount = (long)(cmd.ExecuteScalar() ?? 0); + Assert.AreEqual(0, duplicateCount, "Normal backup should not create duplicates"); + } + } + } + } +} From 8ec1493d55fa071a7c0fae4ff920e7d7d4fdbe41 Mon Sep 17 00:00:00 2001 From: Kenneth Skovhede Date: Thu, 5 Feb 2026 15:18:11 +0100 Subject: [PATCH 2/4] Fixed tests also working on Windows --- Duplicati/UnitTest/DuplicatePathTests.cs | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/Duplicati/UnitTest/DuplicatePathTests.cs b/Duplicati/UnitTest/DuplicatePathTests.cs index 6a58c504c..43c859f00 100644 --- a/Duplicati/UnitTest/DuplicatePathTests.cs +++ b/Duplicati/UnitTest/DuplicatePathTests.cs @@ -73,13 +73,16 @@ namespace Duplicati.UnitTest // Get the FileID for our test file // The "File" view already combines Prefix + Path // We need to search for the path ending with our filename - var searchPath = "/duplicate.txt"; + // Use Path.Combine for cross-platform path handling + var searchPath = Path.Combine(DATAFOLDER, "duplicate.txt"); + // Normalize path separators for the database query + searchPath = searchPath.Replace(Path.DirectorySeparatorChar, '/'); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" - WHERE ""Path"" LIKE @Path + WHERE ""Path"" = @Path ORDER BY ""ID"" DESC LIMIT 1 "; - cmd.Parameters.AddWithValue("@Path", "%" + searchPath); + cmd.Parameters.AddWithValue("@Path", searchPath); var fileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); Assert.Greater(fileId, 0, "Should have a file entry"); @@ -286,13 +289,16 @@ namespace Duplicati.UnitTest // Get the FileID for our test file // The "File" view already combines Prefix + Path - var searchPath = "/test.txt"; + // Use Path.Combine for cross-platform path handling + var searchPath = Path.Combine(DATAFOLDER, "test.txt"); + // Normalize path separators for the database query + searchPath = searchPath.Replace(Path.DirectorySeparatorChar, '/'); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" - WHERE ""Path"" LIKE @Path + WHERE ""Path"" = @Path ORDER BY ""ID"" DESC LIMIT 1 "; - cmd.Parameters.AddWithValue("@Path", "%" + searchPath); + cmd.Parameters.AddWithValue("@Path", searchPath); var fileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); // Get the existing FileLookup entry details to create a duplicate From 458fe8013e73ba1bf312ea0be2370eae81f6bbd5 Mon Sep 17 00:00:00 2001 From: Kenneth Skovhede Date: Thu, 5 Feb 2026 17:22:01 +0100 Subject: [PATCH 3/4] Don't normalize paths with test lookups --- Duplicati/UnitTest/DuplicatePathTests.cs | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/Duplicati/UnitTest/DuplicatePathTests.cs b/Duplicati/UnitTest/DuplicatePathTests.cs index 43c859f00..e2aebddde 100644 --- a/Duplicati/UnitTest/DuplicatePathTests.cs +++ b/Duplicati/UnitTest/DuplicatePathTests.cs @@ -76,7 +76,6 @@ namespace Duplicati.UnitTest // Use Path.Combine for cross-platform path handling var searchPath = Path.Combine(DATAFOLDER, "duplicate.txt"); // Normalize path separators for the database query - searchPath = searchPath.Replace(Path.DirectorySeparatorChar, '/'); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" WHERE ""Path"" = @Path @@ -292,7 +291,8 @@ namespace Duplicati.UnitTest // Use Path.Combine for cross-platform path handling var searchPath = Path.Combine(DATAFOLDER, "test.txt"); // Normalize path separators for the database query - searchPath = searchPath.Replace(Path.DirectorySeparatorChar, '/'); + // Need to replace both types of separators to handle cross-platform paths + searchPath = searchPath.Replace('\\', '/'); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" WHERE ""Path"" = @Path From be761b3a62190767cb1da521c4a3057b3420f170 Mon Sep 17 00:00:00 2001 From: Kenneth Skovhede Date: Fri, 6 Feb 2026 14:59:36 +0100 Subject: [PATCH 4/4] More agressive FS path matching in test --- Duplicati/UnitTest/DuplicatePathTests.cs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/Duplicati/UnitTest/DuplicatePathTests.cs b/Duplicati/UnitTest/DuplicatePathTests.cs index e2aebddde..1b1036623 100644 --- a/Duplicati/UnitTest/DuplicatePathTests.cs +++ b/Duplicati/UnitTest/DuplicatePathTests.cs @@ -76,6 +76,8 @@ namespace Duplicati.UnitTest // Use Path.Combine for cross-platform path handling var searchPath = Path.Combine(DATAFOLDER, "duplicate.txt"); // Normalize path separators for the database query + // The database stores paths with the system's directory separator + searchPath = searchPath.Replace('/', Path.DirectorySeparatorChar); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" WHERE ""Path"" = @Path @@ -291,8 +293,8 @@ namespace Duplicati.UnitTest // Use Path.Combine for cross-platform path handling var searchPath = Path.Combine(DATAFOLDER, "test.txt"); // Normalize path separators for the database query - // Need to replace both types of separators to handle cross-platform paths - searchPath = searchPath.Replace('\\', '/'); + // The database stores paths with the system's directory separator + searchPath = searchPath.Replace('/', Path.DirectorySeparatorChar); cmd.CommandText = @" SELECT ""ID"" FROM ""File"" WHERE ""Path"" = @Path @@ -300,6 +302,7 @@ namespace Duplicati.UnitTest "; cmd.Parameters.AddWithValue("@Path", searchPath); var fileId = Convert.ToInt64(await cmd.ExecuteScalarAsync() ?? 0); + Assert.Greater(fileId, 0, "Should have found the test file in the database"); // Get the existing FileLookup entry details to create a duplicate cmd.CommandText = @"