From 04d2896b61908ddf8bd7839883969254f61f55b2 Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 28 Sep 2026 14:29:33 +0000 Subject: [PATCH] fix: skip named pipes, sockets and device nodes when scanning [patch] On Unix, Directory.EnumerateFiles yields every entry that is not a directory, and .NET marks named pipes, sockets and device nodes with the same Normal attribute as a regular file. The hasher then opened a named pipe for reading, which blocks until a writer appears, so a single pipe anywhere in the tree hung Scan, DryRun, Stats and Deduplicate. ScanForFiles now keeps only regular files. The file type is read through the runtime's SystemNative_LStat shim, whose FileStatus layout the runtime defines with fixed-width fields and normalized type bits on every Unix, so no per-platform struct stat layout is involved. Fixes ktsu-dev/FileDeduplicator#136 Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01PNmrp6FP3tBU47owLsiovc --- .../FileScannerAndHasherTests.cs | 62 ++++++++++++++++++ FileDeduplicator/FileDeduplicator.csproj | 2 + FileDeduplicator/FileScanner.cs | 11 ++++ FileDeduplicator/FileType.cs | 65 +++++++++++++++++++ 4 files changed, 140 insertions(+) create mode 100644 FileDeduplicator/FileType.cs diff --git a/FileDeduplicator.Test/FileScannerAndHasherTests.cs b/FileDeduplicator.Test/FileScannerAndHasherTests.cs index 31e2890..cb97c57 100644 --- a/FileDeduplicator.Test/FileScannerAndHasherTests.cs +++ b/FileDeduplicator.Test/FileScannerAndHasherTests.cs @@ -2,6 +2,8 @@ namespace ktsu.FileDeduplicator.Test; +using System.Diagnostics; + using ktsu.Semantics.Paths; using ktsu.Semantics.Strings; using Microsoft.VisualStudio.TestTools.UnitTesting; @@ -127,6 +129,50 @@ public void ScanTerminatesOnADirectorySymlinkCycle() Assert.Contains(real, files); } + /// + /// A named pipe must not reach the hasher. Opening one for reading blocks until a writer + /// appears, so a single pipe anywhere in the tree hung every verb. + /// + [TestMethod] + public void ScanLeavesOutANamedPipe() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath regular = tree.Write("regular.txt", "content"); + AbsoluteFilePath empty = tree.Write("nested/empty.txt", string.Empty); + MakeNamedPipe(Path.Combine(tree.Root.WeakString, "nested", "pipe")); + + // Act + IReadOnlyList files = FileScanner.ScanForFiles(tree.Root); + + // Assert -- regular files, including an empty one, are still found + Assert.HasCount(2, files); + Assert.Contains(regular, files); + Assert.Contains(empty, files); + } + + /// + /// Scanning and hashing a tree that holds a named pipe must finish, which is what the user sees + /// of the fix. A hang fails the test after a timeout rather than stalling the run. + /// + [TestMethod] + public void ScanAndHashFinishWhenTheTreeHoldsANamedPipe() + { + // Arrange + using TempTree tree = new(); + AbsoluteFilePath regular = tree.Write("a.txt", "a"); + MakeNamedPipe(Path.Combine(tree.Root.WeakString, "pipe")); + + // Act + Task> pass = Task.Run(() => FileHasher.HashFiles(FileScanner.ScanForFiles(tree.Root))); + bool finished = pass.Wait(TimeSpan.FromSeconds(30)); + + // Assert + Assert.IsTrue(finished, "Scanning and hashing did not finish, so the named pipe was opened."); + Assert.ContainsSingle(pass.Result); + Assert.IsTrue(pass.Result.ContainsKey(regular)); + } + /// /// Identical content must hash identically regardless of the file's name or location. /// @@ -257,4 +303,20 @@ public void HashingSkipsAnUnreadableFileAndStillHashesTheRest() Assert.AreEqual(FileHasher.ComputeHash(alsoReadable), hashes[alsoReadable]); Assert.IsFalse(hashes.ContainsKey(unreadable)); } + + /// + /// Creates a named pipe with mkfifo, or marks the test inconclusive where there is none. + /// + /// Where to create the pipe. + private static void MakeNamedPipe(string path) + { + if (OperatingSystem.IsWindows()) + { + Assert.Inconclusive("Windows keeps named pipes out of the file system, so there is nothing to stage."); + } + + using Process mkfifo = Process.Start("mkfifo", [path]); + mkfifo.WaitForExit(); + Assert.AreEqual(0, mkfifo.ExitCode, $"mkfifo could not create {path}."); + } } diff --git a/FileDeduplicator/FileDeduplicator.csproj b/FileDeduplicator/FileDeduplicator.csproj index 80e032e..9c809fe 100644 --- a/FileDeduplicator/FileDeduplicator.csproj +++ b/FileDeduplicator/FileDeduplicator.csproj @@ -8,6 +8,8 @@ $(NoWarn);CA1812 + true + diff --git a/FileDeduplicator/FileScanner.cs b/FileDeduplicator/FileScanner.cs index 9c10612..c9db82f 100644 --- a/FileDeduplicator/FileScanner.cs +++ b/FileDeduplicator/FileScanner.cs @@ -30,6 +30,10 @@ internal static class FileScanner /// not skipped, which a bare new EnumerationOptions() would do. Those files were scanned /// before this change, and a duplicate among them is still a duplicate. /// + /// + /// No attribute marks a named pipe, socket or device node, so filters + /// those out itself through . + /// /// private static readonly EnumerationOptions WalkOptions = new() { @@ -49,6 +53,13 @@ internal static IReadOnlyList ScanForFiles(AbsoluteDirectoryPa List files = []; foreach (string file in Directory.EnumerateFiles(path.WeakString, "*", WalkOptions)) { + // On Unix the enumeration also yields named pipes, sockets and device nodes. Opening a + // pipe to hash it blocks until a writer appears, and none of them is a copy of anything. + if (!FileType.IsRegularFile(file)) + { + continue; + } + files.Add(file.As()); } diff --git a/FileDeduplicator/FileType.cs b/FileDeduplicator/FileType.cs new file mode 100644 index 0000000..b0e1e60 --- /dev/null +++ b/FileDeduplicator/FileType.cs @@ -0,0 +1,65 @@ +// Copyright (c) 2023-2026 ktsu-dev contributors + +namespace ktsu.FileDeduplicator; + +using System.Runtime.InteropServices; + +/// +/// Tells regular files apart from the other entries a Unix directory can hold. +/// +/// +/// On Unix, returns every +/// entry that is not a directory, so named pipes, sockets and device nodes arrive alongside +/// regular files. .NET reports all of them with , and exposes no +/// managed way to read the file type. Opening a named pipe for reading blocks until something +/// opens it for writing, which hung every verb; a device node such as /dev/zero never +/// reaches end of file. +/// +/// The type is read through SystemNative_LStat, the shim the runtime itself uses to +/// implement on Unix. Unlike lstat(2), whose struct stat layout +/// differs between operating systems and architectures, the shim fills a structure the runtime +/// defines with fixed-width fields, and normalizes the file type bits to the same values on every +/// platform. Only the leading Mode field is read, and the buffer is oversized, so a field +/// the runtime appends later cannot overrun it. +/// +/// +internal static partial class FileType +{ + private const int TypeMask = 0xF000; + private const int RegularFile = 0x8000; + + /// + /// Gets whether the entry at is a regular file, the only kind whose + /// content can be hashed and whose deletion reclaims space. + /// + /// The entry to inspect. A symlink is inspected, not followed. + /// + /// for a named pipe, socket or device node. for a + /// regular file, on Windows, and when the entry cannot be inspected, so that hashing it reports + /// the problem as it did before. + /// + internal static bool IsRegularFile(string path) + { + if (OperatingSystem.IsWindows()) + { + return true; + } + + return LStat(path, out FileStatus status) != 0 || (status.Mode & TypeMask) == RegularFile; + } + + [LibraryImport("libSystem.Native", EntryPoint = "SystemNative_LStat", StringMarshalling = StringMarshalling.Utf8)] + [DefaultDllImportSearchPaths(DllImportSearchPath.SafeDirectories)] + private static partial int LStat(string path, out FileStatus output); + + /// + /// The runtime's FileStatus, of which only Mode is read. It follows the 32-bit + /// Flags field, and both have led the structure since the shim was introduced. + /// + [StructLayout(LayoutKind.Explicit, Size = 256)] + private struct FileStatus + { + [FieldOffset(4)] + internal int Mode; + } +}