diff --git a/ImageDescriber.Test/ImportTests.cs b/ImageDescriber.Test/ImportTests.cs index 91cb4db..b9b175e 100644 --- a/ImageDescriber.Test/ImportTests.cs +++ b/ImageDescriber.Test/ImportTests.cs @@ -9,6 +9,10 @@ namespace ktsu.ImageDescriber.Tests; [TestClass] public class ImportTests { + private static readonly string[] ExpectedHeader = ["h1", "h2"]; + private static readonly string[] ExpectedMultiLineRecord = ["a\nb", "c"]; + private static readonly string[] ExpectedEscapedQuoteRecord = ["d", "e\"f"]; + [TestMethod] public void ParseCsvLineSimpleFields() { @@ -184,4 +188,72 @@ public void MergeKnownPathsReturnsFalseWhenSourceEmpty() Assert.IsFalse(result); Assert.AreEqual(1, target.KnownPaths.Count); } + + [TestMethod] + public void CsvExportThenImportRoundTripsEveryField() + { + string tempDir = Path.GetTempPath(); + ImageDescription[] originals = + [ + new() + { + Hash = new string('a', 64), + SuggestedFileName = "dog-on-beach.jpg".As(), + KnownPaths = [Path.Combine(tempDir, "a.jpg").As(), Path.Combine(tempDir, "b.jpg").As()], + Model = "llama3.2-vision".As(), + DescribedAt = new DateTime(2026, 9, 26, 7, 10, 12, DateTimeKind.Utc), + FileSizeBytes = 12345, + Description = "A dog on a beach.\n\nThe sky is blue, and the dog says \"woof\".", + }, + new() + { + Hash = new string('b', 64), + SuggestedFileName = "dog, cat 'n' friends.jpg".As(), + KnownPaths = [Path.Combine(tempDir, "c.jpg").As()], + Model = "llava".As(), + DescribedAt = new DateTime(2026, 1, 2, 3, 4, 5, DateTimeKind.Utc), + FileSizeBytes = 1, + Description = "Two animals.\r\nOn a sofa.", + }, + new() + { + Hash = new string('c', 64), + SuggestedFileName = "plain.jpg".As(), + KnownPaths = [Path.Combine(tempDir, "d.jpg").As()], + Model = "llava".As(), + DescribedAt = new DateTime(2026, 1, 2, 3, 4, 5, DateTimeKind.Utc), + FileSizeBytes = 2, + Description = "Simple.", + }, + ]; + + List imported = Import.ParseCsv(Export.BuildCsv(originals)); + + Assert.HasCount(originals.Length, imported); + for (int i = 0; i < originals.Length; i++) + { + ImageDescription expected = originals[i]; + ImageDescription actual = imported[i]; + Assert.AreEqual(expected.Hash, actual.Hash); + Assert.AreEqual(expected.SuggestedFileName, actual.SuggestedFileName); + Assert.AreSequenceEqual(expected.KnownPaths, actual.KnownPaths); + Assert.AreEqual(expected.Model, actual.Model); + Assert.AreEqual(expected.DescribedAt, actual.DescribedAt); + Assert.AreEqual(expected.FileSizeBytes, actual.FileSizeBytes); + Assert.AreEqual(expected.Description, actual.Description); + } + } + + [TestMethod] + public void ParseCsvRecordsKeepsLineBreaksInsideQuotedFields() + { + List<(int LineNumber, List Fields)> records = Import.ParseCsvRecords("h1,h2\r\n\"a\nb\",c\r\n\r\nd,\"e\"\"f\"\n"); + + Assert.HasCount(3, records); + Assert.AreSequenceEqual(ExpectedHeader, records[0].Fields); + Assert.AreSequenceEqual(ExpectedMultiLineRecord, records[1].Fields); + Assert.AreEqual(2, records[1].LineNumber); + Assert.AreSequenceEqual(ExpectedEscapedQuoteRecord, records[2].Fields); + Assert.AreEqual(5, records[2].LineNumber); + } } diff --git a/ImageDescriber/Verbs/Export.cs b/ImageDescriber/Verbs/Export.cs index bb1bcca..f20a7e2 100644 --- a/ImageDescriber/Verbs/Export.cs +++ b/ImageDescriber/Verbs/Export.cs @@ -3,6 +3,7 @@ namespace ktsu.ImageDescriber.Verbs; using System.Collections.Generic; +using System.Globalization; using System.IO; using System.Linq; using System.Text; @@ -63,19 +64,35 @@ private static void ExportJson(AbsoluteFilePath outputPath, Dictionary descriptions) + private static void ExportCsv(AbsoluteFilePath outputPath, Dictionary descriptions) => + File.WriteAllText(outputPath.WeakString, BuildCsv(descriptions.Values), Encoding.UTF8); + + internal static string BuildCsv(IEnumerable descriptions) { StringBuilder sb = new(); sb.AppendLine("Hash,SuggestedFileName,KnownPaths,Model,DescribedAt,FileSizeBytes,Description"); - foreach (ImageDescription desc in descriptions.Values) + foreach (ImageDescription desc in descriptions) { - string escapedDescription = $"\"{desc.Description.Replace("\"", "\"\"", StringComparison.Ordinal)}\""; string joinedPaths = string.Join("; ", desc.KnownPaths.Select(p => p.WeakString)); - string escapedPaths = $"\"{joinedPaths.Replace("\"", "\"\"", StringComparison.Ordinal)}\""; - sb.AppendLine($"{desc.Hash},{desc.SuggestedFileName},{escapedPaths},{desc.Model},{desc.DescribedAt:O},{desc.FileSizeBytes},{escapedDescription}"); + string[] fields = + [ + desc.Hash, + desc.SuggestedFileName.WeakString, + joinedPaths, + desc.Model.WeakString, + desc.DescribedAt.ToString("O", CultureInfo.InvariantCulture), + desc.FileSizeBytes.ToString(CultureInfo.InvariantCulture), + desc.Description, + ]; + + // Quote every field: file names may contain commas and quotes, and descriptions + // usually contain paragraph breaks, so no field is safe to write bare. + sb.AppendLine(string.Join(',', fields.Select(QuoteCsvField))); } - File.WriteAllText(outputPath.WeakString, sb.ToString(), Encoding.UTF8); + return sb.ToString(); } + + private static string QuoteCsvField(string value) => $"\"{value.Replace("\"", "\"\"", StringComparison.Ordinal)}\""; } diff --git a/ImageDescriber/Verbs/Import.cs b/ImageDescriber/Verbs/Import.cs index 9ad48e6..fa2dd8b 100644 --- a/ImageDescriber/Verbs/Import.cs +++ b/ImageDescriber/Verbs/Import.cs @@ -138,32 +138,31 @@ private static List ImportJson(AbsoluteFilePath inputPath) return JsonSerializer.Deserialize>(json, JsonOptions) ?? []; } - private static List ImportCsv(AbsoluteFilePath inputPath) + private static List ImportCsv(AbsoluteFilePath inputPath) => + ParseCsv(File.ReadAllText(inputPath.WeakString)); + + internal static List ParseCsv(string text) { List entries = []; - string[] lines = File.ReadAllLines(inputPath.WeakString); - if (lines.Length < 2) + // Parse the whole text rather than line by line, because a quoted field may span lines. + List<(int LineNumber, List Fields)> records = [.. ParseCsvRecords(text) + .Where(r => r.Fields.Count > 1 || !string.IsNullOrWhiteSpace(r.Fields[0]))]; + + if (records.Count < 2) { Console.WriteLine("CSV file is empty or has no data rows."); return entries; } // Skip header row - for (int i = 1; i < lines.Length; i++) + foreach ((int lineNumber, List fields) in records.Skip(1)) { - string line = lines[i]; - if (string.IsNullOrWhiteSpace(line)) - { - continue; - } - try { - List fields = ParseCsvLine(line); if (fields.Count < 7) { - Console.WriteLine($" Skipping line {i + 1}: not enough fields."); + Console.WriteLine($" Skipping line {lineNumber}: not enough fields."); continue; } @@ -191,89 +190,109 @@ private static List ImportCsv(AbsoluteFilePath inputPath) } catch (FormatException ex) { - Console.WriteLine($" Skipping line {i + 1}: {ex.Message}"); + Console.WriteLine($" Skipping line {lineNumber}: {ex.Message}"); } catch (OverflowException ex) { - Console.WriteLine($" Skipping line {i + 1}: {ex.Message}"); + Console.WriteLine($" Skipping line {lineNumber}: {ex.Message}"); } catch (ArgumentException ex) { - Console.WriteLine($" Skipping line {i + 1}: {ex.Message}"); + Console.WriteLine($" Skipping line {lineNumber}: {ex.Message}"); } } return entries; } - internal static List ParseCsvLine(string line) + internal static List ParseCsvLine(string line) => + ParseCsvRecords(line).Select(r => r.Fields).FirstOrDefault() ?? []; + + /// + /// Splits CSV text into records per RFC 4180: a quoted field may contain commas, doubled + /// quotes and line breaks. Each record carries the 1-based line it starts on. + /// + internal static List<(int LineNumber, List Fields)> ParseCsvRecords(string text) { + List<(int LineNumber, List Fields)> records = []; List fields = []; - int i = 0; + StringBuilder field = new(); + bool recordHasContent = false; + int line = 1; + int recordLine = 1; - while (i < line.Length) + int i = 0; + while (i < text.Length) { - if (line[i] == '"') - { - fields.Add(ParseQuotedField(line, ref i)); - } - else + char c = text[i++]; + switch (c) { - fields.Add(ParseUnquotedField(line, ref i)); + case '"': + i = ReadQuotedField(text, i, field, ref line); + recordHasContent = true; + break; + case ',': + fields.Add(field.ToString()); + field.Clear(); + recordHasContent = true; + break; + case '\r': + break; + case '\n': + EndRecord(); + line++; + recordLine = line; + break; + default: + field.Append(c); + recordHasContent = true; + break; } } - return fields; - } - - private static string ParseQuotedField(string line, ref int i) - { - i++; // skip opening quote - StringBuilder field = new(); + EndRecord(); + return records; - while (i < line.Length) + void EndRecord() { - if (line[i] == '"') + if (recordHasContent) { - if (i + 1 < line.Length && line[i + 1] == '"') - { - field.Append('"'); - i += 2; - } - else - { - i++; // closing quote - break; - } + fields.Add(field.ToString()); + records.Add((recordLine, fields)); + fields = []; } - else - { - field.Append(line[i]); - i++; - } - } - // Skip comma after quoted field - if (i < line.Length && line[i] == ',') - { - i++; + field.Clear(); + recordHasContent = false; } - - return field.ToString(); } - private static string ParseUnquotedField(string line, ref int i) + /// + /// Reads a quoted field's contents into , starting just after its + /// opening quote, and returns the index just after its closing quote. + /// + private static int ReadQuotedField(string text, int start, StringBuilder field, ref int line) { - int commaIndex = line.IndexOf(',', i); - if (commaIndex < 0) + int i = start; + while (i < text.Length) { - string result = line[i..]; - i = line.Length; - return result; + char c = text[i++]; + if (c != '"') + { + line += c == '\n' ? 1 : 0; + field.Append(c); + } + else if (i < text.Length && text[i] == '"') + { + field.Append('"'); + i++; + } + else + { + return i; + } } - string field = line[i..commaIndex]; - i = commaIndex + 1; - return field; + return i; } }