From 3a7657fa4f1b2974c48b2861011fc6a5fdd0a97d Mon Sep 17 00:00:00 2001 From: chsword Date: Wed, 16 Sep 2026 13:10:13 +0000 Subject: [PATCH 1/3] =?UTF-8?q?=E5=AF=BC=E5=85=A5=E5=89=8D=E7=9A=84?= =?UTF-8?q?=E5=88=97=E8=AF=8A=E6=96=AD=E4=B8=8E=E5=8F=AA=E8=AF=BB=E8=A1=A8?= =?UTF-8?q?=E5=A4=B4=EF=BC=8C=E5=91=BD=E4=BB=A4=E8=A1=8C=E5=B7=A5=E5=85=B7?= =?UTF-8?q?=E6=94=B9=E4=B8=BA=E9=80=90=E8=A1=8C=E8=AF=BB=E5=87=BA?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 标题差一个空格、多一个单位(「金额」与「金额(元)」),导入此前不报错,只是那一列悄悄全是默认值——OnCellError 只到单元格层面,管不到这里。新增 OnMissingColumn:读过表头之后、取第一行之前,每个对不上的标题上报一次,并给出表头上与之相近的标题(只按一方包含另一方判定,不多猜)。不设回调时行为与既有版本一致;在回调中抛出即可拒绝该文件。 新增 ReadHeader:只读到表头那一行为止,给出表名与各列标题,用于导入前核对列或据表头生成模型。 命令行工具随之改为逐行读出:表头单独读,数据走流式导入,类型推断改成边读边推断,写 JSON 也改为边读边写。二十万行五列:convert 13.5 秒 / 2412 MB → 2.6 秒 / 215 MB,--typed 12.9 秒 / 2428 MB → 4.1 秒 / 329 MB,generate-model 11.9 秒 / 2245 MB → 2.7 秒 / 204 MB,三者输出与此前逐字节相同。 顺带修掉一处:表头两侧有空白时,convert 写出的属性名保留空白而值永远为空——表头照原样取、数据却以去掉空白的标题为键,两边对不上。现在同出一处。 版本 2.12.0。 Generated with [Claude Code](https://claude.ai/code) via [Happy](https://happy.engineering) Co-Authored-By: Claude Opus 5 Co-Authored-By: Happy --- Chsword.Excel2Object.Cli/ConvertCommand.cs | 75 ++++--- .../GenerateModelCommand.cs | 8 +- Chsword.Excel2Object.Cli/SheetData.cs | 63 +++--- Chsword.Excel2Object.Cli/TypeInference.cs | 63 ++++-- .../ImportDiagnosticsTest.cs | 187 ++++++++++++++++++ .../Chsword.Excel2Object.csproj | 2 +- Chsword.Excel2Object/ExcelHelper.cs | 6 + Chsword.Excel2Object/ExcelImporter.cs | 82 ++++++-- .../Internal/ImportContext.cs | 24 +++ Chsword.Excel2Object/Internal/ImportRow.cs | 14 ++ .../Internal/XlsxRowReader.cs | 4 +- .../Options/ExcelColumnMissing.cs | 47 +++++ .../Options/ExcelImporterOptions.cs | 23 +++ .../Options/ExcelSheetHeader.cs | 19 ++ README.md | 7 + README_EN.md | 7 + docs/README.md | 1 + docs/versions/v2.12.0.md | 71 +++++++ 18 files changed, 614 insertions(+), 89 deletions(-) create mode 100644 Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs create mode 100644 Chsword.Excel2Object/Options/ExcelColumnMissing.cs create mode 100644 Chsword.Excel2Object/Options/ExcelSheetHeader.cs create mode 100644 docs/versions/v2.12.0.md diff --git a/Chsword.Excel2Object.Cli/ConvertCommand.cs b/Chsword.Excel2Object.Cli/ConvertCommand.cs index 304a852..e643db4 100644 --- a/Chsword.Excel2Object.Cli/ConvertCommand.cs +++ b/Chsword.Excel2Object.Cli/ConvertCommand.cs @@ -1,5 +1,6 @@ using System.Data; using System.Globalization; +using System.Text; using System.Text.Encodings.Web; using System.Text.Json; using System.Text.Json.Nodes; @@ -9,12 +10,6 @@ namespace Chsword.Excel2Object.Cli; /// excel2obj convert: Excel -> JSON or JSON -> Excel, decided per input by file extension. public static class ConvertCommand { - private static readonly JsonSerializerOptions JsonOptions = new() - { - WriteIndented = true, - Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping - }; - public static int Run(Arguments args, TextWriter output, TextWriter error) { if (args.Positional.Count == 0) throw new UsageException("convert needs at least one input file"); @@ -45,14 +40,13 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) if (SheetData.IsExcel(input)) { - var json = ExcelToJson(input, sheet, typed); if (destination == null) { - output.WriteLine(json); + output.WriteLine(ExcelToJson(input, sheet, typed)); } else { - File.WriteAllText(destination, json); + using (var file = File.Create(destination)) WriteJson(input, sheet, typed, file); error.WriteLine($"{input} -> {destination}"); } } @@ -71,48 +65,75 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) return Excel2ObjCli.Ok; } + /// 写到标准输出时才用得到:那里本就要把整段文本拿在手上。 public static string ExcelToJson(string path, string? sheet, bool typed) + { + using var buffer = new MemoryStream(); + WriteJson(path, sheet, typed, buffer); + return Encoding.UTF8.GetString(buffer.ToArray()); + } + + /// + /// 一行读出、一行写出,中途不把整份数据攒在内存里。--typed 要先知道每列是什么类型, + /// 故先过一遍推断,再过一遍写出——两遍各是一次顺序读。 + /// + public static void WriteJson(string path, string? sheet, bool typed, Stream destination) { var data = SheetData.Load(path, sheet); - var types = typed - ? data.Columns.ToDictionary(c => c, c => TypeInference.Infer(data.ColumnValues(c))) - : null; + var types = typed ? data.Infer() : null; - var array = new JsonArray(); - foreach (var row in data.Rows) + using var writer = new Utf8JsonWriter(destination, + new JsonWriterOptions {Indented = true, Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping}); + writer.WriteStartArray(); + foreach (var row in data.Rows()) { - var item = new JsonObject(); + writer.WriteStartObject(); foreach (var column in data.Columns) { - var text = row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : ""; - item[column] = types == null ? JsonValue.Create(text) : ToJsonValue(text, types[column]); + writer.WritePropertyName(column); + var text = SheetData.Text(row, column); + if (types == null) + writer.WriteStringValue(text); + else + WriteTypedValue(writer, text, types[column].Result); } - array.Add(item); + writer.WriteEndObject(); } - return array.ToJsonString(JsonOptions); + writer.WriteEndArray(); } - private static JsonNode? ToJsonValue(string text, InferredType type) + private static void WriteTypedValue(Utf8JsonWriter writer, string text, InferredType type) { - if (type != InferredType.String && string.IsNullOrWhiteSpace(text)) return null; + if (type != InferredType.String && string.IsNullOrWhiteSpace(text)) + { + writer.WriteNullValue(); + return; + } + switch (type) { case InferredType.Bool: TypeInference.TryParseBool(text, out var flag); - return JsonValue.Create(flag); + writer.WriteBooleanValue(flag); + break; case InferredType.Int: - return JsonValue.Create(int.Parse(text, CultureInfo.InvariantCulture)); + writer.WriteNumberValue(int.Parse(text, CultureInfo.InvariantCulture)); + break; case InferredType.Long: - return JsonValue.Create(long.Parse(text, CultureInfo.InvariantCulture)); + writer.WriteNumberValue(long.Parse(text, CultureInfo.InvariantCulture)); + break; case InferredType.Decimal: - return JsonValue.Create(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture)); + writer.WriteNumberValue(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture)); + break; case InferredType.DateTime: TypeInference.TryParseDateTime(text, out var date); - return JsonValue.Create(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture)); + writer.WriteStringValue(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture)); + break; default: - return JsonValue.Create(text); + writer.WriteStringValue(text); + break; } } diff --git a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs index 6ba684e..67cd8cc 100644 --- a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs +++ b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs @@ -44,12 +44,14 @@ public static string Generate(SheetData data, string className, string? ns) sb.Append("public class ").AppendLine(className); sb.AppendLine("{"); var used = new HashSet(StringComparer.Ordinal) {className}; + var inferences = data.Infer(); for (var i = 0; i < data.Columns.Count; i++) { var title = data.Columns[i]; - var values = data.ColumnValues(title).ToList(); - var type = TypeInference.Infer(values); - var nullable = values.Count == 0 || values.Any(string.IsNullOrWhiteSpace); + var inference = inferences[title]; + var type = inference.Result; + // 一行都没有,或出现过空值,该属性即为可空 + var nullable = !inference.Any || inference.HasBlank; var name = Unique(ToIdentifier(title, $"Column{i + 1}"), used); if (i > 0) sb.AppendLine(); diff --git a/Chsword.Excel2Object.Cli/SheetData.cs b/Chsword.Excel2Object.Cli/SheetData.cs index 66f56b6..3f5162e 100644 --- a/Chsword.Excel2Object.Cli/SheetData.cs +++ b/Chsword.Excel2Object.Cli/SheetData.cs @@ -1,46 +1,65 @@ -using NPOI.SS.UserModel; +using Chsword.Excel2Object.Options; namespace Chsword.Excel2Object.Cli; -/// A sheet as the importer sees it: the header titles in column order and one string per cell. +/// +/// A sheet as the importer sees it: the header titles in column order, and its rows on demand. +/// +/// +/// 行不再一次性读进内存: 每次遍历都逐行读出,命令行因而能处理远大于内存的 +/// 文件。需要过两遍数据的地方(--typed 先推断类型再写出)就遍历两次,各自的开销是一遍 +/// 顺序读。 +/// public sealed class SheetData { - public SheetData(string sheetTitle, List columns, List> rows) + private readonly string _path; + private readonly string? _sheetTitle; + + private SheetData(string path, string? sheetTitle, string title, List columns) { - SheetTitle = sheetTitle; + _path = path; + _sheetTitle = sheetTitle; + SheetTitle = title; Columns = columns; - Rows = rows; } public string SheetTitle { get; } + public List Columns { get; } - public List> Rows { get; } public static SheetData Load(string path, string? sheetTitle) { if (!File.Exists(path)) throw new FileNotFoundException($"input file not found: {path}", path); - var bytes = File.ReadAllBytes(path); - var rows = ExcelHelper.ExcelToObject>(bytes, sheetTitle).ToList(); - var (title, columns) = ReadHeader(bytes, sheetTitle); - return new SheetData(title, columns, rows); + + // 只读到表头那一行为止,后面有多少行数据都不影响这一步的开销 + using var input = File.OpenRead(path); + var header = ExcelHelper.ReadHeader(input, options => options.SheetTitle = sheetTitle); + return new SheetData(path, sheetTitle, header.SheetTitle ?? "", header.Columns.ToList()); } - /// The header row straight from the workbook, so an empty sheet still yields its columns. - private static (string title, List columns) ReadHeader(byte[] bytes, string? sheetTitle) + /// 逐行读出该表。每次遍历都重新读一遍文件。 + public IEnumerable> Rows() { - using var stream = new MemoryStream(bytes); - var workbook = WorkbookFactory.Create(stream); - var sheet = string.IsNullOrEmpty(sheetTitle) ? workbook.GetSheetAt(0) : workbook.GetSheet(sheetTitle); - if (sheet == null) throw new Excel2ObjectException($"The specified sheet:[{sheetTitle}] does not exist"); - var header = sheet.GetRow(sheet.FirstRowNum); - // untrimmed on purpose: the importer keys rows by the exact header text - var columns = header?.Cells.Select(cell => cell.ToString() ?? "").ToList() ?? new List(); - return (sheet.SheetName, columns); + using var input = File.OpenRead(_path); + foreach (var row in ExcelHelper.ExcelStreamToObject>(input, + options => options.SheetTitle = _sheetTitle)) + yield return row; + } + + /// 各列的类型推断,一遍读完。 + public Dictionary Infer() + { + var inferences = Columns.ToDictionary(c => c, _ => new TypeInference.Inference(), StringComparer.Ordinal); + foreach (var row in Rows()) + foreach (var column in Columns) + inferences[column].Observe(Text(row, column)); + + return inferences; } - public IEnumerable ColumnValues(string column) + public static string Text(IReadOnlyDictionary row, string column) { - return Rows.Select(row => row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : ""); + return row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : ""; } public static bool IsExcel(string path) diff --git a/Chsword.Excel2Object.Cli/TypeInference.cs b/Chsword.Excel2Object.Cli/TypeInference.cs index 04b6961..5effc0d 100644 --- a/Chsword.Excel2Object.Cli/TypeInference.cs +++ b/Chsword.Excel2Object.Cli/TypeInference.cs @@ -27,26 +27,57 @@ public static class TypeInference public static InferredType Infer(IEnumerable values) { - var seen = false; - var candidates = new HashSet + var inference = new Inference(); + foreach (var value in values) inference.Observe(value); + return inference.Result; + } + + /// + /// 一列的推断过程:一个值一个值地喂进来,不必先把整列攒在内存里。命令行要处理的文件可能有 + /// 几十万行,逐行读出便是为此。 + /// + public sealed class Inference + { + private readonly HashSet _candidates = new() {InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, InferredType.DateTime}; - foreach (var value in values) + + private bool _seen; + + /// 这一列上是否出现过空值——空值本身不参与类型判断,却决定该属性是否可空。 + public bool HasBlank { get; private set; } + + /// 是否有过任何一行。一行都没有的列按字符串处理,且算作可空。 + public bool Any { get; private set; } + + public InferredType Result { - if (string.IsNullOrWhiteSpace(value)) continue; - seen = true; - candidates.RemoveWhere(candidate => !Fits(candidate, value)); - if (candidates.Count == 0) break; + get + { + if (!_seen) return InferredType.String; + foreach (var candidate in new[] + { + InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, + InferredType.DateTime + }) + if (_candidates.Contains(candidate)) + return candidate; + + return InferredType.String; + } } - if (!seen) return InferredType.String; - foreach (var candidate in new[] - { - InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, - InferredType.DateTime - }) - if (candidates.Contains(candidate)) - return candidate; - return InferredType.String; + public void Observe(string value) + { + Any = true; + if (string.IsNullOrWhiteSpace(value)) + { + HasBlank = true; + return; + } + + _seen = true; + if (_candidates.Count > 0) _candidates.RemoveWhere(candidate => !Fits(candidate, value)); + } } public static bool Fits(InferredType type, string value) diff --git a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs new file mode 100644 index 0000000..9348ce9 --- /dev/null +++ b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs @@ -0,0 +1,187 @@ +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using Chsword.Excel2Object.Options; +using Microsoft.VisualStudio.TestTools.UnitTesting; +using NPOI.SS.UserModel; +using NPOI.XSSF.UserModel; + +namespace Chsword.Excel2Object.Tests; + +/// +/// 导入前的两件事:模型上的列在表头里有没有,以及表头本身是什么。 +/// +[TestClass] +public class ImportDiagnosticsTest +{ + public class Order + { + [ExcelTitle("订单号")] public string No { get; set; } = ""; + [ExcelTitle("金额")] public decimal Amount { get; set; } + [ExcelTitle("备注")] public string? Memo { get; set; } + } + + /// 表头由调用方写定,好把「对不上」这件事摆出来。 + private static byte[] Workbook(string[] titles, string[][] rows, string sheetTitle = "数据") + { + var workbook = new XSSFWorkbook(); + var sheet = workbook.CreateSheet(sheetTitle); + var header = sheet.CreateRow(0); + for (var i = 0; i < titles.Length; i++) header.CreateCell(i).SetCellValue(titles[i]); + for (var r = 0; r < rows.Length; r++) + { + var row = sheet.CreateRow(r + 1); + for (var i = 0; i < rows[r].Length; i++) row.CreateCell(i).SetCellValue(rows[r][i]); + } + + using var bytes = new MemoryStream(); + workbook.Write(bytes, true); + return bytes.ToArray(); + } + + private static byte[] Sample() + { + // 「金额」在表里叫「金额(元)」——导入不会报错,只是那一列悄悄全为空 + return Workbook(new[] {"订单号", "金额(元)", "备注"}, + new[] {new[] {"00123", "100.5", "线上"}, new[] {"00124", "200", ""}}); + } + + [TestMethod] + public void AColumnTheHeaderDoesNotHaveIsReported() + { + var missing = new List(); + var orders = new ExcelImporter() + .ExcelToObject(Sample(), options => options.OnMissingColumn = missing.Add).ToList(); + + // 沿用既有行为:那一列取默认值,导入照常 + Assert.AreEqual(2, orders.Count); + Assert.AreEqual(0m, orders[0].Amount); + Assert.AreEqual("00123", orders[0].No); + + Assert.AreEqual(1, missing.Count); + Assert.AreEqual("金额", missing[0].Title); + Assert.AreEqual(nameof(Order.Amount), missing[0].PropertyName); + Assert.AreEqual("数据", missing[0].SheetTitle); + CollectionAssert.AreEqual(new[] {"订单号", "金额(元)", "备注"}, missing[0].HeaderTitles.ToArray()); + // 相近的标题一并给出,否则「差在哪里」还得自己翻文件 + CollectionAssert.AreEqual(new[] {"金额(元)"}, missing[0].SimilarTitles.ToArray()); + StringAssert.Contains(missing[0].ToString(), "金额(元)"); + } + + [TestMethod] + public void NothingIsReportedWhenEveryColumnIsFound() + { + var missing = new List(); + var bytes = Workbook(new[] {"订单号", "金额", "备注"}, new[] {new[] {"00123", "100.5", "线上"}}); + var orders = new ExcelImporter() + .ExcelToObject(bytes, options => options.OnMissingColumn = missing.Add).ToList(); + + Assert.AreEqual(100.5m, orders[0].Amount); + Assert.AreEqual(0, missing.Count); + } + + [TestMethod] + public void WithoutTheCallbackTheImportIsUnchanged() + { + // 不设回调时一切照旧:这是既有版本的行为 + var orders = new ExcelImporter().ExcelToObject(Sample()).ToList(); + Assert.AreEqual(2, orders.Count); + Assert.AreEqual(0m, orders[0].Amount); + } + + [TestMethod] + public void ThrowingFromTheCallbackRefusesTheFile() + { + var e = Assert.ThrowsException(() => + new ExcelImporter().ExcelToObject(Sample(), + options => options.OnMissingColumn = m => throw new Excel2ObjectException(m.ToString())) + .ToList()); + + StringAssert.Contains(e.Message, "[金额]"); + } + + [TestMethod] + public void TheStreamingImportReportsItToo() + { + var missing = new List(); + using var input = new MemoryStream(Sample()); + var orders = new ExcelImporter() + .ExcelStreamToObject(input, options => options.OnMissingColumn = missing.Add).ToList(); + + Assert.AreEqual(2, orders.Count); + Assert.AreEqual(1, missing.Count); + Assert.AreEqual("金额", missing[0].Title); + CollectionAssert.AreEqual(new[] {"金额(元)"}, missing[0].SimilarTitles.ToArray()); + } + + [TestMethod] + public void TwoColumnsMissingAreReportedOnceEach() + { + var missing = new List(); + var bytes = Workbook(new[] {"订单号"}, new[] {new[] {"00123"}, new[] {"00124"}, new[] {"00125"}}); + new ExcelImporter().ExcelToObject(bytes, options => options.OnMissingColumn = missing.Add).ToList(); + + // 每个对不上的标题只上报一次,与行数无关 + CollectionAssert.AreEquivalent(new[] {"金额", "备注"}, missing.Select(m => m.Title).ToArray()); + Assert.AreEqual(0, missing[0].SimilarTitles.Count, "没有相近的就不要硬凑"); + } + + [TestMethod] + public void TheHeaderCanBeReadOnItsOwn() + { + foreach (var excelType in new[] {ExcelType.Xlsx, ExcelType.Xls}) + { + var bytes = new ExcelExporter().ObjectToExcelBytes( + new[] {new Order {No = "00123", Amount = 100.5m, Memo = "线上"}}, options => + { + options.ExcelType = excelType; + options.SheetTitle = "本月"; + }); + Assert.IsNotNull(bytes); + + using var input = new MemoryStream(bytes!); + var header = ExcelHelper.ReadHeader(input); + Assert.AreEqual("本月", header.SheetTitle, excelType.ToString()); + CollectionAssert.AreEqual(new[] {"订单号", "金额", "备注"}, header.Columns.ToArray(), excelType.ToString()); + } + } + + [TestMethod] + public void TheHeaderHonoursSheetTitleAndSkippedLines() + { + var bytes = Workbook(new[] {"订单号", "金额"}, new[] {new[] {"00123", "1"}}, "上月"); + var workbook = new XSSFWorkbook(new MemoryStream(bytes)); + var second = workbook.CreateSheet("本月"); + second.CreateRow(0).CreateCell(0).SetCellValue("导出说明"); + var header = second.CreateRow(1); + header.CreateCell(0).SetCellValue("城市"); + header.CreateCell(1).SetCellValue("数量"); + using var both = new MemoryStream(); + workbook.Write(both, true); + + using var input = new MemoryStream(both.ToArray()); + var read = ExcelHelper.ReadHeader(input, options => + { + options.SheetTitle = "本月"; + options.TitleSkipLine = 1; + }); + + Assert.AreEqual("本月", read.SheetTitle); + CollectionAssert.AreEqual(new[] {"城市", "数量"}, read.Columns.ToArray()); + } + + [TestMethod] + public void AnEmptySheetStillYieldsItsName() + { + var workbook = new XSSFWorkbook(); + workbook.CreateSheet("空表"); + using var bytes = new MemoryStream(); + workbook.Write(bytes, true); + + using var input = new MemoryStream(bytes.ToArray()); + var header = ExcelHelper.ReadHeader(input); + Assert.AreEqual("空表", header.SheetTitle); + Assert.AreEqual(0, header.Columns.Count); + } +} diff --git a/Chsword.Excel2Object/Chsword.Excel2Object.csproj b/Chsword.Excel2Object/Chsword.Excel2Object.csproj index 12ac32e..9b2dec1 100644 --- a/Chsword.Excel2Object/Chsword.Excel2Object.csproj +++ b/Chsword.Excel2Object/Chsword.Excel2Object.csproj @@ -15,7 +15,7 @@ true Chsword.Excel2Object Library Zou Jian - 2.11.0 + 2.12.0 Copyright ? 2014-2025 https://github.com/chsword/Excel2Object/ README.md diff --git a/Chsword.Excel2Object/ExcelHelper.cs b/Chsword.Excel2Object/ExcelHelper.cs index 074bbdb..12470ed 100644 --- a/Chsword.Excel2Object/ExcelHelper.cs +++ b/Chsword.Excel2Object/ExcelHelper.cs @@ -111,6 +111,12 @@ public static void ObjectToExcel(IEnumerable data, string path, return excelExporter.ObjectToExcelBytes(data, optionsAction); } + /// + public static ExcelSheetHeader ReadHeader(Stream input, Action? optionAction = null) + { + return new ExcelImporter().ReadHeader(input, optionAction); + } + /// public static IEnumerable ExcelStreamToObject(Stream input, Action? optionAction = null) diff --git a/Chsword.Excel2Object/ExcelImporter.cs b/Chsword.Excel2Object/ExcelImporter.cs index e78bc9b..7015719 100644 --- a/Chsword.Excel2Object/ExcelImporter.cs +++ b/Chsword.Excel2Object/ExcelImporter.cs @@ -47,8 +47,8 @@ public IEnumerable ExcelToObject(byte[] bytes, var options = new ExcelImporterOptions(); optionAction?.Invoke(options); var context = new ImportContext(options); - var rows = GetDataRows(bytes, options, context); - return ToModels(rows, context); + var source = GetDataRows(bytes, options, context); + return ToModels(source == null ? null : AtHeader(source, options), context); } public IEnumerable ExcelToObject(byte[] bytes, string? sheetTitle) @@ -95,10 +95,42 @@ public IEnumerable ExcelStreamToObject(Stream input, var options = new ExcelImporterOptions(); optionAction?.Invoke(options); var context = new ImportContext(options); - var rows = XlsxRowReader.Rows(input, options, context).GetEnumerator(); - rows.MoveNext(); - for (var i = 0; i < options.TitleSkipLine; i++) rows.MoveNext(); - return ToModels(rows, context); + return ToModels(AtHeader(XlsxRowReader.Rows(input, options, context), options), context); + } + + /// + /// 只读出表头:表名与各列的标题,按列的先后。用于在导入之前核对列,或据表头生成模型。 + /// + /// + /// 只读到表头那一行为止,后面有多少行数据都不影响其开销。.xls 仍须整份读入——该格式 + /// 的数据并非顺序存放。传入的流由本方法读取,返回前即已读完。 + /// + /// + /// + /// using var file = File.OpenRead("orders.xlsx"); + /// var header = ExcelHelper.ReadHeader(file); + /// Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}"); + /// + /// + public ExcelSheetHeader ReadHeader(Stream input, Action? optionAction = null) + { + var options = new ExcelImporterOptions(); + optionAction?.Invoke(options); + var context = new ImportContext(options); + + var source = LooksLikeXlsx(input) + ? XlsxRowReader.Rows(input, options, context) + : GetDataRows(ReadAll(input), options, context); + if (source == null) return new ExcelSheetHeader(null, new List()); + + using var rows = AtHeader(source, options); + var titleRow = rows.Current; + var columns = new List(); + if (titleRow != null) + foreach (var cell in titleRow.Cells) + columns.Add(TextOf(cell.Value) ?? string.Empty); + + return new ExcelSheetHeader(source.Title, columns); } /// @@ -178,7 +210,7 @@ private static IEnumerable InternalExcelToObject(IEnumerator(result); + var dictColumns = BuildColumnMappings(result, context); while (result.MoveNext()) { @@ -195,21 +227,29 @@ private static IEnumerable InternalExcelToObject(IEnumerator> BuildColumnMappings( - IEnumerator result) + IEnumerator result, ImportContext context) where TModel : class, new() { var dict = ExcelUtil.GetPropertiesAttributesDict(); var dictColumns = new Dictionary>(); var titleRow = result.Current; + if (titleRow == null) return dictColumns; - if (titleRow != null) - foreach (var cell in titleRow.Cells) - { - var title = TextOf(cell.Value); - var prop = dict.FirstOrDefault(c => title == c.Value.Title); - if (prop.Key != null && !dictColumns.ContainsKey(cell.Key)) - dictColumns.Add(cell.Key, prop); - } + var headerTitles = new List(); + foreach (var cell in titleRow.Cells) + { + var title = TextOf(cell.Value) ?? string.Empty; + headerTitles.Add(title); + var prop = dict.FirstOrDefault(c => title == c.Value.Title); + if (prop.Key != null && !dictColumns.ContainsKey(cell.Key)) + dictColumns.Add(cell.Key, prop); + } + + // 模型上写着、表头里却没有的标题:那一列不会被填上,整列都是默认值,此处上报 + var mapped = new HashSet(dictColumns.Values.Select(c => c.Value.Title), StringComparer.Ordinal); + foreach (var pair in dict) + if (!mapped.Contains(pair.Value.Title)) + context.ReportMissingColumn(pair.Value.Title, pair.Key.Name, titleRow.SheetTitle, headerTitles); return dictColumns; } @@ -395,7 +435,7 @@ private static void PopulateModelFromRow(TModel model, IImportRow row, return date.Value.ToString(pattern, CultureInfo.InvariantCulture); } - private static IEnumerator? GetDataRows(byte[]? bytes, ExcelImporterOptions options, + private static SheetSource? GetDataRows(byte[]? bytes, ExcelImporterOptions options, ImportContext context) { if (bytes == null || bytes.Length == 0) @@ -423,7 +463,13 @@ private static void PopulateModelFromRow(TModel model, IImportRow row, throw new Excel2ObjectException($"The specified sheet:[{options.SheetTitle}] does not exist"); } - var rows = NpoiRows(sheet, context).GetEnumerator(); + return new SheetSource(sheet.SheetName, NpoiRows(sheet, context)); + } + + /// 取到停在表头那一行的枚举器:表头之上还可以有若干行说明文字。 + private static IEnumerator AtHeader(SheetSource source, ExcelImporterOptions options) + { + var rows = source.Rows.GetEnumerator(); rows.MoveNext(); for (var i = 0; i < options.TitleSkipLine; i++) rows.MoveNext(); return rows; diff --git a/Chsword.Excel2Object/Internal/ImportContext.cs b/Chsword.Excel2Object/Internal/ImportContext.cs index 602c9a8..58d711e 100644 --- a/Chsword.Excel2Object/Internal/ImportContext.cs +++ b/Chsword.Excel2Object/Internal/ImportContext.cs @@ -61,6 +61,30 @@ public void Report(IRow? row, int columnIndex, Exception exception) Report(row?.Sheet?.SheetName, row?.RowNum ?? -1, columnIndex, exception); } + /// + /// 上报模型上的某个标题在表头中找不到。未设置回调时不作任何输出;回调抛出的异常照旧向外 + /// 传播,调用方据此即可拒绝这样的文件。 + /// + public void ReportMissingColumn(string title, string propertyName, string? sheetTitle, + IReadOnlyList headerTitles) + { + if (_options.OnMissingColumn == null) return; + + // 相近只按「一方包含另一方」判定:「金额」与「金额(元)」够用,再多的猜测不如不猜 + var similar = new List(); + foreach (var header in headerTitles) + { + var one = header.Trim(); + var other = title.Trim(); + if (one.Length == 0 || other.Length == 0 || one == other) continue; + if (one.IndexOf(other, StringComparison.OrdinalIgnoreCase) >= 0 || + other.IndexOf(one, StringComparison.OrdinalIgnoreCase) >= 0) + similar.Add(header); + } + + _options.OnMissingColumn(new ExcelColumnMissing(title, propertyName, sheetTitle, headerTitles, similar)); + } + /// 位置由调用方给出:流式读取没有单元格对象,只有行列号。 public void Report(string? sheetTitle, int rowIndex, int columnIndex, Exception exception) { diff --git a/Chsword.Excel2Object/Internal/ImportRow.cs b/Chsword.Excel2Object/Internal/ImportRow.cs index 2690b47..9feb71a 100644 --- a/Chsword.Excel2Object/Internal/ImportRow.cs +++ b/Chsword.Excel2Object/Internal/ImportRow.cs @@ -21,6 +21,20 @@ internal interface IImportRow CellData Cell(int columnIndex); } +/// 一张工作表的来源:它的名字,以及各行。 +internal sealed class SheetSource +{ + public SheetSource(string? title, IEnumerable rows) + { + Title = title; + Rows = rows; + } + + public string? Title { get; } + + public IEnumerable Rows { get; } +} + /// 整份读入内存时的一行,取值经由 NPOI 的单元格。 internal sealed class NpoiRow : IImportRow { diff --git a/Chsword.Excel2Object/Internal/XlsxRowReader.cs b/Chsword.Excel2Object/Internal/XlsxRowReader.cs index 7f0a677..bab69bc 100644 --- a/Chsword.Excel2Object/Internal/XlsxRowReader.cs +++ b/Chsword.Excel2Object/Internal/XlsxRowReader.cs @@ -26,7 +26,7 @@ internal static class XlsxRowReader private const string RelationshipNamespace = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"; - public static IEnumerable Rows(Stream input, ExcelImporterOptions options, ImportContext context) + public static SheetSource Rows(Stream input, ExcelImporterOptions options, ImportContext context) { OPCPackage package; try @@ -44,7 +44,7 @@ public static IEnumerable Rows(Stream input, ExcelImporterOptions op var reader = new XSSFReader(package); var workbook = ReadWorkbook(reader); var sheet = Locate(workbook, options.SheetTitle); - return Iterate(package, reader, sheet, workbook.Date1904, context); + return new SheetSource(sheet.Title, Iterate(package, reader, sheet, workbook.Date1904, context)); } catch { diff --git a/Chsword.Excel2Object/Options/ExcelColumnMissing.cs b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs new file mode 100644 index 0000000..9a261df --- /dev/null +++ b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs @@ -0,0 +1,47 @@ +namespace Chsword.Excel2Object.Options; + +/// +/// 模型上写着的某个列标题,在表头中找不到。该属性因而不会被填上,整列都是默认值。 +/// +/// +/// 这是导入类问题中最常见的一种:标题差了一个空格、多了个单位(「金额」与「金额(元)」), +/// 导入不会报错,只是那一列悄悄全为空。 即为 +/// 此而设。 +/// +public class ExcelColumnMissing +{ + public ExcelColumnMissing(string title, string propertyName, string? sheetTitle, + IReadOnlyList headerTitles, IReadOnlyList similarTitles) + { + Title = title; + PropertyName = propertyName; + SheetTitle = sheetTitle; + HeaderTitles = headerTitles; + SimilarTitles = similarTitles; + } + + /// 模型上写着的标题。 + public string Title { get; } + + /// 要它的那个属性。 + public string PropertyName { get; } + + public string? SheetTitle { get; } + + /// 表头上实际有的标题,按列的先后。 + public IReadOnlyList HeaderTitles { get; } + + /// + /// 表头上与之相近的标题:一方包含另一方即算(「金额」与「金额(元)」),不作更多猜测。 + /// + public IReadOnlyList SimilarTitles { get; } + + public override string ToString() + { + var where = SheetTitle == null ? "" : $"工作表 [{SheetTitle}] "; + var similar = SimilarTitles.Count == 0 + ? "" + : $",表头上与之相近的是 [{string.Join("]、[", SimilarTitles)}]"; + return $"{where}的表头中没有 [{Title}] 这一列(属性 {PropertyName} 因而不会被填上){similar}。"; + } +} diff --git a/Chsword.Excel2Object/Options/ExcelImporterOptions.cs b/Chsword.Excel2Object/Options/ExcelImporterOptions.cs index 6ed3051..323d2f3 100644 --- a/Chsword.Excel2Object/Options/ExcelImporterOptions.cs +++ b/Chsword.Excel2Object/Options/ExcelImporterOptions.cs @@ -27,4 +27,27 @@ public class ExcelImporterOptions /// options.OnCellError = e => throw e.Exception;。 /// public Action? OnCellError { get; set; } + + /// + /// 模型上写着的某个列标题在表头中找不到时的回调,用于得知哪一列没有对上。默认不设回调, + /// 此时该属性保持其默认值,导入照常进行——与既有版本一致。 + /// + /// + /// + /// 标题差一个空格、多一个单位(「金额」与「金额(元)」),导入并不会报错,只是那一列 + /// 悄悄全为空。这是导入类问题中最常见的一种, 只覆盖到单元格 + /// 层面,对此无能为力。 + /// + /// 回调在读过表头之后、取第一行数据之前调用,每个对不上的标题调用一次。 + /// + /// + /// // 只是记下来 + /// options.OnMissingColumn = missing => logger.Warn(missing.ToString()); + /// + /// // 或者干脆不接受这样的文件 + /// options.OnMissingColumn = missing => throw new Excel2ObjectException(missing.ToString()); + /// + /// + /// + public Action? OnMissingColumn { get; set; } } diff --git a/Chsword.Excel2Object/Options/ExcelSheetHeader.cs b/Chsword.Excel2Object/Options/ExcelSheetHeader.cs new file mode 100644 index 0000000..78fb1f4 --- /dev/null +++ b/Chsword.Excel2Object/Options/ExcelSheetHeader.cs @@ -0,0 +1,19 @@ +namespace Chsword.Excel2Object.Options; + +/// +/// 一张工作表的表头:表名与各列的标题,按列的先后。 +/// +public class ExcelSheetHeader +{ + public ExcelSheetHeader(string? sheetTitle, IReadOnlyList columns) + { + SheetTitle = sheetTitle; + Columns = columns; + } + + /// 实际读的那张表的名字:未指定表名时即第一张表。 + public string? SheetTitle { get; } + + /// 表头上的各列标题,按列的先后;表中没有任何行时为空。 + public IReadOnlyList Columns { get; } +} diff --git a/README.md b/README.md index 8782876..61075b6 100644 --- a/README.md +++ b/README.md @@ -68,6 +68,13 @@ excel2obj generate-model orders.xlsx --class Order # 由表头生成 ### 发布说明 +* **2026.09.16** - v2.12.0 +- [x] ✨ **新增:** 模型上的列在表头中找不到时可以得知:`options.OnMissingColumn = m => logger.Warn(m.ToString())`。标题差一个空格、多一个单位(「金额」与「金额(元)」),导入此前不会报错,只是那一列悄悄全是默认值。回调在读过表头之后、取第一行之前调用,每个对不上的标题一次,并给出表头上与之相近的标题;在回调中抛出即可拒绝这样的文件。默认不设回调时行为与既有版本一致 - 查看 [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md) +- [x] ✨ **新增:** 只读出表头:`ExcelHelper.ReadHeader(stream)` 给出表名与各列标题,只读到表头那一行为止,用于导入前核对列或据表头生成模型 +- [x] 🔧 命令行工具改为逐行读出:二十万行五列的文件,`convert` 由 13.5 秒 / 2412 MB 降到 2.6 秒 / 215 MB,`generate-model` 由 11.9 秒 / 2245 MB 降到 2.7 秒 / 204 MB,输出与此前逐字节相同 +- [x] 🐛 修复命令行工具下表头带空白时取不到值:写出的属性名保留了空白,而值永远为空——表头与数据两边的标题此前不是同一个入口取的 + + * **2026.09.16** - v2.11.0 - [x] ✨ **新增:** 流式导入:`ExcelHelper.ExcelStreamToObject(stream)` 逐行读出 `.xlsx` 的一张工作表,不把整个工作簿建进内存。实测二十万行五列:整份读入峰值 1265 MB / 8.4 秒,流式导入 194 MB / 2.9 秒,两者读出的数据逐字段一致。取到的序列是惰性的,`Take` 一类的操作真的能少读。三处不同:公式格读的是文件中存着的上一次计算结果而非当场求值(本库导出的文件里公式没有这个结果,读作空白,与其他空白格一样)、只有 `.xlsx` 能逐行读出(`.xls` 照旧整份读入)、工作表在一开始就定位 - 查看 [docs/versions/v2.11.0.md](docs/versions/v2.11.0.md) - [x] 🔧 导入的类型转换不再依赖 NPOI 的单元格:整份读入与逐行读出共用同一套转换,两条路的行为不会各自漂移。字典形式的导入随之改为惰性给出;标题行中同名的标题以最左一列为准,非文本的标题不再中断导入 diff --git a/README_EN.md b/README_EN.md index dfc65af..2ce9c4f 100644 --- a/README_EN.md +++ b/README_EN.md @@ -66,6 +66,13 @@ See [Chsword.Excel2Object.Cli/README.md](Chsword.Excel2Object.Cli/README.md). ### Release Notes +* **2026.09.16** - v2.12.0 +- [x] ✨ **NEW:** Learn when a column your model declares is not in the header: `options.OnMissingColumn = m => logger.Warn(m.ToString())`. A title off by a space or carrying a unit ("Amount" against "Amount (USD)") never failed the import before - that property was simply left at its default for every row. The callback runs after the header is read and before the first row, once per title that did not match, and names the header titles closest to it; throw from it to refuse the file. With no callback set the behaviour is unchanged - See [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md) +- [x] ✨ **NEW:** Read just the header: `ExcelHelper.ReadHeader(stream)` returns the sheet name and its column titles, reading no further than the header row - for checking columns before an import, or generating a model from them +- [x] 🔧 The command-line tool now reads row by row: on 200,000 rows of five columns, `convert` went from 13.5 s / 2412 MB to 2.6 s / 215 MB and `generate-model` from 11.9 s / 2245 MB to 2.7 s / 204 MB, byte for byte the same output +- [x] 🐛 Fixed the command-line tool losing values when a header title carries surrounding whitespace: the property it wrote kept the whitespace while its value was always empty, because the header and the rows did not go through the same reader + + * **2026.09.16** - v2.11.0 - [x] ✨ **NEW:** Streaming import: `ExcelHelper.ExcelStreamToObject(stream)` reads one sheet of an `.xlsx` row by row instead of building the whole workbook in memory. Measured on 200,000 rows of five columns: 1265 MB / 8.4 s read whole against 194 MB / 2.9 s streamed, field for field the same data. The sequence is lazy, so `Take` and friends really do read less. Three differences: a formula cell reads the result stored in the file rather than being evaluated (files this library writes carry no such result, so those cells read as blank, like any other blank cell), only `.xlsx` can be read row by row (`.xls` is still read whole), and the sheet is located up front - See [docs/versions/v2.11.0.md](docs/versions/v2.11.0.md) - [x] 🔧 Import conversion no longer depends on NPOI's cell objects: reading whole and reading row by row share one conversion, so the two cannot drift apart. The dictionary form of the import became lazy as a result; duplicate header titles now resolve to the leftmost column, and a non-text header no longer aborts the import diff --git a/docs/README.md b/docs/README.md index 0ed0cf1..48125ff 100644 --- a/docs/README.md +++ b/docs/README.md @@ -6,6 +6,7 @@ | 版本 | 发布日期 | 主要内容 | | --- | --- | --- | +| [v2.12.0](versions/v2.12.0.md) | 2026-09-16 | 导入前的列诊断与只读表头;命令行工具逐行读出 | | [v2.11.0](versions/v2.11.0.md) | 2026-09-16 | 流式导入:逐行读出,不把整个工作簿建进内存 | | [v2.10.0](versions/v2.10.0.md) | 2026-09-16 | 流式导出:边取边写,内存不随行数增长 | | [v2.9.1](versions/v2.9.1.md) | 2026-09-15 | 合并单元格:按列合并连续相同值、指定区域;空引用检查改为强制 | diff --git a/docs/versions/v2.12.0.md b/docs/versions/v2.12.0.md new file mode 100644 index 0000000..c4ff14f --- /dev/null +++ b/docs/versions/v2.12.0.md @@ -0,0 +1,71 @@ +# v2.12.0 + +发布日期:2026-09-16 + +## 概述 + +本版本补上导入前的两件事:模型上写着的列在表头里究竟有没有(`OnMissingColumn`),以及表头本身是什么(`ReadHeader`)。命令行工具也随之改为逐行读出,二十万行的文件转换由 2.4 GB 降到 215 MB。 + +## 变更明细 + +### 1. 新增:模型上的列在表头中找不到时可以得知 + +标题差一个空格、多一个单位——「金额」与「金额(元)」——导入并不会报错,只是那一列悄悄全是默认值。这是导入类问题中最常见的一种,而 `OnCellError` 只覆盖到单元格层面,对此无能为力。 + +```csharp +var orders = new ExcelImporter().ExcelToObject(bytes, options => + options.OnMissingColumn = missing => logger.Warn(missing.ToString())); + +// 工作表 [数据] 的表头中没有 [金额] 这一列(属性 Amount 因而不会被填上), +// 表头上与之相近的是 [金额(元)]。 +``` + +回调在读过表头之后、取第一行数据之前调用,每个对不上的标题调用一次,与行数无关。`ExcelColumnMissing` 给出模型上写的标题、要它的属性名、表名、表头上实际有的全部标题,以及其中与之相近的几个——相近只按「一方包含另一方」判定,不作更多猜测。 + +默认不设回调,此时行为与既有版本完全一致:该属性保持默认值,导入照常。若希望干脆不接受这样的文件,在回调中抛出即可: + +```csharp +options.OnMissingColumn = missing => throw new Excel2ObjectException(missing.ToString()); +``` + +流式导入同样适用。 + +### 2. 新增:只读出表头 + +```csharp +using var file = File.OpenRead("orders.xlsx"); +var header = ExcelHelper.ReadHeader(file); +Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}"); +``` + +只读到表头那一行为止,后面有多少行数据都不影响这一步的开销。用于在导入之前核对列、按表头生成模型,或让使用者自行把表里的列对到模型上。`SheetTitle` 是实际读的那张表的名字——未指定表名时即第一张表;`SheetTitle` 与 `TitleSkipLine` 两个选项照常生效。`.xls` 仍须整份读入,该格式的数据并非顺序存放。 + +### 3. 改进:命令行工具逐行读出 + +`excel2obj` 此前每条命令都把整个工作簿读进内存,且表头还要再打开一次文件才能读到。现改为:表头由 `ReadHeader` 单独读出,数据走流式导入;类型推断(`--typed` 与 `generate-model`)也改成边读边推断,不再把整列的值攒起来。 + +二十万行五列的文件,本机实测: + +| 命令 | 此前 | 现在 | +| --- | --- | --- | +| `convert` | 13.5 秒 / 2412 MB | 2.6 秒 / 215 MB | +| `convert --typed` | 12.9 秒 / 2428 MB | 4.1 秒 / 329 MB | +| `generate-model` | 11.9 秒 / 2245 MB | 2.7 秒 / 204 MB | + +三条命令的输出与此前逐字节相同。`--typed` 要先知道每列是什么类型,故读两遍文件:一遍推断,一遍写出;写 JSON 也改为边读边写,不再先在内存里搭出整棵 JSON 树。 + +### 4. 修复:命令行工具下表头带空白时取不到值 + +表头写成  订单号 (两侧有空格)时,`excel2obj convert` 会写出一个名为 " 订单号 " 的属性,而它的值永远是空的——导入给出的行以去掉空白的标题为键,而表头是照原样取的,两边对不上。现在两处都经由同一个入口取得,不再有这个出入: + +```json +// 此前 +[ { " 订单号 ": "", "金额": "100.5" } ] +// 现在 +[ { "订单号": "00123", "金额": "100.5" } ] +``` + +## 注意事项 + +- `OnMissingColumn` 只在按模型导入时有意义:导入为 `Dictionary` 时列由表头决定,无所谓对不对得上。 +- 表头中重复的标题以最左边那一列为准(v2.11.0 起),因此「某个标题存在但对到了另一列」不属于本次上报的范围。 From 7f1ad2330afbcb08c29703d1fb0b116852e81774 Mon Sep 17 00:00:00 2001 From: chsword Date: Wed, 16 Sep 2026 13:21:49 +0000 Subject: [PATCH 2/3] =?UTF-8?q?=E6=8C=89=20Copilot=20=E7=9A=84=E5=A4=8D?= =?UTF-8?q?=E6=9F=A5=E6=84=8F=E8=A7=81=E4=BF=AE=E6=AD=A3=E4=BA=94=E5=A4=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 写出文件此前是先清空目标再边读边写:--output 指向输入本身时,源文件在读到之前就已被毁掉,中途失败也会留下半个文件。改为先写同目录下的临时文件,成功之后再就位。 CLI 改流式之后公式格取的是文件里存着的结果,而非当场求值——本库导出的文件没有存下结果,那一列会读作空白,与改动前不同。补 --whole:整份读入、公式当场求值,并在 README 与版本文档中写明默认行为。原先「输出逐字节相同」的说法改为「不含公式的文件逐字节相同」。 表里一行都没有时表头也没有,此前直接返回、模型上的每个标题都不上报;现同样上报。ReadHeader 遇到根本不是工作簿的输入改为抛出——此前静默给出空表头,CLI 因而把一份坏文件写成了 [],这是这次改动引入的回退,由新加的用例抓到。 另有 Inference.Result 每次取值都新建候选数组(逐格取,二十万行即上百万次),改为静态;Run 拆出目标路径的判定;测试里不再用 ! 压制可空警告。 Generated with [Claude Code](https://claude.ai/code) via [Happy](https://happy.engineering) Co-Authored-By: Claude Opus 5 Co-Authored-By: Happy --- Chsword.Excel2Object.Cli/ConvertCommand.cs | 67 ++++++++++++++----- .../GenerateModelCommand.cs | 2 +- Chsword.Excel2Object.Cli/README.md | 13 +++- Chsword.Excel2Object.Cli/SheetData.cs | 21 ++++-- Chsword.Excel2Object.Cli/TypeInference.cs | 12 ++-- Chsword.Excel2Object.Tests/CliTest.cs | 59 ++++++++++++++++ .../ImportDiagnosticsTest.cs | 35 +++++++--- Chsword.Excel2Object/ExcelImporter.cs | 44 +++++++----- README.md | 2 +- README_EN.md | 2 +- docs/versions/v2.12.0.md | 13 +++- 11 files changed, 210 insertions(+), 60 deletions(-) diff --git a/Chsword.Excel2Object.Cli/ConvertCommand.cs b/Chsword.Excel2Object.Cli/ConvertCommand.cs index e643db4..1923566 100644 --- a/Chsword.Excel2Object.Cli/ConvertCommand.cs +++ b/Chsword.Excel2Object.Cli/ConvertCommand.cs @@ -25,28 +25,22 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) Directory.CreateDirectory(target); } + var whole = args.Has("whole"); + foreach (var input in inputs) { if (!File.Exists(input)) throw new FileNotFoundException($"input file not found: {input}", input); - string? destination; - if (inputs.Count > 1) - destination = Path.Combine(target!, Path.GetFileNameWithoutExtension(input) + - (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx")); - else if (target != null && Directory.Exists(target)) - destination = Path.Combine(target, Path.GetFileNameWithoutExtension(input) + - (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx")); - else - destination = target; + var destination = Destination(input, target, inputs.Count > 1, xls); if (SheetData.IsExcel(input)) { if (destination == null) { - output.WriteLine(ExcelToJson(input, sheet, typed)); + output.WriteLine(ExcelToJson(input, sheet, typed, whole)); } else { - using (var file = File.Create(destination)) WriteJson(input, sheet, typed, file); + Replace(destination, file => WriteJson(input, sheet, typed, whole, file)); error.WriteLine($"{input} -> {destination}"); } } @@ -57,7 +51,8 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) var excelType = xls || destination.EndsWith(".xls", StringComparison.OrdinalIgnoreCase) ? ExcelType.Xls : ExcelType.Xlsx; - File.WriteAllBytes(destination, JsonToExcel(input, sheet, excelType)); + var bytes = JsonToExcel(input, sheet, excelType); + Replace(destination, file => file.Write(bytes, 0, bytes.Length)); error.WriteLine($"{input} -> {destination}"); } } @@ -65,11 +60,42 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) return Excel2ObjCli.Ok; } + /// 该输入写到哪里去:多个输入时按目录派生文件名,单个输入时即 --output 本身。 + private static string? Destination(string input, string? target, bool severalInputs, bool xls) + { + var name = Path.GetFileNameWithoutExtension(input) + + (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx"); + if (severalInputs) return Path.Combine(target!, name); + if (target != null && Directory.Exists(target)) return Path.Combine(target, name); + return target; + } + + /// + /// 先写到同目录下的临时文件,成功之后再就位。读的是一行一行来的,若直接往目标文件写: + /// 目标与输入是同一个文件时,源在读到之前就已被清空;中途失败也会留下半个文件。 + /// + private static void Replace(string destination, Action write) + { + var temporary = destination + ".tmp" + Path.GetRandomFileName(); + try + { + using (var file = File.Create(temporary)) write(file); + + if (File.Exists(destination)) File.Delete(destination); + File.Move(temporary, destination); + } + catch + { + if (File.Exists(temporary)) File.Delete(temporary); + throw; + } + } + /// 写到标准输出时才用得到:那里本就要把整段文本拿在手上。 - public static string ExcelToJson(string path, string? sheet, bool typed) + public static string ExcelToJson(string path, string? sheet, bool typed, bool whole = false) { using var buffer = new MemoryStream(); - WriteJson(path, sheet, typed, buffer); + WriteJson(path, sheet, typed, whole, buffer); return Encoding.UTF8.GetString(buffer.ToArray()); } @@ -77,10 +103,15 @@ public static string ExcelToJson(string path, string? sheet, bool typed) /// 一行读出、一行写出,中途不把整份数据攒在内存里。--typed 要先知道每列是什么类型, /// 故先过一遍推断,再过一遍写出——两遍各是一次顺序读。 /// - public static void WriteJson(string path, string? sheet, bool typed, Stream destination) + /// + /// 整份读入工作簿,公式当场求值。逐行读出取的是文件里存着的上一次计算结果,没有存下结果的 + /// 公式(本库导出的文件即如此)因而读作空白;确需求值时用这条路,代价是内存随文件增长。 + /// + public static void WriteJson(string path, string? sheet, bool typed, bool whole, Stream destination) { - var data = SheetData.Load(path, sheet); - var types = typed ? data.Infer() : null; + var data = SheetData.Load(path, sheet, whole); + // 每列的类型在此定下,不在写每一格时反复去问 + var types = typed ? data.Infer().ToDictionary(c => c.Key, c => c.Value.Result, StringComparer.Ordinal) : null; using var writer = new Utf8JsonWriter(destination, new JsonWriterOptions {Indented = true, Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping}); @@ -95,7 +126,7 @@ public static void WriteJson(string path, string? sheet, bool typed, Stream dest if (types == null) writer.WriteStringValue(text); else - WriteTypedValue(writer, text, types[column].Result); + WriteTypedValue(writer, text, types[column]); } writer.WriteEndObject(); diff --git a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs index 67cd8cc..4db8e81 100644 --- a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs +++ b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs @@ -11,7 +11,7 @@ public static int Run(Arguments args, TextWriter output, TextWriter error) if (args.Positional.Count != 1) throw new UsageException("generate-model needs exactly one input file"); var input = args.Positional[0]; var sheet = args.Get("sheet"); - var data = SheetData.Load(input, sheet); + var data = SheetData.Load(input, sheet, args.Has("whole")); var className = args.Get("class") ?? ToIdentifier(data.SheetTitle, "Model"); var code = Generate(data, className, args.Get("namespace")); diff --git a/Chsword.Excel2Object.Cli/README.md b/Chsword.Excel2Object.Cli/README.md index fcd1094..90bd065 100644 --- a/Chsword.Excel2Object.Cli/README.md +++ b/Chsword.Excel2Object.Cli/README.md @@ -14,11 +14,18 @@ excel2obj convert orders.xlsx # prints a JSON array to std excel2obj convert orders.xlsx --output orders.json # writes the file excel2obj convert orders.xlsx --sheet Orders --typed excel2obj convert a.xlsx b.xls --output ./json/ # several inputs need an output directory +excel2obj convert orders.xlsx --whole # read the workbook whole and evaluate formulas ``` -Each row becomes an object keyed by the header row. Values are the cell text (formulas are evaluated); with -`--typed`, a column whose non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as -JSON numbers, booleans or `yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`. +Each row becomes an object keyed by the header row. Values are the cell text; with `--typed`, a column whose +non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as JSON numbers, booleans or +`yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`. + +The sheet is read row by row, so a file far larger than memory converts fine. One consequence: a **formula cell +reads the result stored in the file** rather than being evaluated. Excel writes that result when it saves, so +files saved by Excel read as before; a file written by this library carries formulas with no stored result, and +those cells read as blank. Pass `--whole` to read the workbook whole and evaluate formulas instead - the memory +it needs then grows with the file. `--whole` works with `generate-model` as well. ## JSON -> Excel diff --git a/Chsword.Excel2Object.Cli/SheetData.cs b/Chsword.Excel2Object.Cli/SheetData.cs index 3f5162e..4c70c3b 100644 --- a/Chsword.Excel2Object.Cli/SheetData.cs +++ b/Chsword.Excel2Object.Cli/SheetData.cs @@ -14,11 +14,13 @@ public sealed class SheetData { private readonly string _path; private readonly string? _sheetTitle; + private readonly bool _whole; - private SheetData(string path, string? sheetTitle, string title, List columns) + private SheetData(string path, string? sheetTitle, bool whole, string title, List columns) { _path = path; _sheetTitle = sheetTitle; + _whole = whole; SheetTitle = title; Columns = columns; } @@ -27,18 +29,29 @@ private SheetData(string path, string? sheetTitle, string title, List co public List Columns { get; } - public static SheetData Load(string path, string? sheetTitle) + /// + /// 整份读入工作簿,公式当场求值;否则逐行读出,公式取文件里存着的上一次计算结果。 + /// + public static SheetData Load(string path, string? sheetTitle, bool whole = false) { if (!File.Exists(path)) throw new FileNotFoundException($"input file not found: {path}", path); // 只读到表头那一行为止,后面有多少行数据都不影响这一步的开销 using var input = File.OpenRead(path); var header = ExcelHelper.ReadHeader(input, options => options.SheetTitle = sheetTitle); - return new SheetData(path, sheetTitle, header.SheetTitle ?? "", header.Columns.ToList()); + return new SheetData(path, sheetTitle, whole, header.SheetTitle ?? "", header.Columns.ToList()); } - /// 逐行读出该表。每次遍历都重新读一遍文件。 + /// 读出该表的各行。每次遍历都重新读一遍文件。 public IEnumerable> Rows() + { + if (_whole) + return ExcelHelper.ExcelToObject>(File.ReadAllBytes(_path), _sheetTitle); + + return Streamed(); + } + + private IEnumerable> Streamed() { using var input = File.OpenRead(_path); foreach (var row in ExcelHelper.ExcelStreamToObject>(input, diff --git a/Chsword.Excel2Object.Cli/TypeInference.cs b/Chsword.Excel2Object.Cli/TypeInference.cs index 5effc0d..36f50c6 100644 --- a/Chsword.Excel2Object.Cli/TypeInference.cs +++ b/Chsword.Excel2Object.Cli/TypeInference.cs @@ -38,6 +38,12 @@ public static InferredType Infer(IEnumerable values) /// public sealed class Inference { + /// 由窄到宽,取第一个仍然成立的。静态存放:这个取值会被逐格问到。 + private static readonly InferredType[] Order = + { + InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, InferredType.DateTime + }; + private readonly HashSet _candidates = new() {InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, InferredType.DateTime}; @@ -54,11 +60,7 @@ public InferredType Result get { if (!_seen) return InferredType.String; - foreach (var candidate in new[] - { - InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, - InferredType.DateTime - }) + foreach (var candidate in Order) if (_candidates.Contains(candidate)) return candidate; diff --git a/Chsword.Excel2Object.Tests/CliTest.cs b/Chsword.Excel2Object.Tests/CliTest.cs index efb98bf..cf287e4 100644 --- a/Chsword.Excel2Object.Tests/CliTest.cs +++ b/Chsword.Excel2Object.Tests/CliTest.cs @@ -62,6 +62,65 @@ private static (int code, string stdout, string stderr) Run(params string[] args return (code, stdout.ToString(), stderr.ToString()); } + /// 一份带公式的工作簿:公式的结果要等 Excel 打开时才算出,文件里并没有存下。 + private string WriteWorkbookWithFormula() + { + var path = Path.Combine(_dir, "with-formula.xlsx"); + var bytes = new ExcelExporter().ObjectToExcelBytes(Orders, o => + { + o.ExcelType = ExcelType.Xlsx; + o.FormulaColumns.Add(new Options.FormulaColumn {Title = "Double", Formula = c => c["Qty"] * 2}); + }); + File.WriteAllBytes(path, bytes!); + return path; + } + + [TestMethod] + public void ConvertingOntoTheInputItselfKeepsTheData() + { + // 逐行读出意味着边读边写,目标若就是输入本身,先清空目标便会毁掉源文件 + var path = WriteWorkbook(); + var (code, _, stderr) = Run("convert", path, "--output", path); + Assert.AreEqual(Excel2ObjCli.Ok, code, stderr); + + using var doc = JsonDocument.Parse(File.ReadAllText(path)); + var rows = doc.RootElement.EnumerateArray().ToList(); + Assert.AreEqual(2, rows.Count); + Assert.AreEqual("Apple", rows[0].GetProperty("Product").GetString()); + } + + [TestMethod] + public void AFailedConversionLeavesTheDestinationAlone() + { + var destination = Path.Combine(_dir, "keep.json"); + File.WriteAllText(destination, "原样保留"); + + var broken = Path.Combine(_dir, "broken.xlsx"); + File.WriteAllText(broken, "这不是一份工作簿"); + var (code, _, _) = Run("convert", broken, "--output", destination); + + Assert.AreNotEqual(Excel2ObjCli.Ok, code); + Assert.AreEqual("原样保留", File.ReadAllText(destination), "失败时不应留下半个文件"); + } + + [TestMethod] + public void FormulasReadTheirStoredResultUnlessWholeIsAsked() + { + var path = WriteWorkbookWithFormula(); + + // 默认逐行读出:公式取文件里存着的结果,本库写出的文件并没有存下 + var (code, stdout, stderr) = Run("convert", path); + Assert.AreEqual(Excel2ObjCli.Ok, code, stderr); + using (var doc = JsonDocument.Parse(stdout)) + Assert.AreEqual("", doc.RootElement[0].GetProperty("Double").GetString()); + + // --whole 整份读入,公式当场求值 + var (wholeCode, wholeOut, wholeErr) = Run("convert", path, "--whole"); + Assert.AreEqual(Excel2ObjCli.Ok, wholeCode, wholeErr); + using (var doc = JsonDocument.Parse(wholeOut)) + Assert.AreEqual("8", doc.RootElement[0].GetProperty("Double").GetString()); + } + [TestMethod] public void ExcelToJsonWritesCellTextByDefault() { diff --git a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs index 9348ce9..0f9a97f 100644 --- a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs +++ b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs @@ -127,20 +127,39 @@ public void TwoColumnsMissingAreReportedOnceEach() Assert.AreEqual(0, missing[0].SimilarTitles.Count, "没有相近的就不要硬凑"); } + [TestMethod] + public void AnEmptySheetReportsEveryColumnAsMissing() + { + var workbook = new XSSFWorkbook(); + workbook.CreateSheet("空表"); + using var bytes = new MemoryStream(); + workbook.Write(bytes, true); + + var missing = new List(); + var orders = new ExcelImporter() + .ExcelToObject(bytes.ToArray(), options => options.OnMissingColumn = missing.Add).ToList(); + + // 一行都没有,表头自然也没有:模型上的每个标题都对不上 + Assert.AreEqual(0, orders.Count); + CollectionAssert.AreEquivalent(new[] {"订单号", "金额", "备注"}, missing.Select(m => m.Title).ToArray()); + Assert.AreEqual("空表", missing[0].SheetTitle); + Assert.AreEqual(0, missing[0].HeaderTitles.Count); + } + [TestMethod] public void TheHeaderCanBeReadOnItsOwn() { foreach (var excelType in new[] {ExcelType.Xlsx, ExcelType.Xls}) { var bytes = new ExcelExporter().ObjectToExcelBytes( - new[] {new Order {No = "00123", Amount = 100.5m, Memo = "线上"}}, options => - { - options.ExcelType = excelType; - options.SheetTitle = "本月"; - }); - Assert.IsNotNull(bytes); - - using var input = new MemoryStream(bytes!); + new[] {new Order {No = "00123", Amount = 100.5m, Memo = "线上"}}, options => + { + options.ExcelType = excelType; + options.SheetTitle = "本月"; + }) + ?? throw new AssertFailedException($"{excelType} 导出应有内容"); + + using var input = new MemoryStream(bytes); var header = ExcelHelper.ReadHeader(input); Assert.AreEqual("本月", header.SheetTitle, excelType.ToString()); CollectionAssert.AreEqual(new[] {"订单号", "金额", "备注"}, header.Columns.ToArray(), excelType.ToString()); diff --git a/Chsword.Excel2Object/ExcelImporter.cs b/Chsword.Excel2Object/ExcelImporter.cs index 7015719..aea4bff 100644 --- a/Chsword.Excel2Object/ExcelImporter.cs +++ b/Chsword.Excel2Object/ExcelImporter.cs @@ -48,7 +48,7 @@ public IEnumerable ExcelToObject(byte[] bytes, optionAction?.Invoke(options); var context = new ImportContext(options); var source = GetDataRows(bytes, options, context); - return ToModels(source == null ? null : AtHeader(source, options), context); + return ToModels(source == null ? null : AtHeader(source, options), context, source?.Title); } public IEnumerable ExcelToObject(byte[] bytes, string? sheetTitle) @@ -95,7 +95,8 @@ public IEnumerable ExcelStreamToObject(Stream input, var options = new ExcelImporterOptions(); optionAction?.Invoke(options); var context = new ImportContext(options); - return ToModels(AtHeader(XlsxRowReader.Rows(input, options, context), options), context); + var source = XlsxRowReader.Rows(input, options, context); + return ToModels(AtHeader(source, options), context, source.Title); } /// @@ -104,6 +105,8 @@ public IEnumerable ExcelStreamToObject(Stream input, /// /// 只读到表头那一行为止,后面有多少行数据都不影响其开销。.xls 仍须整份读入——该格式 /// 的数据并非顺序存放。传入的流由本方法读取,返回前即已读完。 + /// 表里一行都没有时给出表名与空的列;传入的根本不是工作簿则抛出 + /// 。 /// /// /// @@ -121,7 +124,9 @@ public ExcelSheetHeader ReadHeader(Stream input, Action? o var source = LooksLikeXlsx(input) ? XlsxRowReader.Rows(input, options, context) : GetDataRows(ReadAll(input), options, context); - if (source == null) return new ExcelSheetHeader(null, new List()); + + // 读不出来与「表里没有行」不同:后者给出表名与空列,前者应当说清楚 + if (source == null) throw new Excel2ObjectException("这不是一份能够打开的工作簿。"); using var rows = AtHeader(source, options); var titleRow = rows.Current; @@ -156,13 +161,14 @@ private static byte[] ReadAll(Stream input) } /// 行从哪里来并不影响其后的转换:字典与模型两条路都只认 。 - private static IEnumerable ToModels(IEnumerator? rows, ImportContext context) + private static IEnumerable ToModels(IEnumerator? rows, ImportContext context, + string? sheetTitle) where TModel : class, new() { if (typeof(TModel) == typeof(Dictionary)) return (InternalExcelToDictionary(rows, context) as IEnumerable)!; - return InternalExcelToObject(rows, context); + return InternalExcelToObject(rows, context, sheetTitle); } private static IEnumerable> InternalExcelToDictionary(IEnumerator? result, @@ -201,7 +207,7 @@ private static IEnumerable> InternalExcelToDictionary } private static IEnumerable InternalExcelToObject(IEnumerator? result, - ImportContext context) + ImportContext context, string? sheetTitle) where TModel : class, new() { if (result == null) @@ -210,7 +216,7 @@ private static IEnumerable InternalExcelToObject(IEnumerator(result, context); + var dictColumns = BuildColumnMappings(result, context, sheetTitle); while (result.MoveNext()) { @@ -227,29 +233,31 @@ private static IEnumerable InternalExcelToObject(IEnumerator> BuildColumnMappings( - IEnumerator result, ImportContext context) + IEnumerator result, ImportContext context, string? sheetTitle) where TModel : class, new() { var dict = ExcelUtil.GetPropertiesAttributesDict(); var dictColumns = new Dictionary>(); var titleRow = result.Current; - if (titleRow == null) return dictColumns; + // 表里一行都没有时表头即为空,此时模型上的每个标题都对不上,同样要上报 var headerTitles = new List(); - foreach (var cell in titleRow.Cells) - { - var title = TextOf(cell.Value) ?? string.Empty; - headerTitles.Add(title); - var prop = dict.FirstOrDefault(c => title == c.Value.Title); - if (prop.Key != null && !dictColumns.ContainsKey(cell.Key)) - dictColumns.Add(cell.Key, prop); - } + if (titleRow != null) + foreach (var cell in titleRow.Cells) + { + var title = TextOf(cell.Value) ?? string.Empty; + headerTitles.Add(title); + var prop = dict.FirstOrDefault(c => title == c.Value.Title); + if (prop.Key != null && !dictColumns.ContainsKey(cell.Key)) + dictColumns.Add(cell.Key, prop); + } // 模型上写着、表头里却没有的标题:那一列不会被填上,整列都是默认值,此处上报 var mapped = new HashSet(dictColumns.Values.Select(c => c.Value.Title), StringComparer.Ordinal); foreach (var pair in dict) if (!mapped.Contains(pair.Value.Title)) - context.ReportMissingColumn(pair.Value.Title, pair.Key.Name, titleRow.SheetTitle, headerTitles); + context.ReportMissingColumn(pair.Value.Title, pair.Key.Name, titleRow?.SheetTitle ?? sheetTitle, + headerTitles); return dictColumns; } diff --git a/README.md b/README.md index 61075b6..ca923b6 100644 --- a/README.md +++ b/README.md @@ -71,7 +71,7 @@ excel2obj generate-model orders.xlsx --class Order # 由表头生成 * **2026.09.16** - v2.12.0 - [x] ✨ **新增:** 模型上的列在表头中找不到时可以得知:`options.OnMissingColumn = m => logger.Warn(m.ToString())`。标题差一个空格、多一个单位(「金额」与「金额(元)」),导入此前不会报错,只是那一列悄悄全是默认值。回调在读过表头之后、取第一行之前调用,每个对不上的标题一次,并给出表头上与之相近的标题;在回调中抛出即可拒绝这样的文件。默认不设回调时行为与既有版本一致 - 查看 [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md) - [x] ✨ **新增:** 只读出表头:`ExcelHelper.ReadHeader(stream)` 给出表名与各列标题,只读到表头那一行为止,用于导入前核对列或据表头生成模型 -- [x] 🔧 命令行工具改为逐行读出:二十万行五列的文件,`convert` 由 13.5 秒 / 2412 MB 降到 2.6 秒 / 215 MB,`generate-model` 由 11.9 秒 / 2245 MB 降到 2.7 秒 / 204 MB,输出与此前逐字节相同 +- [x] 🔧 命令行工具改为逐行读出:二十万行五列的文件,`convert` 由 13.5 秒 / 2412 MB 降到 2.6 秒 / 215 MB,`generate-model` 由 11.9 秒 / 2245 MB 降到 2.7 秒 / 204 MB,不含公式的文件输出与此前逐字节相同。公式格取的是文件里存着的结果,确需当场求值时加 `--whole` - [x] 🐛 修复命令行工具下表头带空白时取不到值:写出的属性名保留了空白,而值永远为空——表头与数据两边的标题此前不是同一个入口取的 diff --git a/README_EN.md b/README_EN.md index 2ce9c4f..a3366d3 100644 --- a/README_EN.md +++ b/README_EN.md @@ -69,7 +69,7 @@ See [Chsword.Excel2Object.Cli/README.md](Chsword.Excel2Object.Cli/README.md). * **2026.09.16** - v2.12.0 - [x] ✨ **NEW:** Learn when a column your model declares is not in the header: `options.OnMissingColumn = m => logger.Warn(m.ToString())`. A title off by a space or carrying a unit ("Amount" against "Amount (USD)") never failed the import before - that property was simply left at its default for every row. The callback runs after the header is read and before the first row, once per title that did not match, and names the header titles closest to it; throw from it to refuse the file. With no callback set the behaviour is unchanged - See [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md) - [x] ✨ **NEW:** Read just the header: `ExcelHelper.ReadHeader(stream)` returns the sheet name and its column titles, reading no further than the header row - for checking columns before an import, or generating a model from them -- [x] 🔧 The command-line tool now reads row by row: on 200,000 rows of five columns, `convert` went from 13.5 s / 2412 MB to 2.6 s / 215 MB and `generate-model` from 11.9 s / 2245 MB to 2.7 s / 204 MB, byte for byte the same output +- [x] 🔧 The command-line tool now reads row by row: on 200,000 rows of five columns, `convert` went from 13.5 s / 2412 MB to 2.6 s / 215 MB and `generate-model` from 11.9 s / 2245 MB to 2.7 s / 204 MB, byte for byte the same output for files without formulas. A formula cell now reads the result stored in the file; pass `--whole` to read the workbook whole and evaluate formulas - [x] 🐛 Fixed the command-line tool losing values when a header title carries surrounding whitespace: the property it wrote kept the whitespace while its value was always empty, because the header and the rows did not go through the same reader diff --git a/docs/versions/v2.12.0.md b/docs/versions/v2.12.0.md index c4ff14f..d1c7f5c 100644 --- a/docs/versions/v2.12.0.md +++ b/docs/versions/v2.12.0.md @@ -52,7 +52,16 @@ Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}") | `convert --typed` | 12.9 秒 / 2428 MB | 4.1 秒 / 329 MB | | `generate-model` | 11.9 秒 / 2245 MB | 2.7 秒 / 204 MB | -三条命令的输出与此前逐字节相同。`--typed` 要先知道每列是什么类型,故读两遍文件:一遍推断,一遍写出;写 JSON 也改为边读边写,不再先在内存里搭出整棵 JSON 树。 +不含公式的文件,三条命令的输出与此前逐字节相同。`--typed` 要先知道每列是什么类型,故读两遍文件:一遍推断,一遍写出;写 JSON 也改为边读边写,不再先在内存里搭出整棵 JSON 树。 + +**公式格有一处不同**:逐行读出取的是文件里存着的上一次计算结果,而非当场求值。Excel 存盘时会写下这个结果,故 Excel 保存过的文件照常;本库导出的文件里公式没有这个结果,此时该列读作空白。确需求值时加 `--whole`,整份读入工作簿: + +```bash +excel2obj convert orders.xlsx --whole # 公式当场求值,内存随文件增长 +excel2obj generate-model orders.xlsx --whole +``` + +写出文件时先写到同目录下的临时文件,成功之后再就位:`--output` 指向输入本身时源文件不会在读到之前被清空,中途失败也不会留下半个文件。 ### 4. 修复:命令行工具下表头带空白时取不到值 @@ -67,5 +76,7 @@ Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}") ## 注意事项 +- 表里一行都没有时,表头自然也没有,模型上的每个标题都会被上报一次。 +- `ReadHeader` 在传入的根本不是工作簿时抛出 `Excel2ObjectException`;这与「表里没有行」不同,后者给出表名与空的列。 - `OnMissingColumn` 只在按模型导入时有意义:导入为 `Dictionary` 时列由表头决定,无所谓对不对得上。 - 表头中重复的标题以最左边那一列为准(v2.11.0 起),因此「某个标题存在但对到了另一列」不属于本次上报的范围。 From 0e55529063d807c24a7ca825557889058d888611 Mon Sep 17 00:00:00 2001 From: chsword Date: Wed, 16 Sep 2026 14:03:13 +0000 Subject: [PATCH 3/3] =?UTF-8?q?=E6=8C=89=E5=A4=8D=E6=9F=A5=E6=84=8F?= =?UTF-8?q?=E8=A7=81=E4=BF=AE=E6=AD=A3=E4=B8=89=E5=A4=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --whole 没登记进不带值的开关集合,紧随其后的位置参数会被当成它的值吃掉:excel2obj convert --whole a.xlsx 报「没有输入文件」。只有把它写在末尾才碰巧能用,而 README 与新加的用例恰好都那样写,因而没被发现。现补上登记,并把用例改为把它写在路径之前。 类型推断此前以标题为键收集,表头有重名的列时抛「同名的键已存在」。generate-model 本有 Unique 来给重名列起不同的属性名,这一改反倒让它在这种文件上挂掉。改为按列序号存放。 没有表名时,列缺失那句话会以一个孤零零的「的」开头。 Generated with [Claude Code](https://claude.ai/code) via [Happy](https://happy.engineering) Co-Authored-By: Claude Opus 5 Co-Authored-By: Happy --- Chsword.Excel2Object.Cli/ConvertCommand.cs | 10 +++---- Chsword.Excel2Object.Cli/Excel2ObjCli.cs | 2 +- .../GenerateModelCommand.cs | 2 +- Chsword.Excel2Object.Cli/SheetData.cs | 12 ++++---- Chsword.Excel2Object.Tests/CliTest.cs | 30 +++++++++++++++++-- .../ImportDiagnosticsTest.cs | 10 +++++++ .../Options/ExcelColumnMissing.cs | 4 +-- 7 files changed, 54 insertions(+), 16 deletions(-) diff --git a/Chsword.Excel2Object.Cli/ConvertCommand.cs b/Chsword.Excel2Object.Cli/ConvertCommand.cs index 1923566..e8b91dc 100644 --- a/Chsword.Excel2Object.Cli/ConvertCommand.cs +++ b/Chsword.Excel2Object.Cli/ConvertCommand.cs @@ -111,7 +111,7 @@ public static void WriteJson(string path, string? sheet, bool typed, bool whole, { var data = SheetData.Load(path, sheet, whole); // 每列的类型在此定下,不在写每一格时反复去问 - var types = typed ? data.Infer().ToDictionary(c => c.Key, c => c.Value.Result, StringComparer.Ordinal) : null; + var types = typed ? data.Infer().Select(inference => inference.Result).ToArray() : null; using var writer = new Utf8JsonWriter(destination, new JsonWriterOptions {Indented = true, Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping}); @@ -119,14 +119,14 @@ public static void WriteJson(string path, string? sheet, bool typed, bool whole, foreach (var row in data.Rows()) { writer.WriteStartObject(); - foreach (var column in data.Columns) + for (var i = 0; i < data.Columns.Count; i++) { - writer.WritePropertyName(column); - var text = SheetData.Text(row, column); + writer.WritePropertyName(data.Columns[i]); + var text = SheetData.Text(row, data.Columns[i]); if (types == null) writer.WriteStringValue(text); else - WriteTypedValue(writer, text, types[column]); + WriteTypedValue(writer, text, types[i]); } writer.WriteEndObject(); diff --git a/Chsword.Excel2Object.Cli/Excel2ObjCli.cs b/Chsword.Excel2Object.Cli/Excel2ObjCli.cs index e7eb25c..e4d4b77 100644 --- a/Chsword.Excel2Object.Cli/Excel2ObjCli.cs +++ b/Chsword.Excel2Object.Cli/Excel2ObjCli.cs @@ -130,7 +130,7 @@ public static Arguments Parse(IEnumerable args) } /// Options that never take a value, so a following positional argument is not swallowed. - private static readonly HashSet Switches = new(StringComparer.Ordinal) {"typed", "xls"}; + private static readonly HashSet Switches = new(StringComparer.Ordinal) {"typed", "xls", "whole"}; public bool Has(string name) { diff --git a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs index 4db8e81..65c57a0 100644 --- a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs +++ b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs @@ -48,7 +48,7 @@ public static string Generate(SheetData data, string className, string? ns) for (var i = 0; i < data.Columns.Count; i++) { var title = data.Columns[i]; - var inference = inferences[title]; + var inference = inferences[i]; var type = inference.Result; // 一行都没有,或出现过空值,该属性即为可空 var nullable = !inference.Any || inference.HasBlank; diff --git a/Chsword.Excel2Object.Cli/SheetData.cs b/Chsword.Excel2Object.Cli/SheetData.cs index 4c70c3b..dd87c8e 100644 --- a/Chsword.Excel2Object.Cli/SheetData.cs +++ b/Chsword.Excel2Object.Cli/SheetData.cs @@ -59,13 +59,15 @@ private IEnumerable> Streamed() yield return row; } - /// 各列的类型推断,一遍读完。 - public Dictionary Infer() + /// + /// 各列的类型推断,一遍读完,按列的先后存放——表头允许有重名的列,故不以标题为键。 + /// + public TypeInference.Inference[] Infer() { - var inferences = Columns.ToDictionary(c => c, _ => new TypeInference.Inference(), StringComparer.Ordinal); + var inferences = Columns.Select(_ => new TypeInference.Inference()).ToArray(); foreach (var row in Rows()) - foreach (var column in Columns) - inferences[column].Observe(Text(row, column)); + for (var i = 0; i < Columns.Count; i++) + inferences[i].Observe(Text(row, Columns[i])); return inferences; } diff --git a/Chsword.Excel2Object.Tests/CliTest.cs b/Chsword.Excel2Object.Tests/CliTest.cs index cf287e4..f94d8f7 100644 --- a/Chsword.Excel2Object.Tests/CliTest.cs +++ b/Chsword.Excel2Object.Tests/CliTest.cs @@ -114,13 +114,39 @@ public void FormulasReadTheirStoredResultUnlessWholeIsAsked() using (var doc = JsonDocument.Parse(stdout)) Assert.AreEqual("", doc.RootElement[0].GetProperty("Double").GetString()); - // --whole 整份读入,公式当场求值 - var (wholeCode, wholeOut, wholeErr) = Run("convert", path, "--whole"); + // --whole 是个开关,放在路径之前也不应把路径吞掉 + var (wholeCode, wholeOut, wholeErr) = Run("convert", "--whole", path); Assert.AreEqual(Excel2ObjCli.Ok, wholeCode, wholeErr); using (var doc = JsonDocument.Parse(wholeOut)) Assert.AreEqual("8", doc.RootElement[0].GetProperty("Double").GetString()); } + [TestMethod] + public void DuplicateHeaderTitlesStillGenerateAModel() + { + // 表头允许有重名的列,生成的属性名靠 Unique 区分 + var path = Path.Combine(_dir, "dup.xlsx"); + var workbook = new NPOI.XSSF.UserModel.XSSFWorkbook(); + var sheet = workbook.CreateSheet("Dup"); + var header = sheet.CreateRow(0); + header.CreateCell(0).SetCellValue("Name"); + header.CreateCell(1).SetCellValue("Qty"); + header.CreateCell(2).SetCellValue("Name"); + var row = sheet.CreateRow(1); + row.CreateCell(0).SetCellValue("甲"); + row.CreateCell(1).SetCellValue(2); + row.CreateCell(2).SetCellValue("乙"); + using (var file = File.Create(path)) workbook.Write(file, false); + + var (code, stdout, stderr) = Run("generate-model", path, "--class=Dup"); + Assert.AreEqual(Excel2ObjCli.Ok, code, stderr); + StringAssert.Contains(stdout, "public string? Name { get; set; }"); + StringAssert.Contains(stdout, "public string? Name2 { get; set; }"); + + var (typedCode, _, typedErr) = Run("convert", path, "--typed"); + Assert.AreEqual(Excel2ObjCli.Ok, typedCode, typedErr); + } + [TestMethod] public void ExcelToJsonWritesCellTextByDefault() { diff --git a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs index 0f9a97f..b055828 100644 --- a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs +++ b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs @@ -146,6 +146,16 @@ public void AnEmptySheetReportsEveryColumnAsMissing() Assert.AreEqual(0, missing[0].HeaderTitles.Count); } + [TestMethod] + public void TheMessageReadsWellWithAndWithoutASheetName() + { + var withSheet = new ExcelColumnMissing("金额", "Amount", "数据", new[] {"金额(元)"}, new[] {"金额(元)"}); + StringAssert.StartsWith(withSheet.ToString(), "工作表 [数据] 的表头中没有 [金额]"); + + var withoutSheet = new ExcelColumnMissing("金额", "Amount", null, new string[0], new string[0]); + StringAssert.StartsWith(withoutSheet.ToString(), "表头中没有 [金额]"); + } + [TestMethod] public void TheHeaderCanBeReadOnItsOwn() { diff --git a/Chsword.Excel2Object/Options/ExcelColumnMissing.cs b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs index 9a261df..0692bf9 100644 --- a/Chsword.Excel2Object/Options/ExcelColumnMissing.cs +++ b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs @@ -38,10 +38,10 @@ public ExcelColumnMissing(string title, string propertyName, string? sheetTitle, public override string ToString() { - var where = SheetTitle == null ? "" : $"工作表 [{SheetTitle}] "; + var where = SheetTitle == null ? "" : $"工作表 [{SheetTitle}] 的"; var similar = SimilarTitles.Count == 0 ? "" : $",表头上与之相近的是 [{string.Join("]、[", SimilarTitles)}]"; - return $"{where}的表头中没有 [{Title}] 这一列(属性 {PropertyName} 因而不会被填上){similar}。"; + return $"{where}表头中没有 [{Title}] 这一列(属性 {PropertyName} 因而不会被填上){similar}。"; } }