diff --git a/Chsword.Excel2Object.Cli/ConvertCommand.cs b/Chsword.Excel2Object.Cli/ConvertCommand.cs
index 304a852..e8b91dc 100644
--- a/Chsword.Excel2Object.Cli/ConvertCommand.cs
+++ b/Chsword.Excel2Object.Cli/ConvertCommand.cs
@@ -1,5 +1,6 @@
using System.Data;
using System.Globalization;
+using System.Text;
using System.Text.Encodings.Web;
using System.Text.Json;
using System.Text.Json.Nodes;
@@ -9,12 +10,6 @@ namespace Chsword.Excel2Object.Cli;
/// excel2obj convert: Excel -> JSON or JSON -> Excel, decided per input by file extension.
public static class ConvertCommand
{
- private static readonly JsonSerializerOptions JsonOptions = new()
- {
- WriteIndented = true,
- Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping
- };
-
public static int Run(Arguments args, TextWriter output, TextWriter error)
{
if (args.Positional.Count == 0) throw new UsageException("convert needs at least one input file");
@@ -30,29 +25,22 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
Directory.CreateDirectory(target);
}
+ var whole = args.Has("whole");
+
foreach (var input in inputs)
{
if (!File.Exists(input)) throw new FileNotFoundException($"input file not found: {input}", input);
- string? destination;
- if (inputs.Count > 1)
- destination = Path.Combine(target!, Path.GetFileNameWithoutExtension(input) +
- (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx"));
- else if (target != null && Directory.Exists(target))
- destination = Path.Combine(target, Path.GetFileNameWithoutExtension(input) +
- (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx"));
- else
- destination = target;
+ var destination = Destination(input, target, inputs.Count > 1, xls);
if (SheetData.IsExcel(input))
{
- var json = ExcelToJson(input, sheet, typed);
if (destination == null)
{
- output.WriteLine(json);
+ output.WriteLine(ExcelToJson(input, sheet, typed, whole));
}
else
{
- File.WriteAllText(destination, json);
+ Replace(destination, file => WriteJson(input, sheet, typed, whole, file));
error.WriteLine($"{input} -> {destination}");
}
}
@@ -63,7 +51,8 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
var excelType = xls || destination.EndsWith(".xls", StringComparison.OrdinalIgnoreCase)
? ExcelType.Xls
: ExcelType.Xlsx;
- File.WriteAllBytes(destination, JsonToExcel(input, sheet, excelType));
+ var bytes = JsonToExcel(input, sheet, excelType);
+ Replace(destination, file => file.Write(bytes, 0, bytes.Length));
error.WriteLine($"{input} -> {destination}");
}
}
@@ -71,48 +60,111 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
return Excel2ObjCli.Ok;
}
- public static string ExcelToJson(string path, string? sheet, bool typed)
+ /// 该输入写到哪里去:多个输入时按目录派生文件名,单个输入时即 --output 本身。
+ private static string? Destination(string input, string? target, bool severalInputs, bool xls)
{
- var data = SheetData.Load(path, sheet);
- var types = typed
- ? data.Columns.ToDictionary(c => c, c => TypeInference.Infer(data.ColumnValues(c)))
- : null;
+ var name = Path.GetFileNameWithoutExtension(input) +
+ (SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx");
+ if (severalInputs) return Path.Combine(target!, name);
+ if (target != null && Directory.Exists(target)) return Path.Combine(target, name);
+ return target;
+ }
- var array = new JsonArray();
- foreach (var row in data.Rows)
+ ///
+ /// 先写到同目录下的临时文件,成功之后再就位。读的是一行一行来的,若直接往目标文件写:
+ /// 目标与输入是同一个文件时,源在读到之前就已被清空;中途失败也会留下半个文件。
+ ///
+ private static void Replace(string destination, Action write)
+ {
+ var temporary = destination + ".tmp" + Path.GetRandomFileName();
+ try
+ {
+ using (var file = File.Create(temporary)) write(file);
+
+ if (File.Exists(destination)) File.Delete(destination);
+ File.Move(temporary, destination);
+ }
+ catch
{
- var item = new JsonObject();
- foreach (var column in data.Columns)
+ if (File.Exists(temporary)) File.Delete(temporary);
+ throw;
+ }
+ }
+
+ /// 写到标准输出时才用得到:那里本就要把整段文本拿在手上。
+ public static string ExcelToJson(string path, string? sheet, bool typed, bool whole = false)
+ {
+ using var buffer = new MemoryStream();
+ WriteJson(path, sheet, typed, whole, buffer);
+ return Encoding.UTF8.GetString(buffer.ToArray());
+ }
+
+ ///
+ /// 一行读出、一行写出,中途不把整份数据攒在内存里。--typed 要先知道每列是什么类型,
+ /// 故先过一遍推断,再过一遍写出——两遍各是一次顺序读。
+ ///
+ ///
+ /// 整份读入工作簿,公式当场求值。逐行读出取的是文件里存着的上一次计算结果,没有存下结果的
+ /// 公式(本库导出的文件即如此)因而读作空白;确需求值时用这条路,代价是内存随文件增长。
+ ///
+ public static void WriteJson(string path, string? sheet, bool typed, bool whole, Stream destination)
+ {
+ var data = SheetData.Load(path, sheet, whole);
+ // 每列的类型在此定下,不在写每一格时反复去问
+ var types = typed ? data.Infer().Select(inference => inference.Result).ToArray() : null;
+
+ using var writer = new Utf8JsonWriter(destination,
+ new JsonWriterOptions {Indented = true, Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping});
+ writer.WriteStartArray();
+ foreach (var row in data.Rows())
+ {
+ writer.WriteStartObject();
+ for (var i = 0; i < data.Columns.Count; i++)
{
- var text = row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "";
- item[column] = types == null ? JsonValue.Create(text) : ToJsonValue(text, types[column]);
+ writer.WritePropertyName(data.Columns[i]);
+ var text = SheetData.Text(row, data.Columns[i]);
+ if (types == null)
+ writer.WriteStringValue(text);
+ else
+ WriteTypedValue(writer, text, types[i]);
}
- array.Add(item);
+ writer.WriteEndObject();
}
- return array.ToJsonString(JsonOptions);
+ writer.WriteEndArray();
}
- private static JsonNode? ToJsonValue(string text, InferredType type)
+ private static void WriteTypedValue(Utf8JsonWriter writer, string text, InferredType type)
{
- if (type != InferredType.String && string.IsNullOrWhiteSpace(text)) return null;
+ if (type != InferredType.String && string.IsNullOrWhiteSpace(text))
+ {
+ writer.WriteNullValue();
+ return;
+ }
+
switch (type)
{
case InferredType.Bool:
TypeInference.TryParseBool(text, out var flag);
- return JsonValue.Create(flag);
+ writer.WriteBooleanValue(flag);
+ break;
case InferredType.Int:
- return JsonValue.Create(int.Parse(text, CultureInfo.InvariantCulture));
+ writer.WriteNumberValue(int.Parse(text, CultureInfo.InvariantCulture));
+ break;
case InferredType.Long:
- return JsonValue.Create(long.Parse(text, CultureInfo.InvariantCulture));
+ writer.WriteNumberValue(long.Parse(text, CultureInfo.InvariantCulture));
+ break;
case InferredType.Decimal:
- return JsonValue.Create(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture));
+ writer.WriteNumberValue(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture));
+ break;
case InferredType.DateTime:
TypeInference.TryParseDateTime(text, out var date);
- return JsonValue.Create(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture));
+ writer.WriteStringValue(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture));
+ break;
default:
- return JsonValue.Create(text);
+ writer.WriteStringValue(text);
+ break;
}
}
diff --git a/Chsword.Excel2Object.Cli/Excel2ObjCli.cs b/Chsword.Excel2Object.Cli/Excel2ObjCli.cs
index e7eb25c..e4d4b77 100644
--- a/Chsword.Excel2Object.Cli/Excel2ObjCli.cs
+++ b/Chsword.Excel2Object.Cli/Excel2ObjCli.cs
@@ -130,7 +130,7 @@ public static Arguments Parse(IEnumerable args)
}
/// Options that never take a value, so a following positional argument is not swallowed.
- private static readonly HashSet Switches = new(StringComparer.Ordinal) {"typed", "xls"};
+ private static readonly HashSet Switches = new(StringComparer.Ordinal) {"typed", "xls", "whole"};
public bool Has(string name)
{
diff --git a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs
index 6ba684e..65c57a0 100644
--- a/Chsword.Excel2Object.Cli/GenerateModelCommand.cs
+++ b/Chsword.Excel2Object.Cli/GenerateModelCommand.cs
@@ -11,7 +11,7 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
if (args.Positional.Count != 1) throw new UsageException("generate-model needs exactly one input file");
var input = args.Positional[0];
var sheet = args.Get("sheet");
- var data = SheetData.Load(input, sheet);
+ var data = SheetData.Load(input, sheet, args.Has("whole"));
var className = args.Get("class") ?? ToIdentifier(data.SheetTitle, "Model");
var code = Generate(data, className, args.Get("namespace"));
@@ -44,12 +44,14 @@ public static string Generate(SheetData data, string className, string? ns)
sb.Append("public class ").AppendLine(className);
sb.AppendLine("{");
var used = new HashSet(StringComparer.Ordinal) {className};
+ var inferences = data.Infer();
for (var i = 0; i < data.Columns.Count; i++)
{
var title = data.Columns[i];
- var values = data.ColumnValues(title).ToList();
- var type = TypeInference.Infer(values);
- var nullable = values.Count == 0 || values.Any(string.IsNullOrWhiteSpace);
+ var inference = inferences[i];
+ var type = inference.Result;
+ // 一行都没有,或出现过空值,该属性即为可空
+ var nullable = !inference.Any || inference.HasBlank;
var name = Unique(ToIdentifier(title, $"Column{i + 1}"), used);
if (i > 0) sb.AppendLine();
diff --git a/Chsword.Excel2Object.Cli/README.md b/Chsword.Excel2Object.Cli/README.md
index fcd1094..90bd065 100644
--- a/Chsword.Excel2Object.Cli/README.md
+++ b/Chsword.Excel2Object.Cli/README.md
@@ -14,11 +14,18 @@ excel2obj convert orders.xlsx # prints a JSON array to std
excel2obj convert orders.xlsx --output orders.json # writes the file
excel2obj convert orders.xlsx --sheet Orders --typed
excel2obj convert a.xlsx b.xls --output ./json/ # several inputs need an output directory
+excel2obj convert orders.xlsx --whole # read the workbook whole and evaluate formulas
```
-Each row becomes an object keyed by the header row. Values are the cell text (formulas are evaluated); with
-`--typed`, a column whose non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as
-JSON numbers, booleans or `yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`.
+Each row becomes an object keyed by the header row. Values are the cell text; with `--typed`, a column whose
+non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as JSON numbers, booleans or
+`yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`.
+
+The sheet is read row by row, so a file far larger than memory converts fine. One consequence: a **formula cell
+reads the result stored in the file** rather than being evaluated. Excel writes that result when it saves, so
+files saved by Excel read as before; a file written by this library carries formulas with no stored result, and
+those cells read as blank. Pass `--whole` to read the workbook whole and evaluate formulas instead - the memory
+it needs then grows with the file. `--whole` works with `generate-model` as well.
## JSON -> Excel
diff --git a/Chsword.Excel2Object.Cli/SheetData.cs b/Chsword.Excel2Object.Cli/SheetData.cs
index 66f56b6..dd87c8e 100644
--- a/Chsword.Excel2Object.Cli/SheetData.cs
+++ b/Chsword.Excel2Object.Cli/SheetData.cs
@@ -1,46 +1,80 @@
-using NPOI.SS.UserModel;
+using Chsword.Excel2Object.Options;
namespace Chsword.Excel2Object.Cli;
-/// A sheet as the importer sees it: the header titles in column order and one string per cell.
+///
+/// A sheet as the importer sees it: the header titles in column order, and its rows on demand.
+///
+///
+/// 行不再一次性读进内存: 每次遍历都逐行读出,命令行因而能处理远大于内存的
+/// 文件。需要过两遍数据的地方(--typed 先推断类型再写出)就遍历两次,各自的开销是一遍
+/// 顺序读。
+///
public sealed class SheetData
{
- public SheetData(string sheetTitle, List columns, List> rows)
+ private readonly string _path;
+ private readonly string? _sheetTitle;
+ private readonly bool _whole;
+
+ private SheetData(string path, string? sheetTitle, bool whole, string title, List columns)
{
- SheetTitle = sheetTitle;
+ _path = path;
+ _sheetTitle = sheetTitle;
+ _whole = whole;
+ SheetTitle = title;
Columns = columns;
- Rows = rows;
}
public string SheetTitle { get; }
+
public List Columns { get; }
- public List> Rows { get; }
- public static SheetData Load(string path, string? sheetTitle)
+ ///
+ /// 整份读入工作簿,公式当场求值;否则逐行读出,公式取文件里存着的上一次计算结果。
+ ///
+ public static SheetData Load(string path, string? sheetTitle, bool whole = false)
{
if (!File.Exists(path)) throw new FileNotFoundException($"input file not found: {path}", path);
- var bytes = File.ReadAllBytes(path);
- var rows = ExcelHelper.ExcelToObject>(bytes, sheetTitle).ToList();
- var (title, columns) = ReadHeader(bytes, sheetTitle);
- return new SheetData(title, columns, rows);
+
+ // 只读到表头那一行为止,后面有多少行数据都不影响这一步的开销
+ using var input = File.OpenRead(path);
+ var header = ExcelHelper.ReadHeader(input, options => options.SheetTitle = sheetTitle);
+ return new SheetData(path, sheetTitle, whole, header.SheetTitle ?? "", header.Columns.ToList());
}
- /// The header row straight from the workbook, so an empty sheet still yields its columns.
- private static (string title, List columns) ReadHeader(byte[] bytes, string? sheetTitle)
+ /// 读出该表的各行。每次遍历都重新读一遍文件。
+ public IEnumerable> Rows()
{
- using var stream = new MemoryStream(bytes);
- var workbook = WorkbookFactory.Create(stream);
- var sheet = string.IsNullOrEmpty(sheetTitle) ? workbook.GetSheetAt(0) : workbook.GetSheet(sheetTitle);
- if (sheet == null) throw new Excel2ObjectException($"The specified sheet:[{sheetTitle}] does not exist");
- var header = sheet.GetRow(sheet.FirstRowNum);
- // untrimmed on purpose: the importer keys rows by the exact header text
- var columns = header?.Cells.Select(cell => cell.ToString() ?? "").ToList() ?? new List();
- return (sheet.SheetName, columns);
+ if (_whole)
+ return ExcelHelper.ExcelToObject>(File.ReadAllBytes(_path), _sheetTitle);
+
+ return Streamed();
+ }
+
+ private IEnumerable> Streamed()
+ {
+ using var input = File.OpenRead(_path);
+ foreach (var row in ExcelHelper.ExcelStreamToObject>(input,
+ options => options.SheetTitle = _sheetTitle))
+ yield return row;
+ }
+
+ ///
+ /// 各列的类型推断,一遍读完,按列的先后存放——表头允许有重名的列,故不以标题为键。
+ ///
+ public TypeInference.Inference[] Infer()
+ {
+ var inferences = Columns.Select(_ => new TypeInference.Inference()).ToArray();
+ foreach (var row in Rows())
+ for (var i = 0; i < Columns.Count; i++)
+ inferences[i].Observe(Text(row, Columns[i]));
+
+ return inferences;
}
- public IEnumerable ColumnValues(string column)
+ public static string Text(IReadOnlyDictionary row, string column)
{
- return Rows.Select(row => row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "");
+ return row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "";
}
public static bool IsExcel(string path)
diff --git a/Chsword.Excel2Object.Cli/TypeInference.cs b/Chsword.Excel2Object.Cli/TypeInference.cs
index 04b6961..36f50c6 100644
--- a/Chsword.Excel2Object.Cli/TypeInference.cs
+++ b/Chsword.Excel2Object.Cli/TypeInference.cs
@@ -27,26 +27,59 @@ public static class TypeInference
public static InferredType Infer(IEnumerable values)
{
- var seen = false;
- var candidates = new HashSet
+ var inference = new Inference();
+ foreach (var value in values) inference.Observe(value);
+ return inference.Result;
+ }
+
+ ///
+ /// 一列的推断过程:一个值一个值地喂进来,不必先把整列攒在内存里。命令行要处理的文件可能有
+ /// 几十万行,逐行读出便是为此。
+ ///
+ public sealed class Inference
+ {
+ /// 由窄到宽,取第一个仍然成立的。静态存放:这个取值会被逐格问到。
+ private static readonly InferredType[] Order =
+ {
+ InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, InferredType.DateTime
+ };
+
+ private readonly HashSet _candidates = new()
{InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal, InferredType.DateTime};
- foreach (var value in values)
+
+ private bool _seen;
+
+ /// 这一列上是否出现过空值——空值本身不参与类型判断,却决定该属性是否可空。
+ public bool HasBlank { get; private set; }
+
+ /// 是否有过任何一行。一行都没有的列按字符串处理,且算作可空。
+ public bool Any { get; private set; }
+
+ public InferredType Result
{
- if (string.IsNullOrWhiteSpace(value)) continue;
- seen = true;
- candidates.RemoveWhere(candidate => !Fits(candidate, value));
- if (candidates.Count == 0) break;
+ get
+ {
+ if (!_seen) return InferredType.String;
+ foreach (var candidate in Order)
+ if (_candidates.Contains(candidate))
+ return candidate;
+
+ return InferredType.String;
+ }
}
- if (!seen) return InferredType.String;
- foreach (var candidate in new[]
- {
- InferredType.Bool, InferredType.Int, InferredType.Long, InferredType.Decimal,
- InferredType.DateTime
- })
- if (candidates.Contains(candidate))
- return candidate;
- return InferredType.String;
+ public void Observe(string value)
+ {
+ Any = true;
+ if (string.IsNullOrWhiteSpace(value))
+ {
+ HasBlank = true;
+ return;
+ }
+
+ _seen = true;
+ if (_candidates.Count > 0) _candidates.RemoveWhere(candidate => !Fits(candidate, value));
+ }
}
public static bool Fits(InferredType type, string value)
diff --git a/Chsword.Excel2Object.Tests/CliTest.cs b/Chsword.Excel2Object.Tests/CliTest.cs
index efb98bf..f94d8f7 100644
--- a/Chsword.Excel2Object.Tests/CliTest.cs
+++ b/Chsword.Excel2Object.Tests/CliTest.cs
@@ -62,6 +62,91 @@ private static (int code, string stdout, string stderr) Run(params string[] args
return (code, stdout.ToString(), stderr.ToString());
}
+ /// 一份带公式的工作簿:公式的结果要等 Excel 打开时才算出,文件里并没有存下。
+ private string WriteWorkbookWithFormula()
+ {
+ var path = Path.Combine(_dir, "with-formula.xlsx");
+ var bytes = new ExcelExporter().ObjectToExcelBytes(Orders, o =>
+ {
+ o.ExcelType = ExcelType.Xlsx;
+ o.FormulaColumns.Add(new Options.FormulaColumn {Title = "Double", Formula = c => c["Qty"] * 2});
+ });
+ File.WriteAllBytes(path, bytes!);
+ return path;
+ }
+
+ [TestMethod]
+ public void ConvertingOntoTheInputItselfKeepsTheData()
+ {
+ // 逐行读出意味着边读边写,目标若就是输入本身,先清空目标便会毁掉源文件
+ var path = WriteWorkbook();
+ var (code, _, stderr) = Run("convert", path, "--output", path);
+ Assert.AreEqual(Excel2ObjCli.Ok, code, stderr);
+
+ using var doc = JsonDocument.Parse(File.ReadAllText(path));
+ var rows = doc.RootElement.EnumerateArray().ToList();
+ Assert.AreEqual(2, rows.Count);
+ Assert.AreEqual("Apple", rows[0].GetProperty("Product").GetString());
+ }
+
+ [TestMethod]
+ public void AFailedConversionLeavesTheDestinationAlone()
+ {
+ var destination = Path.Combine(_dir, "keep.json");
+ File.WriteAllText(destination, "原样保留");
+
+ var broken = Path.Combine(_dir, "broken.xlsx");
+ File.WriteAllText(broken, "这不是一份工作簿");
+ var (code, _, _) = Run("convert", broken, "--output", destination);
+
+ Assert.AreNotEqual(Excel2ObjCli.Ok, code);
+ Assert.AreEqual("原样保留", File.ReadAllText(destination), "失败时不应留下半个文件");
+ }
+
+ [TestMethod]
+ public void FormulasReadTheirStoredResultUnlessWholeIsAsked()
+ {
+ var path = WriteWorkbookWithFormula();
+
+ // 默认逐行读出:公式取文件里存着的结果,本库写出的文件并没有存下
+ var (code, stdout, stderr) = Run("convert", path);
+ Assert.AreEqual(Excel2ObjCli.Ok, code, stderr);
+ using (var doc = JsonDocument.Parse(stdout))
+ Assert.AreEqual("", doc.RootElement[0].GetProperty("Double").GetString());
+
+ // --whole 是个开关,放在路径之前也不应把路径吞掉
+ var (wholeCode, wholeOut, wholeErr) = Run("convert", "--whole", path);
+ Assert.AreEqual(Excel2ObjCli.Ok, wholeCode, wholeErr);
+ using (var doc = JsonDocument.Parse(wholeOut))
+ Assert.AreEqual("8", doc.RootElement[0].GetProperty("Double").GetString());
+ }
+
+ [TestMethod]
+ public void DuplicateHeaderTitlesStillGenerateAModel()
+ {
+ // 表头允许有重名的列,生成的属性名靠 Unique 区分
+ var path = Path.Combine(_dir, "dup.xlsx");
+ var workbook = new NPOI.XSSF.UserModel.XSSFWorkbook();
+ var sheet = workbook.CreateSheet("Dup");
+ var header = sheet.CreateRow(0);
+ header.CreateCell(0).SetCellValue("Name");
+ header.CreateCell(1).SetCellValue("Qty");
+ header.CreateCell(2).SetCellValue("Name");
+ var row = sheet.CreateRow(1);
+ row.CreateCell(0).SetCellValue("甲");
+ row.CreateCell(1).SetCellValue(2);
+ row.CreateCell(2).SetCellValue("乙");
+ using (var file = File.Create(path)) workbook.Write(file, false);
+
+ var (code, stdout, stderr) = Run("generate-model", path, "--class=Dup");
+ Assert.AreEqual(Excel2ObjCli.Ok, code, stderr);
+ StringAssert.Contains(stdout, "public string? Name { get; set; }");
+ StringAssert.Contains(stdout, "public string? Name2 { get; set; }");
+
+ var (typedCode, _, typedErr) = Run("convert", path, "--typed");
+ Assert.AreEqual(Excel2ObjCli.Ok, typedCode, typedErr);
+ }
+
[TestMethod]
public void ExcelToJsonWritesCellTextByDefault()
{
diff --git a/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs
new file mode 100644
index 0000000..b055828
--- /dev/null
+++ b/Chsword.Excel2Object.Tests/ImportDiagnosticsTest.cs
@@ -0,0 +1,216 @@
+using System;
+using System.Collections.Generic;
+using System.IO;
+using System.Linq;
+using Chsword.Excel2Object.Options;
+using Microsoft.VisualStudio.TestTools.UnitTesting;
+using NPOI.SS.UserModel;
+using NPOI.XSSF.UserModel;
+
+namespace Chsword.Excel2Object.Tests;
+
+///
+/// 导入前的两件事:模型上的列在表头里有没有,以及表头本身是什么。
+///
+[TestClass]
+public class ImportDiagnosticsTest
+{
+ public class Order
+ {
+ [ExcelTitle("订单号")] public string No { get; set; } = "";
+ [ExcelTitle("金额")] public decimal Amount { get; set; }
+ [ExcelTitle("备注")] public string? Memo { get; set; }
+ }
+
+ /// 表头由调用方写定,好把「对不上」这件事摆出来。
+ private static byte[] Workbook(string[] titles, string[][] rows, string sheetTitle = "数据")
+ {
+ var workbook = new XSSFWorkbook();
+ var sheet = workbook.CreateSheet(sheetTitle);
+ var header = sheet.CreateRow(0);
+ for (var i = 0; i < titles.Length; i++) header.CreateCell(i).SetCellValue(titles[i]);
+ for (var r = 0; r < rows.Length; r++)
+ {
+ var row = sheet.CreateRow(r + 1);
+ for (var i = 0; i < rows[r].Length; i++) row.CreateCell(i).SetCellValue(rows[r][i]);
+ }
+
+ using var bytes = new MemoryStream();
+ workbook.Write(bytes, true);
+ return bytes.ToArray();
+ }
+
+ private static byte[] Sample()
+ {
+ // 「金额」在表里叫「金额(元)」——导入不会报错,只是那一列悄悄全为空
+ return Workbook(new[] {"订单号", "金额(元)", "备注"},
+ new[] {new[] {"00123", "100.5", "线上"}, new[] {"00124", "200", ""}});
+ }
+
+ [TestMethod]
+ public void AColumnTheHeaderDoesNotHaveIsReported()
+ {
+ var missing = new List();
+ var orders = new ExcelImporter()
+ .ExcelToObject(Sample(), options => options.OnMissingColumn = missing.Add).ToList();
+
+ // 沿用既有行为:那一列取默认值,导入照常
+ Assert.AreEqual(2, orders.Count);
+ Assert.AreEqual(0m, orders[0].Amount);
+ Assert.AreEqual("00123", orders[0].No);
+
+ Assert.AreEqual(1, missing.Count);
+ Assert.AreEqual("金额", missing[0].Title);
+ Assert.AreEqual(nameof(Order.Amount), missing[0].PropertyName);
+ Assert.AreEqual("数据", missing[0].SheetTitle);
+ CollectionAssert.AreEqual(new[] {"订单号", "金额(元)", "备注"}, missing[0].HeaderTitles.ToArray());
+ // 相近的标题一并给出,否则「差在哪里」还得自己翻文件
+ CollectionAssert.AreEqual(new[] {"金额(元)"}, missing[0].SimilarTitles.ToArray());
+ StringAssert.Contains(missing[0].ToString(), "金额(元)");
+ }
+
+ [TestMethod]
+ public void NothingIsReportedWhenEveryColumnIsFound()
+ {
+ var missing = new List();
+ var bytes = Workbook(new[] {"订单号", "金额", "备注"}, new[] {new[] {"00123", "100.5", "线上"}});
+ var orders = new ExcelImporter()
+ .ExcelToObject(bytes, options => options.OnMissingColumn = missing.Add).ToList();
+
+ Assert.AreEqual(100.5m, orders[0].Amount);
+ Assert.AreEqual(0, missing.Count);
+ }
+
+ [TestMethod]
+ public void WithoutTheCallbackTheImportIsUnchanged()
+ {
+ // 不设回调时一切照旧:这是既有版本的行为
+ var orders = new ExcelImporter().ExcelToObject(Sample()).ToList();
+ Assert.AreEqual(2, orders.Count);
+ Assert.AreEqual(0m, orders[0].Amount);
+ }
+
+ [TestMethod]
+ public void ThrowingFromTheCallbackRefusesTheFile()
+ {
+ var e = Assert.ThrowsException(() =>
+ new ExcelImporter().ExcelToObject(Sample(),
+ options => options.OnMissingColumn = m => throw new Excel2ObjectException(m.ToString()))
+ .ToList());
+
+ StringAssert.Contains(e.Message, "[金额]");
+ }
+
+ [TestMethod]
+ public void TheStreamingImportReportsItToo()
+ {
+ var missing = new List();
+ using var input = new MemoryStream(Sample());
+ var orders = new ExcelImporter()
+ .ExcelStreamToObject(input, options => options.OnMissingColumn = missing.Add).ToList();
+
+ Assert.AreEqual(2, orders.Count);
+ Assert.AreEqual(1, missing.Count);
+ Assert.AreEqual("金额", missing[0].Title);
+ CollectionAssert.AreEqual(new[] {"金额(元)"}, missing[0].SimilarTitles.ToArray());
+ }
+
+ [TestMethod]
+ public void TwoColumnsMissingAreReportedOnceEach()
+ {
+ var missing = new List();
+ var bytes = Workbook(new[] {"订单号"}, new[] {new[] {"00123"}, new[] {"00124"}, new[] {"00125"}});
+ new ExcelImporter().ExcelToObject(bytes, options => options.OnMissingColumn = missing.Add).ToList();
+
+ // 每个对不上的标题只上报一次,与行数无关
+ CollectionAssert.AreEquivalent(new[] {"金额", "备注"}, missing.Select(m => m.Title).ToArray());
+ Assert.AreEqual(0, missing[0].SimilarTitles.Count, "没有相近的就不要硬凑");
+ }
+
+ [TestMethod]
+ public void AnEmptySheetReportsEveryColumnAsMissing()
+ {
+ var workbook = new XSSFWorkbook();
+ workbook.CreateSheet("空表");
+ using var bytes = new MemoryStream();
+ workbook.Write(bytes, true);
+
+ var missing = new List();
+ var orders = new ExcelImporter()
+ .ExcelToObject(bytes.ToArray(), options => options.OnMissingColumn = missing.Add).ToList();
+
+ // 一行都没有,表头自然也没有:模型上的每个标题都对不上
+ Assert.AreEqual(0, orders.Count);
+ CollectionAssert.AreEquivalent(new[] {"订单号", "金额", "备注"}, missing.Select(m => m.Title).ToArray());
+ Assert.AreEqual("空表", missing[0].SheetTitle);
+ Assert.AreEqual(0, missing[0].HeaderTitles.Count);
+ }
+
+ [TestMethod]
+ public void TheMessageReadsWellWithAndWithoutASheetName()
+ {
+ var withSheet = new ExcelColumnMissing("金额", "Amount", "数据", new[] {"金额(元)"}, new[] {"金额(元)"});
+ StringAssert.StartsWith(withSheet.ToString(), "工作表 [数据] 的表头中没有 [金额]");
+
+ var withoutSheet = new ExcelColumnMissing("金额", "Amount", null, new string[0], new string[0]);
+ StringAssert.StartsWith(withoutSheet.ToString(), "表头中没有 [金额]");
+ }
+
+ [TestMethod]
+ public void TheHeaderCanBeReadOnItsOwn()
+ {
+ foreach (var excelType in new[] {ExcelType.Xlsx, ExcelType.Xls})
+ {
+ var bytes = new ExcelExporter().ObjectToExcelBytes(
+ new[] {new Order {No = "00123", Amount = 100.5m, Memo = "线上"}}, options =>
+ {
+ options.ExcelType = excelType;
+ options.SheetTitle = "本月";
+ })
+ ?? throw new AssertFailedException($"{excelType} 导出应有内容");
+
+ using var input = new MemoryStream(bytes);
+ var header = ExcelHelper.ReadHeader(input);
+ Assert.AreEqual("本月", header.SheetTitle, excelType.ToString());
+ CollectionAssert.AreEqual(new[] {"订单号", "金额", "备注"}, header.Columns.ToArray(), excelType.ToString());
+ }
+ }
+
+ [TestMethod]
+ public void TheHeaderHonoursSheetTitleAndSkippedLines()
+ {
+ var bytes = Workbook(new[] {"订单号", "金额"}, new[] {new[] {"00123", "1"}}, "上月");
+ var workbook = new XSSFWorkbook(new MemoryStream(bytes));
+ var second = workbook.CreateSheet("本月");
+ second.CreateRow(0).CreateCell(0).SetCellValue("导出说明");
+ var header = second.CreateRow(1);
+ header.CreateCell(0).SetCellValue("城市");
+ header.CreateCell(1).SetCellValue("数量");
+ using var both = new MemoryStream();
+ workbook.Write(both, true);
+
+ using var input = new MemoryStream(both.ToArray());
+ var read = ExcelHelper.ReadHeader(input, options =>
+ {
+ options.SheetTitle = "本月";
+ options.TitleSkipLine = 1;
+ });
+
+ Assert.AreEqual("本月", read.SheetTitle);
+ CollectionAssert.AreEqual(new[] {"城市", "数量"}, read.Columns.ToArray());
+ }
+
+ [TestMethod]
+ public void AnEmptySheetStillYieldsItsName()
+ {
+ var workbook = new XSSFWorkbook();
+ workbook.CreateSheet("空表");
+ using var bytes = new MemoryStream();
+ workbook.Write(bytes, true);
+
+ using var input = new MemoryStream(bytes.ToArray());
+ var header = ExcelHelper.ReadHeader(input);
+ Assert.AreEqual("空表", header.SheetTitle);
+ Assert.AreEqual(0, header.Columns.Count);
+ }
+}
diff --git a/Chsword.Excel2Object/Chsword.Excel2Object.csproj b/Chsword.Excel2Object/Chsword.Excel2Object.csproj
index 12ac32e..9b2dec1 100644
--- a/Chsword.Excel2Object/Chsword.Excel2Object.csproj
+++ b/Chsword.Excel2Object/Chsword.Excel2Object.csproj
@@ -15,7 +15,7 @@
true
Chsword.Excel2Object Library
Zou Jian
- 2.11.0
+ 2.12.0
Copyright ? 2014-2025
https://github.com/chsword/Excel2Object/
README.md
diff --git a/Chsword.Excel2Object/ExcelHelper.cs b/Chsword.Excel2Object/ExcelHelper.cs
index 074bbdb..12470ed 100644
--- a/Chsword.Excel2Object/ExcelHelper.cs
+++ b/Chsword.Excel2Object/ExcelHelper.cs
@@ -111,6 +111,12 @@ public static void ObjectToExcel(IEnumerable data, string path,
return excelExporter.ObjectToExcelBytes(data, optionsAction);
}
+ ///
+ public static ExcelSheetHeader ReadHeader(Stream input, Action? optionAction = null)
+ {
+ return new ExcelImporter().ReadHeader(input, optionAction);
+ }
+
///
public static IEnumerable ExcelStreamToObject(Stream input,
Action? optionAction = null)
diff --git a/Chsword.Excel2Object/ExcelImporter.cs b/Chsword.Excel2Object/ExcelImporter.cs
index e78bc9b..aea4bff 100644
--- a/Chsword.Excel2Object/ExcelImporter.cs
+++ b/Chsword.Excel2Object/ExcelImporter.cs
@@ -47,8 +47,8 @@ public IEnumerable ExcelToObject(byte[] bytes,
var options = new ExcelImporterOptions();
optionAction?.Invoke(options);
var context = new ImportContext(options);
- var rows = GetDataRows(bytes, options, context);
- return ToModels(rows, context);
+ var source = GetDataRows(bytes, options, context);
+ return ToModels(source == null ? null : AtHeader(source, options), context, source?.Title);
}
public IEnumerable ExcelToObject(byte[] bytes, string? sheetTitle)
@@ -95,10 +95,47 @@ public IEnumerable ExcelStreamToObject(Stream input,
var options = new ExcelImporterOptions();
optionAction?.Invoke(options);
var context = new ImportContext(options);
- var rows = XlsxRowReader.Rows(input, options, context).GetEnumerator();
- rows.MoveNext();
- for (var i = 0; i < options.TitleSkipLine; i++) rows.MoveNext();
- return ToModels(rows, context);
+ var source = XlsxRowReader.Rows(input, options, context);
+ return ToModels(AtHeader(source, options), context, source.Title);
+ }
+
+ ///
+ /// 只读出表头:表名与各列的标题,按列的先后。用于在导入之前核对列,或据表头生成模型。
+ ///
+ ///
+ /// 只读到表头那一行为止,后面有多少行数据都不影响其开销。.xls 仍须整份读入——该格式
+ /// 的数据并非顺序存放。传入的流由本方法读取,返回前即已读完。
+ /// 表里一行都没有时给出表名与空的列;传入的根本不是工作簿则抛出
+ /// 。
+ ///
+ ///
+ ///
+ /// using var file = File.OpenRead("orders.xlsx");
+ /// var header = ExcelHelper.ReadHeader(file);
+ /// Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}");
+ ///
+ ///
+ public ExcelSheetHeader ReadHeader(Stream input, Action? optionAction = null)
+ {
+ var options = new ExcelImporterOptions();
+ optionAction?.Invoke(options);
+ var context = new ImportContext(options);
+
+ var source = LooksLikeXlsx(input)
+ ? XlsxRowReader.Rows(input, options, context)
+ : GetDataRows(ReadAll(input), options, context);
+
+ // 读不出来与「表里没有行」不同:后者给出表名与空列,前者应当说清楚
+ if (source == null) throw new Excel2ObjectException("这不是一份能够打开的工作簿。");
+
+ using var rows = AtHeader(source, options);
+ var titleRow = rows.Current;
+ var columns = new List();
+ if (titleRow != null)
+ foreach (var cell in titleRow.Cells)
+ columns.Add(TextOf(cell.Value) ?? string.Empty);
+
+ return new ExcelSheetHeader(source.Title, columns);
}
///
@@ -124,13 +161,14 @@ private static byte[] ReadAll(Stream input)
}
/// 行从哪里来并不影响其后的转换:字典与模型两条路都只认 。
- private static IEnumerable ToModels(IEnumerator? rows, ImportContext context)
+ private static IEnumerable ToModels(IEnumerator? rows, ImportContext context,
+ string? sheetTitle)
where TModel : class, new()
{
if (typeof(TModel) == typeof(Dictionary))
return (InternalExcelToDictionary(rows, context) as IEnumerable)!;
- return InternalExcelToObject(rows, context);
+ return InternalExcelToObject(rows, context, sheetTitle);
}
private static IEnumerable> InternalExcelToDictionary(IEnumerator? result,
@@ -169,7 +207,7 @@ private static IEnumerable> InternalExcelToDictionary
}
private static IEnumerable InternalExcelToObject(IEnumerator? result,
- ImportContext context)
+ ImportContext context, string? sheetTitle)
where TModel : class, new()
{
if (result == null)
@@ -178,7 +216,7 @@ private static IEnumerable InternalExcelToObject(IEnumerator(result);
+ var dictColumns = BuildColumnMappings(result, context, sheetTitle);
while (result.MoveNext())
{
@@ -195,22 +233,32 @@ private static IEnumerable InternalExcelToObject(IEnumerator> BuildColumnMappings(
- IEnumerator result)
+ IEnumerator result, ImportContext context, string? sheetTitle)
where TModel : class, new()
{
var dict = ExcelUtil.GetPropertiesAttributesDict();
var dictColumns = new Dictionary>();
var titleRow = result.Current;
+ // 表里一行都没有时表头即为空,此时模型上的每个标题都对不上,同样要上报
+ var headerTitles = new List();
if (titleRow != null)
foreach (var cell in titleRow.Cells)
{
- var title = TextOf(cell.Value);
+ var title = TextOf(cell.Value) ?? string.Empty;
+ headerTitles.Add(title);
var prop = dict.FirstOrDefault(c => title == c.Value.Title);
if (prop.Key != null && !dictColumns.ContainsKey(cell.Key))
dictColumns.Add(cell.Key, prop);
}
+ // 模型上写着、表头里却没有的标题:那一列不会被填上,整列都是默认值,此处上报
+ var mapped = new HashSet(dictColumns.Values.Select(c => c.Value.Title), StringComparer.Ordinal);
+ foreach (var pair in dict)
+ if (!mapped.Contains(pair.Value.Title))
+ context.ReportMissingColumn(pair.Value.Title, pair.Key.Name, titleRow?.SheetTitle ?? sheetTitle,
+ headerTitles);
+
return dictColumns;
}
@@ -395,7 +443,7 @@ private static void PopulateModelFromRow(TModel model, IImportRow row,
return date.Value.ToString(pattern, CultureInfo.InvariantCulture);
}
- private static IEnumerator? GetDataRows(byte[]? bytes, ExcelImporterOptions options,
+ private static SheetSource? GetDataRows(byte[]? bytes, ExcelImporterOptions options,
ImportContext context)
{
if (bytes == null || bytes.Length == 0)
@@ -423,7 +471,13 @@ private static void PopulateModelFromRow(TModel model, IImportRow row,
throw new Excel2ObjectException($"The specified sheet:[{options.SheetTitle}] does not exist");
}
- var rows = NpoiRows(sheet, context).GetEnumerator();
+ return new SheetSource(sheet.SheetName, NpoiRows(sheet, context));
+ }
+
+ /// 取到停在表头那一行的枚举器:表头之上还可以有若干行说明文字。
+ private static IEnumerator AtHeader(SheetSource source, ExcelImporterOptions options)
+ {
+ var rows = source.Rows.GetEnumerator();
rows.MoveNext();
for (var i = 0; i < options.TitleSkipLine; i++) rows.MoveNext();
return rows;
diff --git a/Chsword.Excel2Object/Internal/ImportContext.cs b/Chsword.Excel2Object/Internal/ImportContext.cs
index 602c9a8..58d711e 100644
--- a/Chsword.Excel2Object/Internal/ImportContext.cs
+++ b/Chsword.Excel2Object/Internal/ImportContext.cs
@@ -61,6 +61,30 @@ public void Report(IRow? row, int columnIndex, Exception exception)
Report(row?.Sheet?.SheetName, row?.RowNum ?? -1, columnIndex, exception);
}
+ ///
+ /// 上报模型上的某个标题在表头中找不到。未设置回调时不作任何输出;回调抛出的异常照旧向外
+ /// 传播,调用方据此即可拒绝这样的文件。
+ ///
+ public void ReportMissingColumn(string title, string propertyName, string? sheetTitle,
+ IReadOnlyList headerTitles)
+ {
+ if (_options.OnMissingColumn == null) return;
+
+ // 相近只按「一方包含另一方」判定:「金额」与「金额(元)」够用,再多的猜测不如不猜
+ var similar = new List();
+ foreach (var header in headerTitles)
+ {
+ var one = header.Trim();
+ var other = title.Trim();
+ if (one.Length == 0 || other.Length == 0 || one == other) continue;
+ if (one.IndexOf(other, StringComparison.OrdinalIgnoreCase) >= 0 ||
+ other.IndexOf(one, StringComparison.OrdinalIgnoreCase) >= 0)
+ similar.Add(header);
+ }
+
+ _options.OnMissingColumn(new ExcelColumnMissing(title, propertyName, sheetTitle, headerTitles, similar));
+ }
+
/// 位置由调用方给出:流式读取没有单元格对象,只有行列号。
public void Report(string? sheetTitle, int rowIndex, int columnIndex, Exception exception)
{
diff --git a/Chsword.Excel2Object/Internal/ImportRow.cs b/Chsword.Excel2Object/Internal/ImportRow.cs
index 2690b47..9feb71a 100644
--- a/Chsword.Excel2Object/Internal/ImportRow.cs
+++ b/Chsword.Excel2Object/Internal/ImportRow.cs
@@ -21,6 +21,20 @@ internal interface IImportRow
CellData Cell(int columnIndex);
}
+/// 一张工作表的来源:它的名字,以及各行。
+internal sealed class SheetSource
+{
+ public SheetSource(string? title, IEnumerable rows)
+ {
+ Title = title;
+ Rows = rows;
+ }
+
+ public string? Title { get; }
+
+ public IEnumerable Rows { get; }
+}
+
/// 整份读入内存时的一行,取值经由 NPOI 的单元格。
internal sealed class NpoiRow : IImportRow
{
diff --git a/Chsword.Excel2Object/Internal/XlsxRowReader.cs b/Chsword.Excel2Object/Internal/XlsxRowReader.cs
index 7f0a677..bab69bc 100644
--- a/Chsword.Excel2Object/Internal/XlsxRowReader.cs
+++ b/Chsword.Excel2Object/Internal/XlsxRowReader.cs
@@ -26,7 +26,7 @@ internal static class XlsxRowReader
private const string RelationshipNamespace =
"http://schemas.openxmlformats.org/officeDocument/2006/relationships";
- public static IEnumerable Rows(Stream input, ExcelImporterOptions options, ImportContext context)
+ public static SheetSource Rows(Stream input, ExcelImporterOptions options, ImportContext context)
{
OPCPackage package;
try
@@ -44,7 +44,7 @@ public static IEnumerable Rows(Stream input, ExcelImporterOptions op
var reader = new XSSFReader(package);
var workbook = ReadWorkbook(reader);
var sheet = Locate(workbook, options.SheetTitle);
- return Iterate(package, reader, sheet, workbook.Date1904, context);
+ return new SheetSource(sheet.Title, Iterate(package, reader, sheet, workbook.Date1904, context));
}
catch
{
diff --git a/Chsword.Excel2Object/Options/ExcelColumnMissing.cs b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs
new file mode 100644
index 0000000..0692bf9
--- /dev/null
+++ b/Chsword.Excel2Object/Options/ExcelColumnMissing.cs
@@ -0,0 +1,47 @@
+namespace Chsword.Excel2Object.Options;
+
+///
+/// 模型上写着的某个列标题,在表头中找不到。该属性因而不会被填上,整列都是默认值。
+///
+///
+/// 这是导入类问题中最常见的一种:标题差了一个空格、多了个单位(「金额」与「金额(元)」),
+/// 导入不会报错,只是那一列悄悄全为空。 即为
+/// 此而设。
+///
+public class ExcelColumnMissing
+{
+ public ExcelColumnMissing(string title, string propertyName, string? sheetTitle,
+ IReadOnlyList headerTitles, IReadOnlyList similarTitles)
+ {
+ Title = title;
+ PropertyName = propertyName;
+ SheetTitle = sheetTitle;
+ HeaderTitles = headerTitles;
+ SimilarTitles = similarTitles;
+ }
+
+ /// 模型上写着的标题。
+ public string Title { get; }
+
+ /// 要它的那个属性。
+ public string PropertyName { get; }
+
+ public string? SheetTitle { get; }
+
+ /// 表头上实际有的标题,按列的先后。
+ public IReadOnlyList HeaderTitles { get; }
+
+ ///
+ /// 表头上与之相近的标题:一方包含另一方即算(「金额」与「金额(元)」),不作更多猜测。
+ ///
+ public IReadOnlyList SimilarTitles { get; }
+
+ public override string ToString()
+ {
+ var where = SheetTitle == null ? "" : $"工作表 [{SheetTitle}] 的";
+ var similar = SimilarTitles.Count == 0
+ ? ""
+ : $",表头上与之相近的是 [{string.Join("]、[", SimilarTitles)}]";
+ return $"{where}表头中没有 [{Title}] 这一列(属性 {PropertyName} 因而不会被填上){similar}。";
+ }
+}
diff --git a/Chsword.Excel2Object/Options/ExcelImporterOptions.cs b/Chsword.Excel2Object/Options/ExcelImporterOptions.cs
index 6ed3051..323d2f3 100644
--- a/Chsword.Excel2Object/Options/ExcelImporterOptions.cs
+++ b/Chsword.Excel2Object/Options/ExcelImporterOptions.cs
@@ -27,4 +27,27 @@ public class ExcelImporterOptions
/// options.OnCellError = e => throw e.Exception;。
///
public Action? OnCellError { get; set; }
+
+ ///
+ /// 模型上写着的某个列标题在表头中找不到时的回调,用于得知哪一列没有对上。默认不设回调,
+ /// 此时该属性保持其默认值,导入照常进行——与既有版本一致。
+ ///
+ ///
+ ///
+ /// 标题差一个空格、多一个单位(「金额」与「金额(元)」),导入并不会报错,只是那一列
+ /// 悄悄全为空。这是导入类问题中最常见的一种, 只覆盖到单元格
+ /// 层面,对此无能为力。
+ ///
+ /// 回调在读过表头之后、取第一行数据之前调用,每个对不上的标题调用一次。
+ ///
+ ///
+ /// // 只是记下来
+ /// options.OnMissingColumn = missing => logger.Warn(missing.ToString());
+ ///
+ /// // 或者干脆不接受这样的文件
+ /// options.OnMissingColumn = missing => throw new Excel2ObjectException(missing.ToString());
+ ///
+ ///
+ ///
+ public Action? OnMissingColumn { get; set; }
}
diff --git a/Chsword.Excel2Object/Options/ExcelSheetHeader.cs b/Chsword.Excel2Object/Options/ExcelSheetHeader.cs
new file mode 100644
index 0000000..78fb1f4
--- /dev/null
+++ b/Chsword.Excel2Object/Options/ExcelSheetHeader.cs
@@ -0,0 +1,19 @@
+namespace Chsword.Excel2Object.Options;
+
+///
+/// 一张工作表的表头:表名与各列的标题,按列的先后。
+///
+public class ExcelSheetHeader
+{
+ public ExcelSheetHeader(string? sheetTitle, IReadOnlyList columns)
+ {
+ SheetTitle = sheetTitle;
+ Columns = columns;
+ }
+
+ /// 实际读的那张表的名字:未指定表名时即第一张表。
+ public string? SheetTitle { get; }
+
+ /// 表头上的各列标题,按列的先后;表中没有任何行时为空。
+ public IReadOnlyList Columns { get; }
+}
diff --git a/README.md b/README.md
index 8782876..ca923b6 100644
--- a/README.md
+++ b/README.md
@@ -68,6 +68,13 @@ excel2obj generate-model orders.xlsx --class Order # 由表头生成
### 发布说明
+* **2026.09.16** - v2.12.0
+- [x] ✨ **新增:** 模型上的列在表头中找不到时可以得知:`options.OnMissingColumn = m => logger.Warn(m.ToString())`。标题差一个空格、多一个单位(「金额」与「金额(元)」),导入此前不会报错,只是那一列悄悄全是默认值。回调在读过表头之后、取第一行之前调用,每个对不上的标题一次,并给出表头上与之相近的标题;在回调中抛出即可拒绝这样的文件。默认不设回调时行为与既有版本一致 - 查看 [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md)
+- [x] ✨ **新增:** 只读出表头:`ExcelHelper.ReadHeader(stream)` 给出表名与各列标题,只读到表头那一行为止,用于导入前核对列或据表头生成模型
+- [x] 🔧 命令行工具改为逐行读出:二十万行五列的文件,`convert` 由 13.5 秒 / 2412 MB 降到 2.6 秒 / 215 MB,`generate-model` 由 11.9 秒 / 2245 MB 降到 2.7 秒 / 204 MB,不含公式的文件输出与此前逐字节相同。公式格取的是文件里存着的结果,确需当场求值时加 `--whole`
+- [x] 🐛 修复命令行工具下表头带空白时取不到值:写出的属性名保留了空白,而值永远为空——表头与数据两边的标题此前不是同一个入口取的
+
+
* **2026.09.16** - v2.11.0
- [x] ✨ **新增:** 流式导入:`ExcelHelper.ExcelStreamToObject(stream)` 逐行读出 `.xlsx` 的一张工作表,不把整个工作簿建进内存。实测二十万行五列:整份读入峰值 1265 MB / 8.4 秒,流式导入 194 MB / 2.9 秒,两者读出的数据逐字段一致。取到的序列是惰性的,`Take` 一类的操作真的能少读。三处不同:公式格读的是文件中存着的上一次计算结果而非当场求值(本库导出的文件里公式没有这个结果,读作空白,与其他空白格一样)、只有 `.xlsx` 能逐行读出(`.xls` 照旧整份读入)、工作表在一开始就定位 - 查看 [docs/versions/v2.11.0.md](docs/versions/v2.11.0.md)
- [x] 🔧 导入的类型转换不再依赖 NPOI 的单元格:整份读入与逐行读出共用同一套转换,两条路的行为不会各自漂移。字典形式的导入随之改为惰性给出;标题行中同名的标题以最左一列为准,非文本的标题不再中断导入
diff --git a/README_EN.md b/README_EN.md
index dfc65af..a3366d3 100644
--- a/README_EN.md
+++ b/README_EN.md
@@ -66,6 +66,13 @@ See [Chsword.Excel2Object.Cli/README.md](Chsword.Excel2Object.Cli/README.md).
### Release Notes
+* **2026.09.16** - v2.12.0
+- [x] ✨ **NEW:** Learn when a column your model declares is not in the header: `options.OnMissingColumn = m => logger.Warn(m.ToString())`. A title off by a space or carrying a unit ("Amount" against "Amount (USD)") never failed the import before - that property was simply left at its default for every row. The callback runs after the header is read and before the first row, once per title that did not match, and names the header titles closest to it; throw from it to refuse the file. With no callback set the behaviour is unchanged - See [docs/versions/v2.12.0.md](docs/versions/v2.12.0.md)
+- [x] ✨ **NEW:** Read just the header: `ExcelHelper.ReadHeader(stream)` returns the sheet name and its column titles, reading no further than the header row - for checking columns before an import, or generating a model from them
+- [x] 🔧 The command-line tool now reads row by row: on 200,000 rows of five columns, `convert` went from 13.5 s / 2412 MB to 2.6 s / 215 MB and `generate-model` from 11.9 s / 2245 MB to 2.7 s / 204 MB, byte for byte the same output for files without formulas. A formula cell now reads the result stored in the file; pass `--whole` to read the workbook whole and evaluate formulas
+- [x] 🐛 Fixed the command-line tool losing values when a header title carries surrounding whitespace: the property it wrote kept the whitespace while its value was always empty, because the header and the rows did not go through the same reader
+
+
* **2026.09.16** - v2.11.0
- [x] ✨ **NEW:** Streaming import: `ExcelHelper.ExcelStreamToObject(stream)` reads one sheet of an `.xlsx` row by row instead of building the whole workbook in memory. Measured on 200,000 rows of five columns: 1265 MB / 8.4 s read whole against 194 MB / 2.9 s streamed, field for field the same data. The sequence is lazy, so `Take` and friends really do read less. Three differences: a formula cell reads the result stored in the file rather than being evaluated (files this library writes carry no such result, so those cells read as blank, like any other blank cell), only `.xlsx` can be read row by row (`.xls` is still read whole), and the sheet is located up front - See [docs/versions/v2.11.0.md](docs/versions/v2.11.0.md)
- [x] 🔧 Import conversion no longer depends on NPOI's cell objects: reading whole and reading row by row share one conversion, so the two cannot drift apart. The dictionary form of the import became lazy as a result; duplicate header titles now resolve to the leftmost column, and a non-text header no longer aborts the import
diff --git a/docs/README.md b/docs/README.md
index 0ed0cf1..48125ff 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -6,6 +6,7 @@
| 版本 | 发布日期 | 主要内容 |
| --- | --- | --- |
+| [v2.12.0](versions/v2.12.0.md) | 2026-09-16 | 导入前的列诊断与只读表头;命令行工具逐行读出 |
| [v2.11.0](versions/v2.11.0.md) | 2026-09-16 | 流式导入:逐行读出,不把整个工作簿建进内存 |
| [v2.10.0](versions/v2.10.0.md) | 2026-09-16 | 流式导出:边取边写,内存不随行数增长 |
| [v2.9.1](versions/v2.9.1.md) | 2026-09-15 | 合并单元格:按列合并连续相同值、指定区域;空引用检查改为强制 |
diff --git a/docs/versions/v2.12.0.md b/docs/versions/v2.12.0.md
new file mode 100644
index 0000000..d1c7f5c
--- /dev/null
+++ b/docs/versions/v2.12.0.md
@@ -0,0 +1,82 @@
+# v2.12.0
+
+发布日期:2026-09-16
+
+## 概述
+
+本版本补上导入前的两件事:模型上写着的列在表头里究竟有没有(`OnMissingColumn`),以及表头本身是什么(`ReadHeader`)。命令行工具也随之改为逐行读出,二十万行的文件转换由 2.4 GB 降到 215 MB。
+
+## 变更明细
+
+### 1. 新增:模型上的列在表头中找不到时可以得知
+
+标题差一个空格、多一个单位——「金额」与「金额(元)」——导入并不会报错,只是那一列悄悄全是默认值。这是导入类问题中最常见的一种,而 `OnCellError` 只覆盖到单元格层面,对此无能为力。
+
+```csharp
+var orders = new ExcelImporter().ExcelToObject(bytes, options =>
+ options.OnMissingColumn = missing => logger.Warn(missing.ToString()));
+
+// 工作表 [数据] 的表头中没有 [金额] 这一列(属性 Amount 因而不会被填上),
+// 表头上与之相近的是 [金额(元)]。
+```
+
+回调在读过表头之后、取第一行数据之前调用,每个对不上的标题调用一次,与行数无关。`ExcelColumnMissing` 给出模型上写的标题、要它的属性名、表名、表头上实际有的全部标题,以及其中与之相近的几个——相近只按「一方包含另一方」判定,不作更多猜测。
+
+默认不设回调,此时行为与既有版本完全一致:该属性保持默认值,导入照常。若希望干脆不接受这样的文件,在回调中抛出即可:
+
+```csharp
+options.OnMissingColumn = missing => throw new Excel2ObjectException(missing.ToString());
+```
+
+流式导入同样适用。
+
+### 2. 新增:只读出表头
+
+```csharp
+using var file = File.OpenRead("orders.xlsx");
+var header = ExcelHelper.ReadHeader(file);
+Console.WriteLine($"{header.SheetTitle}:{string.Join("、", header.Columns)}");
+```
+
+只读到表头那一行为止,后面有多少行数据都不影响这一步的开销。用于在导入之前核对列、按表头生成模型,或让使用者自行把表里的列对到模型上。`SheetTitle` 是实际读的那张表的名字——未指定表名时即第一张表;`SheetTitle` 与 `TitleSkipLine` 两个选项照常生效。`.xls` 仍须整份读入,该格式的数据并非顺序存放。
+
+### 3. 改进:命令行工具逐行读出
+
+`excel2obj` 此前每条命令都把整个工作簿读进内存,且表头还要再打开一次文件才能读到。现改为:表头由 `ReadHeader` 单独读出,数据走流式导入;类型推断(`--typed` 与 `generate-model`)也改成边读边推断,不再把整列的值攒起来。
+
+二十万行五列的文件,本机实测:
+
+| 命令 | 此前 | 现在 |
+| --- | --- | --- |
+| `convert` | 13.5 秒 / 2412 MB | 2.6 秒 / 215 MB |
+| `convert --typed` | 12.9 秒 / 2428 MB | 4.1 秒 / 329 MB |
+| `generate-model` | 11.9 秒 / 2245 MB | 2.7 秒 / 204 MB |
+
+不含公式的文件,三条命令的输出与此前逐字节相同。`--typed` 要先知道每列是什么类型,故读两遍文件:一遍推断,一遍写出;写 JSON 也改为边读边写,不再先在内存里搭出整棵 JSON 树。
+
+**公式格有一处不同**:逐行读出取的是文件里存着的上一次计算结果,而非当场求值。Excel 存盘时会写下这个结果,故 Excel 保存过的文件照常;本库导出的文件里公式没有这个结果,此时该列读作空白。确需求值时加 `--whole`,整份读入工作簿:
+
+```bash
+excel2obj convert orders.xlsx --whole # 公式当场求值,内存随文件增长
+excel2obj generate-model orders.xlsx --whole
+```
+
+写出文件时先写到同目录下的临时文件,成功之后再就位:`--output` 指向输入本身时源文件不会在读到之前被清空,中途失败也不会留下半个文件。
+
+### 4. 修复:命令行工具下表头带空白时取不到值
+
+表头写成 订单号 (两侧有空格)时,`excel2obj convert` 会写出一个名为 " 订单号 " 的属性,而它的值永远是空的——导入给出的行以去掉空白的标题为键,而表头是照原样取的,两边对不上。现在两处都经由同一个入口取得,不再有这个出入:
+
+```json
+// 此前
+[ { " 订单号 ": "", "金额": "100.5" } ]
+// 现在
+[ { "订单号": "00123", "金额": "100.5" } ]
+```
+
+## 注意事项
+
+- 表里一行都没有时,表头自然也没有,模型上的每个标题都会被上报一次。
+- `ReadHeader` 在传入的根本不是工作簿时抛出 `Excel2ObjectException`;这与「表里没有行」不同,后者给出表名与空的列。
+- `OnMissingColumn` 只在按模型导入时有意义:导入为 `Dictionary` 时列由表头决定,无所谓对不对得上。
+- 表头中重复的标题以最左边那一列为准(v2.11.0 起),因此「某个标题存在但对到了另一列」不属于本次上报的范围。