Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
132 changes: 92 additions & 40 deletions Chsword.Excel2Object.Cli/ConvertCommand.cs
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
using System.Data;
using System.Globalization;
using System.Text;
using System.Text.Encodings.Web;
using System.Text.Json;
using System.Text.Json.Nodes;
Expand All @@ -9,12 +10,6 @@ namespace Chsword.Excel2Object.Cli;
/// <summary>excel2obj convert: Excel -> JSON or JSON -> Excel, decided per input by file extension.</summary>
public static class ConvertCommand
{
private static readonly JsonSerializerOptions JsonOptions = new()
{
WriteIndented = true,
Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping
};

public static int Run(Arguments args, TextWriter output, TextWriter error)
{
if (args.Positional.Count == 0) throw new UsageException("convert needs at least one input file");
Expand All @@ -30,29 +25,22 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
Directory.CreateDirectory(target);
}

var whole = args.Has("whole");

foreach (var input in inputs)
{
if (!File.Exists(input)) throw new FileNotFoundException($"input file not found: {input}", input);
string? destination;
if (inputs.Count > 1)
destination = Path.Combine(target!, Path.GetFileNameWithoutExtension(input) +
(SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx"));
else if (target != null && Directory.Exists(target))
destination = Path.Combine(target, Path.GetFileNameWithoutExtension(input) +
(SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx"));
else
destination = target;
var destination = Destination(input, target, inputs.Count > 1, xls);

if (SheetData.IsExcel(input))
{
var json = ExcelToJson(input, sheet, typed);
if (destination == null)
{
output.WriteLine(json);
output.WriteLine(ExcelToJson(input, sheet, typed, whole));
}
else
{
File.WriteAllText(destination, json);
Replace(destination, file => WriteJson(input, sheet, typed, whole, file));
error.WriteLine($"{input} -> {destination}");
}
}
Expand All @@ -63,56 +51,120 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
var excelType = xls || destination.EndsWith(".xls", StringComparison.OrdinalIgnoreCase)
? ExcelType.Xls
: ExcelType.Xlsx;
File.WriteAllBytes(destination, JsonToExcel(input, sheet, excelType));
var bytes = JsonToExcel(input, sheet, excelType);
Replace(destination, file => file.Write(bytes, 0, bytes.Length));
error.WriteLine($"{input} -> {destination}");
}
}

return Excel2ObjCli.Ok;
}

public static string ExcelToJson(string path, string? sheet, bool typed)
/// <summary>该输入写到哪里去:多个输入时按目录派生文件名,单个输入时即 --output 本身。</summary>
private static string? Destination(string input, string? target, bool severalInputs, bool xls)
{
var data = SheetData.Load(path, sheet);
var types = typed
? data.Columns.ToDictionary(c => c, c => TypeInference.Infer(data.ColumnValues(c)))
: null;
var name = Path.GetFileNameWithoutExtension(input) +
(SheetData.IsExcel(input) ? ".json" : xls ? ".xls" : ".xlsx");
if (severalInputs) return Path.Combine(target!, name);
if (target != null && Directory.Exists(target)) return Path.Combine(target, name);
return target;
}

var array = new JsonArray();
foreach (var row in data.Rows)
/// <summary>
/// 先写到同目录下的临时文件,成功之后再就位。读的是一行一行来的,若直接往目标文件写:
/// 目标与输入是同一个文件时,源在读到之前就已被清空;中途失败也会留下半个文件。
/// </summary>
private static void Replace(string destination, Action<Stream> write)
{
var temporary = destination + ".tmp" + Path.GetRandomFileName();
try
{
using (var file = File.Create(temporary)) write(file);

if (File.Exists(destination)) File.Delete(destination);
File.Move(temporary, destination);
}
catch
{
var item = new JsonObject();
foreach (var column in data.Columns)
if (File.Exists(temporary)) File.Delete(temporary);
throw;
}
}

/// <summary>写到标准输出时才用得到:那里本就要把整段文本拿在手上。</summary>
public static string ExcelToJson(string path, string? sheet, bool typed, bool whole = false)
{
using var buffer = new MemoryStream();
WriteJson(path, sheet, typed, whole, buffer);
return Encoding.UTF8.GetString(buffer.ToArray());
}

/// <summary>
/// 一行读出、一行写出,中途不把整份数据攒在内存里。<c>--typed</c> 要先知道每列是什么类型,
/// 故先过一遍推断,再过一遍写出——两遍各是一次顺序读。
/// </summary>
/// <param name="whole">
/// 整份读入工作簿,公式当场求值。逐行读出取的是文件里存着的上一次计算结果,没有存下结果的
/// 公式(本库导出的文件即如此)因而读作空白;确需求值时用这条路,代价是内存随文件增长。
/// </param>
public static void WriteJson(string path, string? sheet, bool typed, bool whole, Stream destination)
{
var data = SheetData.Load(path, sheet, whole);
// 每列的类型在此定下,不在写每一格时反复去问
var types = typed ? data.Infer().Select(inference => inference.Result).ToArray() : null;

using var writer = new Utf8JsonWriter(destination,
new JsonWriterOptions {Indented = true, Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping});
writer.WriteStartArray();
foreach (var row in data.Rows())
{
writer.WriteStartObject();
for (var i = 0; i < data.Columns.Count; i++)
{
var text = row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "";
item[column] = types == null ? JsonValue.Create(text) : ToJsonValue(text, types[column]);
writer.WritePropertyName(data.Columns[i]);
var text = SheetData.Text(row, data.Columns[i]);
if (types == null)
writer.WriteStringValue(text);
else
WriteTypedValue(writer, text, types[i]);
}

array.Add(item);
writer.WriteEndObject();
}

return array.ToJsonString(JsonOptions);
writer.WriteEndArray();
}

private static JsonNode? ToJsonValue(string text, InferredType type)
private static void WriteTypedValue(Utf8JsonWriter writer, string text, InferredType type)
{
if (type != InferredType.String && string.IsNullOrWhiteSpace(text)) return null;
if (type != InferredType.String && string.IsNullOrWhiteSpace(text))
{
writer.WriteNullValue();
return;
}

switch (type)
{
case InferredType.Bool:
TypeInference.TryParseBool(text, out var flag);
return JsonValue.Create(flag);
writer.WriteBooleanValue(flag);
break;
case InferredType.Int:
return JsonValue.Create(int.Parse(text, CultureInfo.InvariantCulture));
writer.WriteNumberValue(int.Parse(text, CultureInfo.InvariantCulture));
break;
case InferredType.Long:
return JsonValue.Create(long.Parse(text, CultureInfo.InvariantCulture));
writer.WriteNumberValue(long.Parse(text, CultureInfo.InvariantCulture));
break;
case InferredType.Decimal:
return JsonValue.Create(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture));
writer.WriteNumberValue(decimal.Parse(text, NumberStyles.Float, CultureInfo.InvariantCulture));
break;
case InferredType.DateTime:
TypeInference.TryParseDateTime(text, out var date);
return JsonValue.Create(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture));
writer.WriteStringValue(date.ToString("yyyy-MM-ddTHH:mm:ss", CultureInfo.InvariantCulture));
break;
default:
return JsonValue.Create(text);
writer.WriteStringValue(text);
break;
}
}

Expand Down
2 changes: 1 addition & 1 deletion Chsword.Excel2Object.Cli/Excel2ObjCli.cs
Original file line number Diff line number Diff line change
Expand Up @@ -130,7 +130,7 @@ public static Arguments Parse(IEnumerable<string> args)
}

/// <summary>Options that never take a value, so a following positional argument is not swallowed.</summary>
private static readonly HashSet<string> Switches = new(StringComparer.Ordinal) {"typed", "xls"};
private static readonly HashSet<string> Switches = new(StringComparer.Ordinal) {"typed", "xls", "whole"};

public bool Has(string name)
{
Expand Down
10 changes: 6 additions & 4 deletions Chsword.Excel2Object.Cli/GenerateModelCommand.cs
Original file line number Diff line number Diff line change
Expand Up @@ -11,7 +11,7 @@ public static int Run(Arguments args, TextWriter output, TextWriter error)
if (args.Positional.Count != 1) throw new UsageException("generate-model needs exactly one input file");
var input = args.Positional[0];
var sheet = args.Get("sheet");
var data = SheetData.Load(input, sheet);
var data = SheetData.Load(input, sheet, args.Has("whole"));
var className = args.Get("class") ?? ToIdentifier(data.SheetTitle, "Model");
var code = Generate(data, className, args.Get("namespace"));

Expand Down Expand Up @@ -44,12 +44,14 @@ public static string Generate(SheetData data, string className, string? ns)
sb.Append("public class ").AppendLine(className);
sb.AppendLine("{");
var used = new HashSet<string>(StringComparer.Ordinal) {className};
var inferences = data.Infer();
for (var i = 0; i < data.Columns.Count; i++)
{
var title = data.Columns[i];
var values = data.ColumnValues(title).ToList();
var type = TypeInference.Infer(values);
var nullable = values.Count == 0 || values.Any(string.IsNullOrWhiteSpace);
var inference = inferences[i];
var type = inference.Result;
// 一行都没有,或出现过空值,该属性即为可空
var nullable = !inference.Any || inference.HasBlank;
var name = Unique(ToIdentifier(title, $"Column{i + 1}"), used);

if (i > 0) sb.AppendLine();
Expand Down
13 changes: 10 additions & 3 deletions Chsword.Excel2Object.Cli/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,11 +14,18 @@ excel2obj convert orders.xlsx # prints a JSON array to std
excel2obj convert orders.xlsx --output orders.json # writes the file
excel2obj convert orders.xlsx --sheet Orders --typed
excel2obj convert a.xlsx b.xls --output ./json/ # several inputs need an output directory
excel2obj convert orders.xlsx --whole # read the workbook whole and evaluate formulas
```

Each row becomes an object keyed by the header row. Values are the cell text (formulas are evaluated); with
`--typed`, a column whose non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as
JSON numbers, booleans or `yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`.
Each row becomes an object keyed by the header row. Values are the cell text; with `--typed`, a column whose
non-empty values are all integers, decimals, `TRUE`/`FALSE` or ISO dates is emitted as JSON numbers, booleans or
`yyyy-MM-ddTHH:mm:ss` strings, and empty cells become `null`.

The sheet is read row by row, so a file far larger than memory converts fine. One consequence: a **formula cell
reads the result stored in the file** rather than being evaluated. Excel writes that result when it saves, so
files saved by Excel read as before; a file written by this library carries formulas with no stored result, and
those cells read as blank. Pass `--whole` to read the workbook whole and evaluate formulas instead - the memory
it needs then grows with the file. `--whole` works with `generate-model` as well.

## JSON -> Excel

Expand Down
80 changes: 57 additions & 23 deletions Chsword.Excel2Object.Cli/SheetData.cs
Original file line number Diff line number Diff line change
@@ -1,46 +1,80 @@
using NPOI.SS.UserModel;
using Chsword.Excel2Object.Options;

namespace Chsword.Excel2Object.Cli;

/// <summary>A sheet as the importer sees it: the header titles in column order and one string per cell.</summary>
/// <summary>
/// A sheet as the importer sees it: the header titles in column order, and its rows on demand.
/// </summary>
/// <remarks>
/// 行不再一次性读进内存:<see cref="Rows" /> 每次遍历都逐行读出,命令行因而能处理远大于内存的
/// 文件。需要过两遍数据的地方(<c>--typed</c> 先推断类型再写出)就遍历两次,各自的开销是一遍
/// 顺序读。
/// </remarks>
public sealed class SheetData
{
public SheetData(string sheetTitle, List<string> columns, List<Dictionary<string, object>> rows)
private readonly string _path;
private readonly string? _sheetTitle;
private readonly bool _whole;

private SheetData(string path, string? sheetTitle, bool whole, string title, List<string> columns)
{
SheetTitle = sheetTitle;
_path = path;
_sheetTitle = sheetTitle;
_whole = whole;
SheetTitle = title;
Columns = columns;
Rows = rows;
}

public string SheetTitle { get; }

public List<string> Columns { get; }
public List<Dictionary<string, object>> Rows { get; }

public static SheetData Load(string path, string? sheetTitle)
/// <param name="whole">
/// 整份读入工作簿,公式当场求值;否则逐行读出,公式取文件里存着的上一次计算结果。
/// </param>
public static SheetData Load(string path, string? sheetTitle, bool whole = false)
{
if (!File.Exists(path)) throw new FileNotFoundException($"input file not found: {path}", path);
var bytes = File.ReadAllBytes(path);
var rows = ExcelHelper.ExcelToObject<Dictionary<string, object>>(bytes, sheetTitle).ToList();
var (title, columns) = ReadHeader(bytes, sheetTitle);
return new SheetData(title, columns, rows);

// 只读到表头那一行为止,后面有多少行数据都不影响这一步的开销
using var input = File.OpenRead(path);
var header = ExcelHelper.ReadHeader(input, options => options.SheetTitle = sheetTitle);
return new SheetData(path, sheetTitle, whole, header.SheetTitle ?? "", header.Columns.ToList());
}

/// <summary>The header row straight from the workbook, so an empty sheet still yields its columns.</summary>
private static (string title, List<string> columns) ReadHeader(byte[] bytes, string? sheetTitle)
/// <summary>读出该表的各行。每次遍历都重新读一遍文件。</summary>
public IEnumerable<Dictionary<string, object>> Rows()
{
using var stream = new MemoryStream(bytes);
var workbook = WorkbookFactory.Create(stream);
var sheet = string.IsNullOrEmpty(sheetTitle) ? workbook.GetSheetAt(0) : workbook.GetSheet(sheetTitle);
if (sheet == null) throw new Excel2ObjectException($"The specified sheet:[{sheetTitle}] does not exist");
var header = sheet.GetRow(sheet.FirstRowNum);
// untrimmed on purpose: the importer keys rows by the exact header text
var columns = header?.Cells.Select(cell => cell.ToString() ?? "").ToList() ?? new List<string>();
return (sheet.SheetName, columns);
if (_whole)
return ExcelHelper.ExcelToObject<Dictionary<string, object>>(File.ReadAllBytes(_path), _sheetTitle);

return Streamed();
}

private IEnumerable<Dictionary<string, object>> Streamed()
{
using var input = File.OpenRead(_path);
foreach (var row in ExcelHelper.ExcelStreamToObject<Dictionary<string, object>>(input,
options => options.SheetTitle = _sheetTitle))
Comment on lines +57 to +58
yield return row;
}

/// <summary>
/// 各列的类型推断,一遍读完,按列的先后存放——表头允许有重名的列,故不以标题为键。
/// </summary>
public TypeInference.Inference[] Infer()
{
var inferences = Columns.Select(_ => new TypeInference.Inference()).ToArray();
foreach (var row in Rows())
for (var i = 0; i < Columns.Count; i++)
inferences[i].Observe(Text(row, Columns[i]));

return inferences;
}

public IEnumerable<string> ColumnValues(string column)
public static string Text(IReadOnlyDictionary<string, object> row, string column)
{
return Rows.Select(row => row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "");
return row.TryGetValue(column, out var value) ? value?.ToString() ?? "" : "";
}

public static bool IsExcel(string path)
Expand Down
Loading