From b8661ced0a9e5f59628cf756f4a7936b6350e366 Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Mon, 13 Jul 2026 19:16:03 -0700 Subject: [PATCH 1/7] feat: add JSON lines, Arrow IPC stream, and Parquet output formats --- Cargo.lock | 135 ++++++++++++++++++++++++++++++++++++ Cargo.toml | 1 + README.md | 5 +- docs/index.md | 2 +- docs/reference.md | 13 ++-- docs/tutorial.md | 2 +- src/output.rs | 169 +++++++++++++++++++++++++++++++++++++++++++++- 7 files changed, 317 insertions(+), 10 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index d2f6993..9b2d440 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -70,6 +70,21 @@ dependencies = [ "memchr", ] +[[package]] +name = "alloc-no-stdlib" +version = "2.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cc7bb162ec39d46ab1ca8c77bf72e890535becd1751bb45f64c597edb4c8c6b3" + +[[package]] +name = "alloc-stdlib" +version = "0.2.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e76a019e91224d279006ff972f1e984179a6e9feb050adba6ce8274aef23195" +dependencies = [ + "alloc-no-stdlib", +] + [[package]] name = "android_system_properties" version = "0.1.6" @@ -389,6 +404,27 @@ dependencies = [ "serde_core", ] +[[package]] +name = "brotli" +version = "8.0.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5cc91aac060a7a1e25823bdccbfb6af1875b88f17c6daac97894eed8207166b3" +dependencies = [ + "alloc-no-stdlib", + "alloc-stdlib", + "brotli-decompressor", +] + +[[package]] +name = "brotli-decompressor" +version = "5.0.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a32acac15fe1967bc3986b2a6347dffc965602354ea6f450ad07e8bfd253583" +dependencies = [ + "alloc-no-stdlib", + "alloc-stdlib", +] + [[package]] name = "bumpalo" version = "3.20.3" @@ -408,6 +444,8 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040" dependencies = [ "find-msvc-tools", + "jobserver", + "libc", "shlex", ] @@ -609,6 +647,7 @@ dependencies = [ "clap", "comfy-table", "nu-ansi-term", + "parquet", "reedline", "syntect", "tempfile", @@ -835,6 +874,16 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "jobserver" +version = "0.1.35" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" +dependencies = [ + "getrandom 0.4.3", + "libc", +] + [[package]] name = "js-sys" version = "0.3.106" @@ -958,6 +1007,15 @@ version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" +[[package]] +name = "lz4_flex" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226" +dependencies = [ + "twox-hash", +] + [[package]] name = "memchr" version = "2.8.3" @@ -1096,6 +1154,37 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "parquet" +version = "59.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ff322f54b1a0f9288e614ed1f2d329b380af5476420db19f46ffb865e1163d73" +dependencies = [ + "ahash", + "arrow-array", + "arrow-buffer", + "arrow-data", + "arrow-ipc", + "arrow-schema", + "arrow-select", + "base64", + "brotli", + "bytes", + "chrono", + "flate2", + "half", + "hashbrown", + "lz4_flex", + "num-bigint", + "num-integer", + "num-traits", + "seq-macro", + "simdutf8", + "snap", + "twox-hash", + "zstd", +] + [[package]] name = "path-slash" version = "0.2.1" @@ -1299,6 +1388,12 @@ version = "1.0.28" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8a7852d02fc848982e0c167ef163aaff9cd91dc640ba85e263cb1ce46fae51cd" +[[package]] +name = "seq-macro" +version = "0.3.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1bc711410fbe7399f390ca1c3b60ad0f53f80e95c5eb935e52268a0e2cd49acc" + [[package]] name = "serde" version = "1.0.229" @@ -1418,6 +1513,12 @@ version = "1.16.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f9395f0f0eee849a9b707b2f06bb92a6a422090e2123bb2ef8e87a0e61892a8e" +[[package]] +name = "snap" +version = "1.1.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "199905e6153d6405f9728fe44daace35f8f837bbf830bb6e85fbd5828709a886" + [[package]] name = "strip-ansi-escapes" version = "0.2.1" @@ -1632,6 +1733,12 @@ version = "1.1.2+spec-1.1.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "7d56353a2a665ad0f41a421187180aab746c8c325620617ad883a99a1cbe66d2" +[[package]] +name = "twox-hash" +version = "2.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5283634e518fe9e82c7b20520bb4bc209009fd16c82077c802f8111ecbb0117a" + [[package]] name = "unicode-ident" version = "1.0.26" @@ -1933,3 +2040,31 @@ name = "zmij" version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" + +[[package]] +name = "zstd" +version = "0.13.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a" +dependencies = [ + "zstd-safe", +] + +[[package]] +name = "zstd-safe" +version = "7.3.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882" +dependencies = [ + "zstd-sys", +] + +[[package]] +name = "zstd-sys" +version = "2.1.0+zstd.1.5.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0" +dependencies = [ + "cc", + "pkg-config", +] diff --git a/Cargo.toml b/Cargo.toml index 303b6d3..5de5469 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -20,6 +20,7 @@ arrow-schema = "59.3.0" clap = { version = "4.6.7", features = ["derive"] } comfy-table = "8.0.1" nu-ansi-term = "0.50.3" +parquet = "59.3.0" reedline = "0.52.0" syntect = "5.3.0" terminal-colorsaurus = "1.0.3" diff --git a/README.md b/README.md index 961fb4d..f4aa2d9 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ A command-line tool for querying databases via [ADBC](https://arrow.apache.org/a - **Interactive SQL shell** - Execute SQL queries with command history and intuitive navigation - **Syntax highlighting** - SQL queries highlighted for improved readability - **Formatted output** - Results displayed in clean, aligned tables with dynamic column width -- **File export** - Export query results to JSON, CSV, or Arrow IPC files +- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files - **Fast and lightweight** - Built in Rust for high performance and minimal resource usage ## Installation @@ -107,8 +107,11 @@ Execute a query and output the result to a file: ```sh databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.json +databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.jsonl databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.csv databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.arrow +databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.arrows +databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.parquet ``` ## Reference diff --git a/docs/index.md b/docs/index.md index 06e5a75..81aa504 100644 --- a/docs/index.md +++ b/docs/index.md @@ -19,5 +19,5 @@ databow is a command-line tool for querying databases. - **Interactive SQL shell** - Execute SQL queries with command history and intuitive navigation - **Syntax highlighting** - SQL queries highlighted for improved readability - **Formatted output** - Results displayed in clean, aligned tables with dynamic column width -- **File export** - Export query results to JSON, CSV, or Arrow IPC files +- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files - **Fast and lightweight** - Built in Rust for high performance and minimal resource usage diff --git a/docs/reference.md b/docs/reference.md index 25c203a..7c553e5 100644 --- a/docs/reference.md +++ b/docs/reference.md @@ -108,11 +108,14 @@ databow --driver duckdb --query "SELECT 42 AS the_answer" --output result.json The output format is inferred from the file extension: -| Extension | Format | -|-----------------|-----------| -| `.json` | JSON | -| `.csv` | CSV | -| `.arrow`, `.ipc`| Arrow IPC | +| Extension | Format | +|-----------------|------------------| +| `.json` | JSON | +| `.jsonl` | JSON lines | +| `.csv` | CSV | +| `.arrow`, `.ipc`| Arrow IPC (file) | +| `.arrows` | Arrow IPC stream | +| `.parquet` | Parquet | ## --help diff --git a/docs/tutorial.md b/docs/tutorial.md index 59ec6ae..6bd479f 100644 --- a/docs/tutorial.md +++ b/docs/tutorial.md @@ -143,7 +143,7 @@ $ databow --profile warehouse --file query.sql └───────────────────┘ ``` -Instead of printing query results to stdout, the [`--output` argument](/reference/#-output) can be used to write results to JSON, CSV, or Arrow IPC files: +Instead of printing query results to stdout, the [`--output` argument](/reference/#-output) can be used to write results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files: ```console $ databow --profile warehouse --query "SELECT * FROM penguins" --output penguins.csv diff --git a/src/output.rs b/src/output.rs index 690335a..6b72066 100644 --- a/src/output.rs +++ b/src/output.rs @@ -2,26 +2,33 @@ // SPDX-License-Identifier: Apache-2.0 use arrow::csv::writer::Writer as CsvWriter; -use arrow::ipc::writer::FileWriter as IpcWriter; -use arrow::json::writer::{JsonArray, Writer as JsonWriter}; +use arrow::ipc::writer::{FileWriter as IpcWriter, StreamWriter as IpcStreamWriter}; +use arrow::json::writer::{JsonArray, LineDelimited, Writer as JsonWriter}; use arrow_array::RecordBatch; use arrow_schema::ArrowError; +use parquet::arrow::ArrowWriter; use std::fs::File; use std::path::Path; #[derive(Debug, Clone, Copy, PartialEq)] pub enum OutputFormat { Json, + Jsonl, Csv, Arrow, + ArrowStream, + Parquet, } impl OutputFormat { pub fn from_path(path: &Path) -> Result { match path.extension().and_then(|ext| ext.to_str()) { Some("json") => Ok(OutputFormat::Json), + Some("jsonl") => Ok(OutputFormat::Jsonl), Some("csv") => Ok(OutputFormat::Csv), Some("arrow" | "ipc") => Ok(OutputFormat::Arrow), + Some("arrows") => Ok(OutputFormat::ArrowStream), + Some("parquet") => Ok(OutputFormat::Parquet), Some(ext) => Err(format!("Unsupported file extension: '.{ext}'")), None => Err("Cannot infer format: no file extension".to_string()), } @@ -38,8 +45,11 @@ pub fn write_batches_to_file(batches: &[RecordBatch], path: &Path) -> Result<(), match format { OutputFormat::Json => write_json(batches, file), + OutputFormat::Jsonl => write_jsonl(batches, file), OutputFormat::Csv => write_csv(batches, file), OutputFormat::Arrow => write_arrow_ipc(batches, file), + OutputFormat::ArrowStream => write_arrow_stream(batches, file), + OutputFormat::Parquet => write_parquet(batches, file), } } @@ -53,6 +63,16 @@ fn write_json(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> { Ok(()) } +fn write_jsonl(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> { + let mut writer = JsonWriter::<_, LineDelimited>::new(file); + for batch in batches { + writer.write(batch)?; + } + writer.finish()?; + + Ok(()) +} + fn write_csv(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> { let mut writer = CsvWriter::new(file); for batch in batches { @@ -76,6 +96,39 @@ fn write_arrow_ipc(batches: &[RecordBatch], file: File) -> Result<(), ArrowError Ok(()) } +fn write_arrow_stream(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> { + if batches.is_empty() { + return Ok(()); + } + let schema = batches[0].schema(); + let mut writer = IpcStreamWriter::try_new(file, &schema)?; + for batch in batches { + writer.write(batch)?; + } + writer.finish()?; + + Ok(()) +} + +fn write_parquet(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> { + if batches.is_empty() { + return Ok(()); + } + let schema = batches[0].schema(); + let mut writer = ArrowWriter::try_new(file, schema, None) + .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + for batch in batches { + writer + .write(batch) + .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + } + writer + .close() + .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + + Ok(()) +} + #[cfg(test)] mod tests { use super::*; @@ -243,6 +296,118 @@ mod tests { assert!(!path.exists()); } + #[test] + fn test_output_format_from_path_arrows() { + let path = Path::new("output.arrows"); + assert_eq!( + OutputFormat::from_path(path).unwrap(), + OutputFormat::ArrowStream + ); + } + + #[test] + fn test_output_format_from_path_parquet() { + let path = Path::new("output.parquet"); + assert_eq!( + OutputFormat::from_path(path).unwrap(), + OutputFormat::Parquet + ); + } + + #[test] + fn test_output_format_from_path_jsonl() { + let path = Path::new("output.jsonl"); + assert_eq!(OutputFormat::from_path(path).unwrap(), OutputFormat::Jsonl); + } + + #[test] + fn test_write_arrow_stream() { + let dir = tempdir().unwrap(); + let path = dir.path().join("output.arrows"); + let batch = create_test_batch(); + + write_batches_to_file(std::slice::from_ref(&batch), &path).unwrap(); + + // Verify by reading it back as an IPC stream + let file = File::open(&path).unwrap(); + let reader = arrow::ipc::reader::StreamReader::try_new(file, None).unwrap(); + let read_batches: Vec = reader.map(|r| r.unwrap()).collect(); + + assert_eq!(read_batches.len(), 1); + assert_eq!(read_batches[0].num_rows(), batch.num_rows()); + assert_eq!(read_batches[0].num_columns(), batch.num_columns()); + } + + #[test] + fn test_write_arrow_stream_empty_batches_direct() { + let dir = tempdir().unwrap(); + let path = dir.path().join("output.arrows"); + let file = File::create(&path).unwrap(); + + let result = write_arrow_stream(&[], file); + + assert!(result.is_ok()); + } + + #[test] + fn test_write_parquet() { + use parquet::arrow::arrow_reader::ParquetRecordBatchReaderBuilder; + + let dir = tempdir().unwrap(); + let path = dir.path().join("output.parquet"); + let batch = create_test_batch(); + + write_batches_to_file(std::slice::from_ref(&batch), &path).unwrap(); + + // Verify by reading it back + let file = File::open(&path).unwrap(); + let reader = ParquetRecordBatchReaderBuilder::try_new(file) + .unwrap() + .build() + .unwrap(); + let read_batches: Vec = reader.map(|r| r.unwrap()).collect(); + + let total_rows: usize = read_batches.iter().map(|b| b.num_rows()).sum(); + assert_eq!(total_rows, batch.num_rows()); + assert_eq!(read_batches[0].num_columns(), batch.num_columns()); + } + + #[test] + fn test_write_parquet_empty_batches_direct() { + let dir = tempdir().unwrap(); + let path = dir.path().join("output.parquet"); + let file = File::create(&path).unwrap(); + + let result = write_parquet(&[], file); + + assert!(result.is_ok()); + } + + #[test] + fn test_write_jsonl() { + let dir = tempdir().unwrap(); + let path = dir.path().join("output.jsonl"); + let batch = create_test_batch(); + + write_batches_to_file(std::slice::from_ref(&batch), &path).unwrap(); + + let mut file = File::open(&path).unwrap(); + let mut contents = String::new(); + file.read_to_string(&mut contents).unwrap(); + + // JSON lines: one object per line, not a wrapping array + assert!(!contents.trim_start().starts_with('[')); + let lines: Vec<&str> = contents.lines().filter(|l| !l.is_empty()).collect(); + assert_eq!(lines.len(), 3); + for line in &lines { + assert!(line.trim_start().starts_with('{')); + assert!(line.trim_end().ends_with('}')); + } + assert!(contents.contains("Alice")); + assert!(contents.contains("Bob")); + assert!(contents.contains("Charlie")); + } + #[test] fn test_write_multiple_batches() { let dir = tempdir().unwrap(); From 9c491ba62707e6649cae72ab74793206d00bd294 Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Tue, 25 Aug 2026 10:29:04 -0700 Subject: [PATCH 2/7] docs: remove parens from arrow ipc file reference Co-authored-by: Ian Cook --- docs/reference.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/reference.md b/docs/reference.md index 7c553e5..14bcd52 100644 --- a/docs/reference.md +++ b/docs/reference.md @@ -113,7 +113,7 @@ The output format is inferred from the file extension: | `.json` | JSON | | `.jsonl` | JSON lines | | `.csv` | CSV | -| `.arrow`, `.ipc`| Arrow IPC (file) | +| `.arrow`, `.ipc`| Arrow IPC file | | `.arrows` | Arrow IPC stream | | `.parquet` | Parquet | From ff8ec011dbea7238d4ff03fb5ed5d24985c3ed7b Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Tue, 25 Aug 2026 10:29:44 -0700 Subject: [PATCH 3/7] docs: specify arrow ipc file in tutorial Co-authored-by: Ian Cook --- docs/tutorial.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/tutorial.md b/docs/tutorial.md index 6bd479f..e38cb24 100644 --- a/docs/tutorial.md +++ b/docs/tutorial.md @@ -143,7 +143,7 @@ $ databow --profile warehouse --file query.sql └───────────────────┘ ``` -Instead of printing query results to stdout, the [`--output` argument](/reference/#-output) can be used to write results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files: +Instead of printing query results to stdout, the [`--output` argument](/reference/#-output) can be used to write results to JSON, JSON lines, CSV, Arrow IPC file, Arrow IPC stream, or Parquet files: ```console $ databow --profile warehouse --query "SELECT * FROM penguins" --output penguins.csv From 217d7b7ec942d256d411978430734138b7c19cf6 Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Tue, 25 Aug 2026 10:30:46 -0700 Subject: [PATCH 4/7] docs: update file export line in readme Co-authored-by: Ian Cook --- README.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.md b/README.md index f4aa2d9..1a0cdc3 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ A command-line tool for querying databases via [ADBC](https://arrow.apache.org/a - **Interactive SQL shell** - Execute SQL queries with command history and intuitive navigation - **Syntax highlighting** - SQL queries highlighted for improved readability - **Formatted output** - Results displayed in clean, aligned tables with dynamic column width -- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files +- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, or Parquet files - **Fast and lightweight** - Built in Rust for high performance and minimal resource usage ## Installation From 53acbf75352f93a12c8d78e28877633f4f1d907b Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Thu, 1 Oct 2026 15:08:41 -0700 Subject: [PATCH 5/7] docs: update file export description Co-authored-by: Ian Cook --- docs/index.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/index.md b/docs/index.md index 81aa504..2954f7e 100644 --- a/docs/index.md +++ b/docs/index.md @@ -19,5 +19,5 @@ databow is a command-line tool for querying databases. - **Interactive SQL shell** - Execute SQL queries with command history and intuitive navigation - **Syntax highlighting** - SQL queries highlighted for improved readability - **Formatted output** - Results displayed in clean, aligned tables with dynamic column width -- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, Arrow IPC stream, or Parquet files +- **File export** - Export query results to JSON, JSON lines, CSV, Arrow IPC, or Parquet files - **Fast and lightweight** - Built in Rust for high performance and minimal resource usage From adcc82851761717e96365c8b53a2adba1c05b697 Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Thu, 1 Oct 2026 15:22:23 -0700 Subject: [PATCH 6/7] feat: write parquet with snappy compression and trim parquet features --- Cargo.lock | 90 --------------------------------------------------- Cargo.toml | 2 +- src/output.rs | 25 +++++++------- 3 files changed, 14 insertions(+), 103 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 9b2d440..2ffbe2b 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -70,21 +70,6 @@ dependencies = [ "memchr", ] -[[package]] -name = "alloc-no-stdlib" -version = "2.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "cc7bb162ec39d46ab1ca8c77bf72e890535becd1751bb45f64c597edb4c8c6b3" - -[[package]] -name = "alloc-stdlib" -version = "0.2.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0e76a019e91224d279006ff972f1e984179a6e9feb050adba6ce8274aef23195" -dependencies = [ - "alloc-no-stdlib", -] - [[package]] name = "android_system_properties" version = "0.1.6" @@ -404,27 +389,6 @@ dependencies = [ "serde_core", ] -[[package]] -name = "brotli" -version = "8.0.4" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "5cc91aac060a7a1e25823bdccbfb6af1875b88f17c6daac97894eed8207166b3" -dependencies = [ - "alloc-no-stdlib", - "alloc-stdlib", - "brotli-decompressor", -] - -[[package]] -name = "brotli-decompressor" -version = "5.0.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3a32acac15fe1967bc3986b2a6347dffc965602354ea6f450ad07e8bfd253583" -dependencies = [ - "alloc-no-stdlib", - "alloc-stdlib", -] - [[package]] name = "bumpalo" version = "3.20.3" @@ -444,8 +408,6 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f360145194ee8e21db5ee7f3fcd4fe52210864c75c985dae33218202c8bbe040" dependencies = [ "find-msvc-tools", - "jobserver", - "libc", "shlex", ] @@ -874,16 +836,6 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" -[[package]] -name = "jobserver" -version = "0.1.35" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3" -dependencies = [ - "getrandom 0.4.3", - "libc", -] - [[package]] name = "js-sys" version = "0.3.106" @@ -1007,15 +959,6 @@ version = "0.4.34" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" -[[package]] -name = "lz4_flex" -version = "0.14.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "ecbdfe44b1bd960b68170b417450a628c43f7cf56bb3c5317e61cb230ee7f226" -dependencies = [ - "twox-hash", -] - [[package]] name = "memchr" version = "2.8.3" @@ -1168,21 +1111,16 @@ dependencies = [ "arrow-schema", "arrow-select", "base64", - "brotli", "bytes", "chrono", - "flate2", "half", "hashbrown", - "lz4_flex", "num-bigint", "num-integer", "num-traits", "seq-macro", - "simdutf8", "snap", "twox-hash", - "zstd", ] [[package]] @@ -2040,31 +1978,3 @@ name = "zmij" version = "1.0.23" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" - -[[package]] -name = "zstd" -version = "0.13.3" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e91ee311a569c327171651566e07972200e76fcfe2242a4fa446149a3881c08a" -dependencies = [ - "zstd-safe", -] - -[[package]] -name = "zstd-safe" -version = "7.3.0" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "64d80649ab6db9d9f6f9c80a40becd948eda4714a0a5ac8c4d157a32231c7882" -dependencies = [ - "zstd-sys", -] - -[[package]] -name = "zstd-sys" -version = "2.1.0+zstd.1.5.7" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "0ef0a8027ec3ee71300ab3bcbcd0393f434aa72b91ca6d635a39941deae8eea0" -dependencies = [ - "cc", - "pkg-config", -] diff --git a/Cargo.toml b/Cargo.toml index 5de5469..18874b9 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -20,7 +20,7 @@ arrow-schema = "59.3.0" clap = { version = "4.6.7", features = ["derive"] } comfy-table = "8.0.1" nu-ansi-term = "0.50.3" -parquet = "59.3.0" +parquet = { version = "59.3.0", default-features = false, features = ["arrow", "snap"] } reedline = "0.52.0" syntect = "5.3.0" terminal-colorsaurus = "1.0.3" diff --git a/src/output.rs b/src/output.rs index 6b72066..7264591 100644 --- a/src/output.rs +++ b/src/output.rs @@ -7,6 +7,8 @@ use arrow::json::writer::{JsonArray, LineDelimited, Writer as JsonWriter}; use arrow_array::RecordBatch; use arrow_schema::ArrowError; use parquet::arrow::ArrowWriter; +use parquet::basic::Compression; +use parquet::file::properties::WriterProperties; use std::fs::File; use std::path::Path; @@ -115,16 +117,14 @@ fn write_parquet(batches: &[RecordBatch], file: File) -> Result<(), ArrowError> return Ok(()); } let schema = batches[0].schema(); - let mut writer = ArrowWriter::try_new(file, schema, None) - .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + let props = WriterProperties::builder() + .set_compression(Compression::SNAPPY) + .build(); + let mut writer = ArrowWriter::try_new(file, schema, Some(props))?; for batch in batches { - writer - .write(batch) - .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + writer.write(batch)?; } - writer - .close() - .map_err(|e| ArrowError::ExternalError(Box::new(e)))?; + writer.close()?; Ok(()) } @@ -361,10 +361,11 @@ mod tests { // Verify by reading it back let file = File::open(&path).unwrap(); - let reader = ParquetRecordBatchReaderBuilder::try_new(file) - .unwrap() - .build() - .unwrap(); + let builder = ParquetRecordBatchReaderBuilder::try_new(file).unwrap(); + for column in builder.metadata().row_group(0).columns() { + assert_eq!(column.compression(), Compression::SNAPPY); + } + let reader = builder.build().unwrap(); let read_batches: Vec = reader.map(|r| r.unwrap()).collect(); let total_rows: usize = read_batches.iter().map(|b| b.num_rows()).sum(); From 1a9326f0b22b9030b9e86ea7c66da8774c9d389b Mon Sep 17 00:00:00 2001 From: Emil Sadek Date: Thu, 1 Oct 2026 15:51:26 -0700 Subject: [PATCH 7/7] fix: validate output file extension before connecting --- src/cli.rs | 32 ++++++++++++++++++++++++++++++-- tests/integration_test.rs | 24 ++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 2 deletions(-) diff --git a/src/cli.rs b/src/cli.rs index 8dac594..9dd9061 100644 --- a/src/cli.rs +++ b/src/cli.rs @@ -1,6 +1,7 @@ // Copyright 2026 Columnar Technologies Inc. // SPDX-License-Identifier: Apache-2.0 +use crate::output::OutputFormat; use crate::table::TableMode; use clap::{Arg, ArgAction, Command, value_parser}; use std::path::PathBuf; @@ -84,7 +85,8 @@ pub fn parse_args() -> AppConfig { Arg::new("output") .long("output") .help("Write result to file") - .value_name("file"), + .value_name("file") + .value_parser(parse_output_path), ]; let command = Command::new("databow") .version(env!("CARGO_PKG_VERSION")) @@ -163,7 +165,7 @@ pub fn parse_args() -> AppConfig { .copied() .unwrap_or_default(); - let output_path = matches.get_one::("output").map(PathBuf::from); + let output_path = matches.get_one::("output").cloned(); if output_path.is_some() && matches!(query_source, QuerySource::Interactive) { eprintln!("Error: --output cannot be used in interactive mode"); exit(1); @@ -177,6 +179,12 @@ pub fn parse_args() -> AppConfig { } } +fn parse_output_path(s: &str) -> Result { + let path = PathBuf::from(s); + OutputFormat::from_path(&path)?; + Ok(path) +} + fn uri_has_driver_scheme(uri: &str) -> bool { let Some(idx) = uri.find(':') else { return false; @@ -202,6 +210,26 @@ fn parse_option(option: &str) -> Result<(String, String), String> { mod tests { use super::*; + #[test] + fn test_parse_output_path_valid() { + assert_eq!( + parse_output_path("out.parquet").unwrap(), + PathBuf::from("out.parquet") + ); + } + + #[test] + fn test_parse_output_path_unsupported_extension() { + let err = parse_output_path("out.xyz").unwrap_err(); + assert!(err.contains("Unsupported file extension")); + } + + #[test] + fn test_parse_output_path_no_extension() { + let err = parse_output_path("out").unwrap_err(); + assert!(err.contains("no file extension")); + } + #[test] fn test_connection_source_direct() { let source = ConnectionSource::Direct { diff --git a/tests/integration_test.rs b/tests/integration_test.rs index 302c35c..e68795c 100644 --- a/tests/integration_test.rs +++ b/tests/integration_test.rs @@ -479,3 +479,27 @@ fn test_timestamp_with_time_zone() { stdout ); } + +#[test] +fn test_invalid_output_extension_fails_before_connect() { + let output = Command::new("cargo") + .args([ + "run", + "--", + "--driver", + "nonexistent_driver", + "--query", + "SELECT 1", + "--output", + "out.xyz", + ]) + .output() + .expect("Failed to execute command"); + + // Exit code 2 and "invalid value" come from clap argument parsing, + // which happens before any connection attempt. + assert_eq!(output.status.code(), Some(2)); + let stderr = String::from_utf8_lossy(&output.stderr); + assert!(stderr.contains("invalid value 'out.xyz' for '--output '")); + assert!(stderr.contains("Unsupported file extension")); +}