diff --git a/CHANGELOG.md b/CHANGELOG.md index a1ee5a7..cccb538 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,4 +1,7 @@ # Changelog +## Next +- Add `Target` struct helper for creating targets, and hide the targets in the `target` module. + ## 0.1.0 - 2023 Jan 11 - Initial release diff --git a/README.md b/README.md index e08ae33..1d8db18 100644 --- a/README.md +++ b/README.md @@ -5,6 +5,35 @@ CSV processing library inspired by [csvsc](https://crates.io/crates/csvsc) [![Crates.io](https://img.shields.io/crates/v/csv-pipeline.svg)](https://crates.io/crates/csv-pipeline) [![Documentation](https://docs.rs/csv-pipeline/badge.svg)](https://docs.rs/csv-pipeline) +## Basic Example +```rs +use csv_pipeline::{Pipeline, Transformer}; + +// First create a pipeline from a CSV file path +let csv = Pipeline::from_path("test/Countries.csv") + .unwrap() + // Add a column with values computed from a closure + .add_col("Language", |headers, row| { + match headers.get_field(row, "Country") { + Some("Norway") => Ok("Norwegian".into()), + _ => Ok("Unknown".into()), + } + }) + // Make the "Country" column uppercase + .rename_col("Country", "COUNTRY") + .map_col("COUNTRY", |id_str| Ok(id_str.to_uppercase())) + // Collect the csv into a string + .collect_into_string() + .unwrap(); + +assert_eq!( + csv, + "ID,COUNTRY,Language\n\ + 1,NORWAY,Norwegian\n\ + 2,TUVALU,Unknown\n" +); +``` + ## Dev Instructions ### Get started diff --git a/src/headers.rs b/src/headers.rs index 6b7ee9c..a922ebe 100644 --- a/src/headers.rs +++ b/src/headers.rs @@ -2,6 +2,7 @@ use crate::{Error, Row}; use csv::StringRecordIter; use std::collections::BTreeMap; +/// The headers of a CSV file #[derive(Debug, Clone, PartialEq)] pub struct Headers { indexes: BTreeMap, diff --git a/src/lib.rs b/src/lib.rs index 8a6f288..bbe44a2 100644 --- a/src/lib.rs +++ b/src/lib.rs @@ -1,6 +1,6 @@ //! CSV processing library inspired by [csvsc](https://crates.io/crates/csvsc) //! -//! ## Quickstart +//! ## Get started //! //! The first thing you need is to create a [`Pipeline`]. This can be done by calling [`Pipeline::from_reader`] with a [`csv::Reader`], or [`Pipeline::from_path`] with a path. //! @@ -11,7 +11,7 @@ //! Finally, you probably want to run the pipeline. There are a few options: //! - [`Pipeline::build`] gives you a [`PipelineIter`] which you can iterate through //! - [`Pipeline::run`] runs through the pipeline until it finds an error, or the end -//! - [`Pipeline::collect_into_string`] runs the pipeline and returns the csv as a `Result`. Can be a convenient alternative to flushing to a [`StringTarget`]. +//! - [`Pipeline::collect_into_string`] runs the pipeline and returns the csv as a `Result`. Can be a convenient alternative to flushing to a [`StringTarget`](target::StringTarget). //! //! ## Basic Example //! @@ -84,18 +84,38 @@ //! ``` //! +use std::path::PathBuf; + mod headers; mod pipeline; mod pipeline_iterators; -mod target; mod transform; pub use headers::Headers; pub use pipeline::{Pipeline, PipelineIter}; -pub use target::{PathTarget, StderrTarget, StdoutTarget, StringTarget, Target}; pub use transform::Transformer; +pub mod target; +/// Helper for building a target to flush data into +pub struct Target {} +impl Target { + pub fn path>(path: P) -> target::PathTarget { + target::PathTarget::new(path) + } + pub fn stdout() -> target::StdoutTarget { + target::StdoutTarget::new() + } + pub fn stderr() -> target::StderrTarget { + target::StderrTarget::new() + } + pub fn string<'a>(s: &'a mut String) -> target::StringTarget { + target::StringTarget::new(s) + } +} + +/// Alias of [`csv::StringRecord`] pub type Row = csv::StringRecord; +/// Alias of `Result` pub type RowResult = Result; #[derive(Debug)] diff --git a/src/pipeline.rs b/src/pipeline.rs index 40f300f..2db042b 100644 --- a/src/pipeline.rs +++ b/src/pipeline.rs @@ -2,15 +2,16 @@ use super::headers::Headers; use crate::pipeline_iterators::{ AddCol, Flush, MapCol, MapRow, TransformInto, Validate, ValidateCol, }; -use crate::target::Target; +use crate::target::{StringTarget, Target}; use crate::transform::Transform; -use crate::{Error, Row, RowResult, StringTarget}; +use crate::{Error, Row, RowResult}; use csv::{Reader, ReaderBuilder, StringRecordsIntoIter}; use std::borrow::BorrowMut; use std::collections::BTreeMap; use std::io; use std::path::Path; +/// The main thing pub struct Pipeline<'a> { pub headers: Headers, iterator: Box + 'a>, @@ -303,6 +304,7 @@ impl<'a> IntoIterator for Pipeline<'a> { } } +/// A pipeline you can iterate through. You can get one using [`Pipeline::build`]. pub struct PipelineIter<'a> { pub headers: Headers, pub iterator: Box + 'a>, diff --git a/src/transform.rs b/src/transform.rs index 2c62e7b..1855c2b 100644 --- a/src/transform.rs +++ b/src/transform.rs @@ -3,6 +3,7 @@ use core::fmt::Display; use std::collections::hash_map::DefaultHasher; use std::hash::{Hash, Hasher}; +/// For grouping and reducing rows. pub trait Transform { /// Add the row to the hasher to group this row separately from others fn hash( @@ -24,6 +25,7 @@ pub trait Transform { fn value(&self) -> String; } +/// A struct for building a [`Transform`], which you can use with [`Pipeline::transform_into`](crate::Pipeline::transform_into). pub struct Transformer { name: String, from_col: String, @@ -35,10 +37,12 @@ impl Transformer { from_col: col_name.to_string(), } } + /// Specify which column the transform should be based on pub fn from_col(mut self, col_name: &str) -> Self { self.from_col = col_name.to_string(); self } + /// Keep the unique values from this column pub fn keep_unique(self) -> Box { Box::new(KeepUnique { name: self.name, @@ -46,6 +50,7 @@ impl Transformer { value: "".to_string(), }) } + /// Reduce the values from this column into a single value using a closure pub fn reduce<'a, R, V>(self, reduce: R, init: V) -> Box where R: FnMut(V, &str) -> Result + 'a, @@ -120,7 +125,7 @@ impl Transform for KeepUnique { } } -pub fn compute_hash<'a>( +pub(crate) fn compute_hash<'a>( transformers: &Vec>, headers: &Headers, row: &Row,