WIP This PR covers migration crates/nu-cmd-dataframes to a new plugin ./crates/nu_plugin_polars ## TODO List Other: - [X] Fix examples - [x] Fix Plugin Test Harness - [X] Move Cache to Mutex<BTreeMap> - [X] Logic for disabling/enabling plugin GC based off whether items are cached. - [x] NuExpression custom values - [X] Optimize caching (don't cache every object creation). - [x] Fix dataframe operations (in NuDataFrameCustomValue::operations) - [x] Added plugin_debug! macro that for checking an env variable POLARS_PLUGIN_DEBUG Fix duplicated commands: - [x] There are two polars median commands, one for lazy and one for expr.. there should only be one that works for both. I temporarily called on polars expr-median (inside expressions_macros.rs) - [x] polars quantile (lazy, and expr). the expr one is temporarily expr-median - [x] polars is-in (renamed one series-is-in) Commands: - [x] AppendDF - [x] CastDF - [X] ColumnsDF - [x] DataTypes - [x] Summary - [x] DropDF - [x] DropDuplicates - [x] DropNulls - [x] Dummies - [x] FilterWith - [X] FirstDF - [x] GetDF - [x] LastDF - [X] ListDF - [x] MeltDF - [X] OpenDataFrame - [x] QueryDf - [x] RenameDF - [x] SampleDF - [x] SchemaDF - [x] ShapeDF - [x] SliceDF - [x] TakeDF - [X] ToArrow - [x] ToAvro - [X] ToCSV - [X] ToDataFrame - [X] ToNu - [x] ToParquet - [x] ToJsonLines - [x] WithColumn - [x] ExprAlias - [x] ExprArgWhere - [x] ExprCol - [x] ExprConcatStr - [x] ExprCount - [x] ExprLit - [x] ExprWhen - [x] ExprOtherwise - [x] ExprQuantile - [x] ExprList - [x] ExprAggGroups - [x] ExprCount - [x] ExprIsIn - [x] ExprNot - [x] ExprMax - [x] ExprMin - [x] ExprSum - [x] ExprMean - [x] ExprMedian - [x] ExprStd - [x] ExprVar - [x] ExprDatePart - [X] LazyAggregate - [x] LazyCache - [X] LazyCollect - [x] LazyFetch - [x] LazyFillNA - [x] LazyFillNull - [x] LazyFilter - [x] LazyJoin - [x] LazyQuantile - [x] LazyMedian - [x] LazyReverse - [x] LazySelect - [x] LazySortBy - [x] ToLazyFrame - [x] ToLazyGroupBy - [x] LazyExplode - [x] LazyFlatten - [x] AllFalse - [x] AllTrue - [x] ArgMax - [x] ArgMin - [x] ArgSort - [x] ArgTrue - [x] ArgUnique - [x] AsDate - [x] AsDateTime - [x] Concatenate - [x] Contains - [x] Cumulative - [x] GetDay - [x] GetHour - [x] GetMinute - [x] GetMonth - [x] GetNanosecond - [x] GetOrdinal - [x] GetSecond - [x] GetWeek - [x] GetWeekDay - [x] GetYear - [x] IsDuplicated - [x] IsIn - [x] IsNotNull - [x] IsNull - [x] IsUnique - [x] NNull - [x] NUnique - [x] NotSeries - [x] Replace - [x] ReplaceAll - [x] Rolling - [x] SetSeries - [x] SetWithIndex - [x] Shift - [x] StrLengths - [x] StrSlice - [x] StrFTime - [x] ToLowerCase - [x] ToUpperCase - [x] Unique - [x] ValueCount --------- Co-authored-by: Jack Wright <jack.wright@disqo.com>
134 lines
3.8 KiB
Rust
134 lines
3.8 KiB
Rust
use std::{fs::File, path::PathBuf};
|
|
|
|
use nu_plugin::{EngineInterface, EvaluatedCall, PluginCommand};
|
|
use nu_protocol::{
|
|
Category, Example, LabeledError, PipelineData, ShellError, Signature, Spanned, SyntaxShape,
|
|
Type, Value,
|
|
};
|
|
use polars::prelude::{CsvWriter, SerWriter};
|
|
|
|
use crate::PolarsPlugin;
|
|
|
|
use super::super::values::NuDataFrame;
|
|
|
|
#[derive(Clone)]
|
|
pub struct ToCSV;
|
|
|
|
impl PluginCommand for ToCSV {
|
|
type Plugin = PolarsPlugin;
|
|
|
|
fn name(&self) -> &str {
|
|
"polars to-csv"
|
|
}
|
|
|
|
fn usage(&self) -> &str {
|
|
"Saves dataframe to CSV file."
|
|
}
|
|
|
|
fn signature(&self) -> Signature {
|
|
Signature::build(self.name())
|
|
.required("file", SyntaxShape::Filepath, "file path to save dataframe")
|
|
.named(
|
|
"delimiter",
|
|
SyntaxShape::String,
|
|
"file delimiter character",
|
|
Some('d'),
|
|
)
|
|
.switch("no-header", "Indicates if file doesn't have header", None)
|
|
.input_output_type(Type::Custom("dataframe".into()), Type::Any)
|
|
.category(Category::Custom("dataframe".into()))
|
|
}
|
|
|
|
fn examples(&self) -> Vec<Example> {
|
|
vec![
|
|
Example {
|
|
description: "Saves dataframe to CSV file",
|
|
example: "[[a b]; [1 2] [3 4]] | dfr into-df | dfr to-csv test.csv",
|
|
result: None,
|
|
},
|
|
Example {
|
|
description: "Saves dataframe to CSV file using other delimiter",
|
|
example: "[[a b]; [1 2] [3 4]] | dfr into-df | dfr to-csv test.csv --delimiter '|'",
|
|
result: None,
|
|
},
|
|
]
|
|
}
|
|
|
|
fn run(
|
|
&self,
|
|
plugin: &Self::Plugin,
|
|
_engine: &EngineInterface,
|
|
call: &EvaluatedCall,
|
|
input: PipelineData,
|
|
) -> Result<PipelineData, LabeledError> {
|
|
command(plugin, call, input).map_err(|e| e.into())
|
|
}
|
|
}
|
|
|
|
fn command(
|
|
plugin: &PolarsPlugin,
|
|
call: &EvaluatedCall,
|
|
input: PipelineData,
|
|
) -> Result<PipelineData, ShellError> {
|
|
let file_name: Spanned<PathBuf> = call.req(0)?;
|
|
let delimiter: Option<Spanned<String>> = call.get_flag("delimiter")?;
|
|
let no_header: bool = call.has_flag("no-header")?;
|
|
|
|
let df = NuDataFrame::try_from_pipeline_coerce(plugin, input, call.head)?;
|
|
|
|
let mut file = File::create(&file_name.item).map_err(|e| ShellError::GenericError {
|
|
error: "Error with file name".into(),
|
|
msg: e.to_string(),
|
|
span: Some(file_name.span),
|
|
help: None,
|
|
inner: vec![],
|
|
})?;
|
|
|
|
let writer = CsvWriter::new(&mut file);
|
|
|
|
let writer = if no_header {
|
|
writer.include_header(false)
|
|
} else {
|
|
writer.include_header(true)
|
|
};
|
|
|
|
let mut writer = match delimiter {
|
|
None => writer,
|
|
Some(d) => {
|
|
if d.item.len() != 1 {
|
|
return Err(ShellError::GenericError {
|
|
error: "Incorrect delimiter".into(),
|
|
msg: "Delimiter has to be one char".into(),
|
|
span: Some(d.span),
|
|
help: None,
|
|
inner: vec![],
|
|
});
|
|
} else {
|
|
let delimiter = match d.item.chars().next() {
|
|
Some(d) => d as u8,
|
|
None => unreachable!(),
|
|
};
|
|
|
|
writer.with_separator(delimiter)
|
|
}
|
|
}
|
|
};
|
|
|
|
writer
|
|
.finish(&mut df.to_polars())
|
|
.map_err(|e| ShellError::GenericError {
|
|
error: "Error writing to file".into(),
|
|
msg: e.to_string(),
|
|
span: Some(file_name.span),
|
|
help: None,
|
|
inner: vec![],
|
|
})?;
|
|
|
|
let file_value = Value::string(format!("saved {:?}", &file_name.item), file_name.span);
|
|
|
|
Ok(PipelineData::Value(
|
|
Value::list(vec![file_value], call.head),
|
|
None,
|
|
))
|
|
}
|