From 19a8be09999b128d1ff30d6a2a95861ec93fb014 Mon Sep 17 00:00:00 2001 From: Eric Rodrigues Pires Date: Sun, 19 Oct 2025 22:05:02 -0300 Subject: [PATCH] Improve documentation for Python module and minor fixes --- duper-py/python/duper/__init__.py | 4 +- duper-py/python/duper/_duper.pyi | 34 +++++- duper-py/src/ser.rs | 75 +++++-------- duper-py/tests/test_pickle.py | 25 +++++ duper/src/format.rs | 1 + duper/src/visitor/pretty_printer.rs | 2 +- ...etty_printer_tests__strip_identifiers.snap | 2 +- serde-duper-macros/src/lib.rs | 26 ++--- serde-duper/src/ser.rs | 2 +- serde-duper/src/types/mod.rs | 106 +++++++++++++++++- serde-duper/tests/types_option_none.rs | 16 ++- serde-duper/tests/types_option_some.rs | 9 +- serde-duper/tests/types_plain.rs | 13 ++- 13 files changed, 237 insertions(+), 78 deletions(-) create mode 100644 duper-py/tests/test_pickle.py diff --git a/duper-py/python/duper/__init__.py b/duper-py/python/duper/__init__.py index d51e7f6..be08892 100644 --- a/duper-py/python/duper/__init__.py +++ b/duper-py/python/duper/__init__.py @@ -1,4 +1,6 @@ -"""Utilities for converting to and from Python types into the Duper format.""" +r"""Utilities for converting to and from Python types into the Duper format. + +:mod:`duper` exposes an API similar to :mod:`json`.""" from ._duper import ( dumps, diff --git a/duper-py/python/duper/_duper.pyi b/duper-py/python/duper/_duper.pyi index a4f1d5e..88c056f 100644 --- a/duper-py/python/duper/_duper.pyi +++ b/duper-py/python/duper/_duper.pyi @@ -14,7 +14,15 @@ def dumps( indent: str | int | None = None, strip_identifiers: bool = False, ) -> str: - """Serialize obj as a Duper value formatted str.""" + """Serialize ``obj`` to a Duper value formatted ``str``. + + If ``indent`` is a positive integer, then JSON array elements and + object members will be pretty-printed with that indent level. The + indent may also be specified as a ``str`` containing spaces and/or + tabs. ``None`` is the most compact representation. + + If ``strip_identifiers`` is ``True``, then this function will strip + all identifiers from the serialized value.""" def dump( obj: Any, @@ -23,10 +31,28 @@ def dump( indent: str | int | None = None, strip_identifiers: bool = False, ) -> None: - """Serialize obj as a Duper value formatted stream to fp (a file-like object).""" + """Serialize ``obj`` as a Duper value formatted stream to ``fp`` (a + ``.write()``-supporting file-like object). + + If ``indent`` is a positive integer, then JSON array elements and + object members will be pretty-printed with that indent level. The + indent may also be specified as a ``str`` containing spaces and/or + tabs. ``None`` is the most compact representation. + + If ``strip_identifiers`` is ``True``, then this function will strip + all identifiers from the serialized value.""" def loads(s: str, *, parse_any: bool = False) -> Any: - """Deserialize s (a str instance containing a Duper object or array) to a Python object.""" + """Deserialize ``s`` (a ``str`` instance containing a Duper object or + array) to a Python object. + + If ``parse_any`` is ``True``, then this function will also deserialize + types other than objects and arrays. + """ def load(fp: TextIOBase, *, parse_any: bool = False) -> Any: - """Deserialize fp (a file-like object containing a Duper object or array) to a Python object.""" + """Deserialize ``fp`` (a ``.read()``-supporting file-like object + containing a Duper object or array) to a Python object. + + If ``parse_any`` is ``True``, then this function will also deserialize + types other than objects and arrays.""" diff --git a/duper-py/src/ser.rs b/duper-py/src/ser.rs index a9232d9..8a18847 100644 --- a/duper-py/src/ser.rs +++ b/duper-py/src/ser.rs @@ -116,24 +116,6 @@ pub(crate) fn serialize_pyany<'py>(obj: Bound<'py, PyAny>) -> PyResult(format!( "Unsupported type: {}", @@ -521,7 +503,7 @@ fn serialize_pyclass_identifier<'py>( )?)) .map_err(|error| { PyErr::new::(format!( - "Invalid identifier: {}\n{}", + "Invalid identifier: {} ({})", identifier, error )) })?, @@ -537,7 +519,7 @@ fn serialize_pyclass_identifier<'py>( )?)) .map_err(|error| { PyErr::new::(format!( - "Invalid identifier: {}\n{}", + "Invalid identifier: {} ({})", identifier, error )) })?, @@ -548,31 +530,32 @@ fn serialize_pyclass_identifier<'py>( } fn serialize_pydantic_model<'py>(obj: Bound<'py, PyAny>) -> PyResult> { - if let Ok(class) = obj.getattr("__class__") { - let model_fields = class.getattr("model_fields")?; - if model_fields.is_instance_of::() { - let field_dict = model_fields.downcast::()?; - let fields: PyResult> = field_dict - .iter() - .map(|(field_name, _field_info)| { - let field_name: &Bound<'py, PyString> = field_name.downcast()?; - let value = obj.getattr(field_name)?; - Ok(( - DuperKey::from(Cow::Owned(field_name.to_string())), - serialize_pyany(value)?, - )) - }) - .collect(); - return Ok(DuperValue { - identifier: serialize_pyclass_identifier(&obj)?, - inner: DuperInner::Object( - DuperObject::try_from(fields?).expect("no duplicate keys in pydantic model"), - ), - }); - } + if let Ok(class) = obj.getattr("__class__") + && let model_fields = class.getattr("model_fields")? + && model_fields.is_instance_of::() + { + let field_dict = model_fields.downcast::()?; + let fields: PyResult> = field_dict + .iter() + .map(|(field_name, _field_info)| { + let field_name: &Bound<'py, PyString> = field_name.downcast()?; + let value = obj.getattr(field_name)?; + Ok(( + DuperKey::from(Cow::Owned(field_name.to_string())), + serialize_pyany(value)?, + )) + }) + .collect(); + Ok(DuperValue { + identifier: serialize_pyclass_identifier(&obj)?, + inner: DuperInner::Object( + DuperObject::try_from(fields?).expect("no duplicate keys in pydantic model"), + ), + }) + } else { + Err(PyErr::new::(format!( + "Unsupported type: {}", + obj.get_type() + ))) } - Err(PyErr::new::(format!( - "Unsupported type: {}", - obj.get_type() - ))) } diff --git a/duper-py/tests/test_pickle.py b/duper-py/tests/test_pickle.py new file mode 100644 index 0000000..7760ca0 --- /dev/null +++ b/duper-py/tests/test_pickle.py @@ -0,0 +1,25 @@ +import pickle +import duper + + +class MyClass: + def __init__(self, name: str): + self.name = name + + def greet(self) -> str: + return f"Hello, {self.name}!" + + +def test_pickle(): + obj = MyClass("McDuper") + pic = pickle.dumps(obj) + dup = duper.dumps(pic) + + assert type(dup) is str + assert dup.startswith('b"') + assert dup.endswith('"') + + undup = duper.loads(dup, parse_any=True) + unpic = pickle.loads(undup) + + assert unpic.greet() == "Hello, McDuper!" diff --git a/duper/src/format.rs b/duper/src/format.rs index c05d29e..2fd8c4c 100644 --- a/duper/src/format.rs +++ b/duper/src/format.rs @@ -35,6 +35,7 @@ fn format_cow_str<'a>(string: &Cow<'a, str>) -> Cow<'a, str> { was_hashtag = false; was_quotes = true; chars_to_escape += 1; + max_hashtags = max_hashtags.max(1); } '#' if was_hashtag => { curr_hashtags += 1; diff --git a/duper/src/visitor/pretty_printer.rs b/duper/src/visitor/pretty_printer.rs index 3c681e1..8bd3b40 100644 --- a/duper/src/visitor/pretty_printer.rs +++ b/duper/src/visitor/pretty_printer.rs @@ -576,7 +576,7 @@ mod pretty_printer_tests { DuperIdentifier::try_from(Cow::Borrowed("Float")) .expect("valid identifier"), ), - inner: DuperInner::Float(3.14), + inner: DuperInner::Float(4.2), }, DuperValue { identifier: Some( diff --git a/duper/src/visitor/snapshots/duper__visitor__pretty_printer__pretty_printer_tests__strip_identifiers.snap b/duper/src/visitor/snapshots/duper__visitor__pretty_printer__pretty_printer_tests__strip_identifiers.snap index 9b2b336..fc393ac 100644 --- a/duper/src/visitor/snapshots/duper__visitor__pretty_printer__pretty_printer_tests__strip_identifiers.snap +++ b/duper/src/visitor/snapshots/duper__visitor__pretty_printer__pretty_printer_tests__strip_identifiers.snap @@ -8,7 +8,7 @@ expression: pp string_field: "test", }, array_field: [ - 3.14, + 4.2, true, ], } diff --git a/serde-duper-macros/src/lib.rs b/serde-duper-macros/src/lib.rs index 98e12da..29a957c 100644 --- a/serde-duper-macros/src/lib.rs +++ b/serde-duper-macros/src/lib.rs @@ -109,20 +109,20 @@ fn has_serde_derive_attributes(attrs: &[Attribute]) -> (bool, bool) { let mut has_serialize = false; let mut has_deserialize = false; for attr in attrs { - if attr.path().is_ident("derive") { - if let Meta::List(list) = &attr.meta { - let _ = list.parse_nested_meta(|nested| { - if let Some(segment) = nested.path.segments.last() { - let ident = &segment.ident; - if ident == "Serialize" { - has_serialize = true; - } else if ident == "Deserialize" { - has_deserialize = true; - } + if attr.path().is_ident("derive") + && let Meta::List(list) = &attr.meta + { + let _ = list.parse_nested_meta(|nested| { + if let Some(segment) = nested.path.segments.last() { + let ident = &segment.ident; + if ident == "Serialize" { + has_serialize = true; + } else if ident == "Deserialize" { + has_deserialize = true; } - Ok(()) - }); - } + } + Ok(()) + }); } } (has_serialize, has_deserialize) diff --git a/serde-duper/src/ser.rs b/serde-duper/src/ser.rs index d78a827..26d76fb 100644 --- a/serde-duper/src/ser.rs +++ b/serde-duper/src/ser.rs @@ -48,7 +48,7 @@ where T: Serialize, { Ok(DuperPrettyPrinter::new(false, indent) - .map_err(|error| Error::invalid_value(error))? + .map_err(Error::invalid_value)? .pretty_print(to_duper(value)?)) } diff --git a/serde-duper/src/types/mod.rs b/serde-duper/src/types/mod.rs index 0b52223..e0bc63f 100644 --- a/serde-duper/src/types/mod.rs +++ b/serde-duper/src/types/mod.rs @@ -730,6 +730,7 @@ pub mod DuperRegex { deserializer.deserialize_newtype_struct("Regex", Visitor) } } +#[cfg(feature = "regex")] pub mod DuperOptionRegex { use super::*; use ::regex::Regex as WrappedType; @@ -740,7 +741,7 @@ pub mod DuperOptionRegex { { match value { Some(value) => serializer.serialize_newtype_struct("Regex", value.as_str()), - None => serializer.serialize_newtype_struct("Regex", &Option::<&str>::None), + None => serializer.serialize_newtype_struct("Regex", &Option::<()>::None), } } @@ -789,6 +790,109 @@ pub mod DuperOptionRegex { deserializer.deserialize_newtype_struct("Regex", Visitor) } } +#[cfg(feature = "regex")] +pub mod DuperBytesRegex { + use super::*; + use ::regex::bytes::Regex as WrappedType; + + pub fn serialize(value: &WrappedType, serializer: S) -> Result + where + S: Serializer, + { + serializer.serialize_newtype_struct("BytesRegex", value.as_str()) + } + + pub fn deserialize<'de, D>(deserializer: D) -> Result + where + D: Deserializer<'de>, + { + struct Visitor; + + impl<'de> de::Visitor<'de> for Visitor { + type Value = WrappedType; + + fn expecting(&self, formatter: &mut ::std::fmt::Formatter) -> ::std::fmt::Result { + formatter.write_str("a BytesRegex") + } + + fn visit_str(self, v: &str) -> Result + where + E: de::Error, + { + WrappedType::new(v).map_err(|error| E::custom(error)) + } + + fn visit_newtype_struct(self, deserializer: D) -> Result + where + D: Deserializer<'de>, + { + deserializer.deserialize_str(self) + } + } + + deserializer.deserialize_newtype_struct("BytesRegex", Visitor) + } +} +#[cfg(feature = "regex")] +pub mod DuperOptionBytesRegex { + use super::*; + use ::regex::bytes::Regex as WrappedType; + + pub fn serialize(value: &Option, serializer: S) -> Result + where + S: Serializer, + { + match value { + Some(value) => serializer.serialize_newtype_struct("BytesRegex", value.as_str()), + None => serializer.serialize_newtype_struct("BytesRegex", &Option::<()>::None), + } + } + + pub fn deserialize<'de, D>(deserializer: D) -> Result, D::Error> + where + D: Deserializer<'de>, + { + struct Visitor; + + impl<'de> de::Visitor<'de> for Visitor { + type Value = Option; + + fn expecting(&self, formatter: &mut ::std::fmt::Formatter) -> ::std::fmt::Result { + formatter.write_str("a BytesRegex") + } + + fn visit_str(self, v: &str) -> Result + where + E: de::Error, + { + Some(WrappedType::new(v).map_err(|error| E::custom(error))).transpose() + } + + fn visit_some(self, deserializer: D) -> Result + where + D: Deserializer<'de>, + { + deserializer.deserialize_str(self) + } + + fn visit_none(self) -> Result + where + E: de::Error, + { + Ok(None) + } + + fn visit_newtype_struct(self, deserializer: D) -> Result + where + D: Deserializer<'de>, + { + deserializer.deserialize_option(self) + } + } + + deserializer.deserialize_newtype_struct("BytesRegex", Visitor) + } +} #[cfg(feature = "uuid")] duper_serde_module!(DuperUuid, DuperOptionUuid, ::uuid::Uuid, "Uuid"); diff --git a/serde-duper/tests/types_option_none.rs b/serde-duper/tests/types_option_none.rs index 0d41cd2..cf6d3bc 100644 --- a/serde-duper/tests/types_option_none.rs +++ b/serde-duper/tests/types_option_none.rs @@ -553,18 +553,26 @@ fn none_ipnet() { #[test] #[cfg(feature = "regex")] fn none_regex() { - use regex::Regex; - use serde_duper::types::DuperOptionRegex; + use regex::{Regex, bytes::Regex as BytesRegex}; + use serde_duper::types::{DuperOptionBytesRegex, DuperOptionRegex}; #[derive(Debug, Serialize, Deserialize)] struct Test { #[serde(with = "DuperOptionRegex")] pattern: Option, + #[serde(with = "DuperOptionBytesRegex")] + bytes_pattern: Option, } - let value = Test { pattern: None }; + let value = Test { + pattern: None, + bytes_pattern: None, + }; let serialized = serde_duper::to_string(&value).unwrap(); - assert_eq!(serialized, r##"Test({pattern: Regex(null)})"##); + assert_eq!( + serialized, + r"Test({pattern: Regex(null), bytes_pattern: BytesRegex(null)})" + ); let deserialized: Test = serde_duper::from_string(&serialized).unwrap(); assert!(deserialized.pattern.is_none()); diff --git a/serde-duper/tests/types_option_some.rs b/serde-duper/tests/types_option_some.rs index 6dfd52e..463b61e 100644 --- a/serde-duper/tests/types_option_some.rs +++ b/serde-duper/tests/types_option_some.rs @@ -722,22 +722,25 @@ fn some_ipnet() { #[test] #[cfg(feature = "regex")] fn some_regex() { - use regex::Regex; - use serde_duper::types::DuperOptionRegex; + use regex::{Regex, bytes::Regex as BytesRegex}; + use serde_duper::types::{DuperOptionBytesRegex, DuperOptionRegex}; #[derive(Debug, Serialize, Deserialize)] struct Test { #[serde(with = "DuperOptionRegex")] pattern: Option, + #[serde(with = "DuperOptionBytesRegex")] + bytes_pattern: Option, } let value = Test { pattern: Some(Regex::new(r"Hello (?\w+)!").unwrap()), + bytes_pattern: Some(BytesRegex::new("\x1b\\[\\dm").unwrap()), }; let serialized = serde_duper::to_string(&value).unwrap(); assert_eq!( serialized, - r##"Test({pattern: Regex(r"Hello (?\w+)!")})"## + r#"Test({pattern: Regex(r"Hello (?\w+)!"), bytes_pattern: BytesRegex("\x1b\\[\\dm")})"# ); let deserialized: Test = serde_duper::from_string(&serialized).unwrap(); diff --git a/serde-duper/tests/types_plain.rs b/serde-duper/tests/types_plain.rs index 5a8d046..ddd2854 100644 --- a/serde-duper/tests/types_plain.rs +++ b/serde-duper/tests/types_plain.rs @@ -642,26 +642,33 @@ fn ipnet() { #[test] #[cfg(feature = "regex")] fn regex() { - use regex::Regex; - use serde_duper::types::DuperRegex; + use regex::{Regex, bytes::Regex as BytesRegex}; + use serde_duper::types::{DuperBytesRegex, DuperRegex}; #[derive(Debug, Serialize, Deserialize)] struct Test { #[serde(with = "DuperRegex")] pattern: Regex, + #[serde(with = "DuperBytesRegex")] + bytes_pattern: BytesRegex, } let value = Test { pattern: Regex::new(r"Hello (?\w+)!").unwrap(), + bytes_pattern: BytesRegex::new("\x1b\\[\\dm").unwrap(), }; let serialized = serde_duper::to_string(&value).unwrap(); assert_eq!( serialized, - r#"Test({pattern: Regex(r"Hello (?\w+)!")})"# + r#"Test({pattern: Regex(r"Hello (?\w+)!"), bytes_pattern: BytesRegex("\x1b\\[\\dm")})"# ); let deserialized: Test = serde_duper::from_string(&serialized).unwrap(); assert_eq!(value.pattern.as_str(), deserialized.pattern.as_str()); + assert_eq!( + value.bytes_pattern.as_str(), + deserialized.bytes_pattern.as_str() + ); } #[test] -- 2.51.2