diff --git a/Cargo.lock b/Cargo.lock index c421d9e6d..f8b0da74c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -5181,12 +5181,14 @@ dependencies = [ "knot-cob", "knot-cobs", "knot-git", + "knot-resource", "knot-runtime", "knot-types", "lasso", "proptest", "scc", "serde", + "serde_ipld_dagcbor", "tempfile", "thiserror 2.0.18", "tokio", @@ -5416,6 +5418,7 @@ dependencies = [ "knot-git", "knot-runtime", "knot-types", + "proptest", "serde", "serde_ipld_dagcbor", "serde_json", @@ -5604,12 +5607,16 @@ dependencies = [ name = "knot-types" version = "2.0.0" dependencies = [ + "cid", "gix-hash", "http", + "ipld-core", "jacquard-common", "proptest", "serde", + "serde_ipld_dagcbor", "serde_json", + "sha2 0.11.0", "thiserror 2.0.18", "trusted-proxies", "url", diff --git a/Cargo.toml b/Cargo.toml index 57ca3edf3..0f83fbdf5 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -127,6 +127,7 @@ itertools = "0.14" serde = { version = "1", features = ["derive"] } serde_json = { version = "1", features = ["raw_value"] } serde_ipld_dagcbor = "0.6" +ipld-core = { version = "0.4", features = ["serde"] } serde_bytes = "0.11" tar = "0.4" object_store = { version = "0.12", features = ["aws"] } diff --git a/knot2/crates/knot-types/Cargo.toml b/knot2/crates/knot-types/Cargo.toml index 9cf085536..93d2f682e 100644 --- a/knot2/crates/knot-types/Cargo.toml +++ b/knot2/crates/knot-types/Cargo.toml @@ -6,12 +6,16 @@ rust-version.workspace = true license.workspace = true [dependencies] +cid = { workspace = true } jacquard-common = { workspace = true } gix-hash = { workspace = true } +ipld-core = { workspace = true } http = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } +serde_ipld_dagcbor = { workspace = true } thiserror = { workspace = true } +sha2 = { workspace = true } trusted-proxies = { workspace = true } url = { workspace = true } diff --git a/knot2/crates/knot-types/src/ids.rs b/knot2/crates/knot-types/src/ids.rs index e3ab540f3..fce189b64 100644 --- a/knot2/crates/knot-types/src/ids.rs +++ b/knot2/crates/knot-types/src/ids.rs @@ -2,7 +2,7 @@ use std::fmt; use std::str::FromStr; use jacquard_common::types::crypto::{PublicKey, multikey}; -use jacquard_common::types::string::{Did as SpecDid, Handle, Nsid, Rkey}; +use jacquard_common::types::string::{Did as SpecDid, Handle, Nsid, Rkey, Tid}; use serde::{Deserialize, Serialize}; #[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] @@ -240,6 +240,22 @@ fn parse_rkey(value: &str) -> Option> { Rkey::new_owned(value).ok() } +fn parse_tid(value: &str) -> Option { + Tid::new(value).ok() +} + +fn parse_record_collection(value: &str) -> Option> { + let nsid = parse_nsid(value)?; + + // I try to capture why this isn't just me dicking around, in the test + // a_collection_that_would_overrun_an_mst_key_isnt_a_record_collection. + // Our rkeys are always gonna be 13 bytes, and an MST is not gonna like a key over 256 bytes, + // so we can handle those two things nicely here instead of downstream having to deal with + // len error later. + let room = crate::record::MST_KEY_MAX_BYTES - crate::record::RECORD_RKEY_BYTES - 1; + (nsid.as_str().len() <= room).then_some(nsid) +} + fn parse_did_rkey(value: &str) -> Option { let did = parse_did(value)?; Rkey::::new_owned(did.as_str()).ok()?; @@ -367,6 +383,8 @@ string_id!(RepoRkey, "repo record key", via parse_rkey => Rkey); string_id!(DidRkey, "DID record key", via parse_did_rkey => Did); string_id!(RefName, "ref name", parse_ref_name); string_id!(TypeName, "COB type name", via parse_nsid => Nsid); +string_id!(RecordCollection, "record collection", via parse_record_collection => Nsid); +string_id!(RecordRkey, "record key", via parse_tid => Tid); string_id!(ActorId, "actor public key", parse_multikey); string_id!(KnotHostname, "knot hostname", parse_knot_hostname); string_id!(BranchName, "branch name", parse_branch_name); @@ -383,6 +401,11 @@ impl From<&DidRkey> for RepoDid { } } +impl RecordRkey { + pub fn minted(tid: Tid) -> Self { + Self(tid) + } +} #[derive(Debug, Clone, Default, PartialEq, Eq, serde::Serialize)] #[serde(transparent)] pub struct PushOptions(Vec); diff --git a/knot2/crates/knot-types/src/lib.rs b/knot2/crates/knot-types/src/lib.rs index 40afde127..07610ad5b 100644 --- a/knot2/crates/knot-types/src/lib.rs +++ b/knot2/crates/knot-types/src/lib.rs @@ -9,12 +9,22 @@ pub use ids::{ AccountDid, ActorId, AppviewEndpoint, AuthorName, BranchName, ChangeId, CiLogsAddr, ClonePath, CobId, DidRkey, Email, HttpStatus, KnotHostname, KnotId, KnotServiceUrl, LanguageBytes, LanguageName, LogsHost, LogsPort, ObjectCount, ObjectFormat, OfferedKey, Oid, OriginUrl, - OwnerDid, OwnerRef, ParseError, PushOption, PushOptions, RefName, RefTransition, RepoDid, - RepoName, RepoPath, RepoRkey, ServiceDid, TagName, TypeName, UnixMicros, UnixSeconds, + OwnerDid, OwnerRef, ParseError, PushOption, PushOptions, RecordCollection, RecordRkey, RefName, + RefTransition, RepoDid, RepoName, RepoPath, RepoRkey, ServiceDid, TagName, TypeName, + UnixMicros, UnixSeconds, +}; + +mod record; +pub use record::{ + BodyRef, MAX_RECORD_BYTES, MstKey, MstKeyTooLong, RecordAddress, RecordBody, RecordBodyError, + RecordCid, SubjectNumber, dag_cbor_cid, }; mod policy; -pub use policy::AdmissionPolicy; +pub use policy::{ + AdmissionPolicy, Blocked, ContributionPermission, ContributionPolicy, Decision, Membership, + RepoPolicy, RepoRole, +}; mod hex; pub use hex::{decode_hex, lowercase_hex}; diff --git a/knot2/crates/knot-types/src/newtype.rs b/knot2/crates/knot-types/src/newtype.rs index 901a397e1..a5c107423 100644 --- a/knot2/crates/knot-types/src/newtype.rs +++ b/knot2/crates/knot-types/src/newtype.rs @@ -55,7 +55,6 @@ macro_rules! scalar_order { } }; } - #[macro_export] macro_rules! text_mode { (verbatim, $value:expr) => { diff --git a/knot2/crates/knot-types/src/record.rs b/knot2/crates/knot-types/src/record.rs new file mode 100644 index 000000000..cff0b67e1 --- /dev/null +++ b/knot2/crates/knot-types/src/record.rs @@ -0,0 +1,424 @@ +use std::fmt; +use std::num::NonZeroU32; + +use cid::Cid as BlockCid; +use cid::multihash::Multihash; +use ipld_core::ipld::Ipld; +use jacquard_common::types::crypto::{DAG_CBOR, SHA2_256}; +use jacquard_common::types::string::AtUri; +use serde::{Deserialize, Serialize}; +use sha2::{Digest, Sha256}; + +use crate::ids::{RecordCollection, RecordRkey, RepoDid}; + +pub(crate) const MST_KEY_MAX_BYTES: usize = 256; + +pub(crate) const RECORD_RKEY_BYTES: usize = 13; + +pub const MAX_RECORD_BYTES: usize = 1_000_000; + +const SHA2_256_BYTES: usize = 32; + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +pub struct RecordAddress { + collection: RecordCollection, + rkey: RecordRkey, +} + +impl RecordAddress { + pub const fn new(collection: RecordCollection, rkey: RecordRkey) -> Self { + Self { collection, rkey } + } + + pub const fn collection(&self) -> &RecordCollection { + &self.collection + } + + pub const fn rkey(&self) -> &RecordRkey { + &self.rkey + } + + pub fn mst_key(&self) -> MstKey { + MstKey(format!( + "{}/{}", + self.collection.as_str(), + self.rkey.as_str() + )) + } + + pub fn at_uri(&self, repo: &RepoDid) -> AtUri { + AtUri::from_parts_owned(repo.as_str(), self.collection.as_str(), self.rkey.as_str()) + .expect("Repo DID, collection nsid and TID compose at-uri") + } +} + +impl fmt::Display for RecordAddress { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.pad(self.mst_key().as_str()) + } +} + +#[derive(Debug, thiserror::Error)] +#[error("Mst key of {0} bytes is over {MST_KEY_MAX_BYTES} byte limit")] +pub struct MstKeyTooLong(pub usize); + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub struct MstKey(String); + +impl MstKey { + pub fn new(key: String) -> Result { + if key.len() > MST_KEY_MAX_BYTES { + Err(MstKeyTooLong(key.len())) + } else { + Ok(Self(key)) + } + } + + pub fn as_str(&self) -> &str { + &self.0 + } +} + +impl std::borrow::Borrow for MstKey { + fn borrow(&self) -> &str { + &self.0 + } +} + +impl fmt::Display for MstKey { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.pad(&self.0) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct SubjectNumber(NonZeroU32); + +impl SubjectNumber { + pub const FIRST: Self = Self(NonZeroU32::MIN); + + pub const fn new(value: u32) -> Option { + match NonZeroU32::new(value) { + Some(value) => Some(Self(value)), + None => None, + } + } + + pub const fn get(self) -> u32 { + self.0.get() + } + + pub const fn after(self) -> Option { + match self.0.checked_add(1) { + Some(next) => Some(Self(next)), + None => None, + } + } +} + +impl fmt::Display for SubjectNumber { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!(f, "#{}", self.0.get()) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)] +#[serde(transparent)] +pub struct BodyRef(BlockCid); + +impl BodyRef { + pub fn new(cid: BlockCid) -> Option { + RecordCid::new(cid).is_some().then_some(Self(cid)) + } + + pub const fn cid(self) -> BlockCid { + self.0 + } +} + +impl fmt::Display for BodyRef { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(&self.0, f) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub struct RecordCid([u8; SHA2_256_BYTES]); + +impl RecordCid { + pub fn new(cid: BlockCid) -> Option { + let hash = cid.hash(); + let dag_cbor_sha256 = cid.codec() == DAG_CBOR && hash.code() == SHA2_256; + dag_cbor_sha256 + .then(|| <[u8; SHA2_256_BYTES]>::try_from(hash.digest()).ok()) + .flatten() + .map(Self) + } + + pub fn of(body: &RecordBody) -> Self { + Self(Sha256::digest(body.as_bytes()).into()) + } + + pub fn cid(self) -> BlockCid { + BlockCid::new_v1(DAG_CBOR, sha2_256_multihash(&self.0)) + } +} + +impl Serialize for RecordCid { + fn serialize(&self, serializer: S) -> Result { + self.cid().serialize(serializer) + } +} + +fn sha2_256_multihash(digest: &[u8; SHA2_256_BYTES]) -> Multihash<64> { + Multihash::<64>::wrap(SHA2_256, digest).expect("Sha2-256 digest fits 64 byte multihash") +} + +pub fn dag_cbor_cid(block: &[u8]) -> BlockCid { + BlockCid::new_v1(DAG_CBOR, sha2_256_multihash(&Sha256::digest(block).into())) +} + +impl<'de> Deserialize<'de> for RecordCid { + fn deserialize>(deserializer: D) -> Result { + let cid = BlockCid::deserialize(deserializer)?; + Self::new(cid).ok_or_else(|| { + serde::de::Error::custom("Record CID names dag-cbor bytes hashed with sha2-256") + }) + } +} + +#[derive(Debug, thiserror::Error)] +pub enum RecordBodyError { + #[error("Record body is dag-cbor and these bytes aren't: {0}")] + NotDagCbor(String), + #[error( + "record body is canonical dag-cbor, and these {0} bytes aren't canonical encoding of their value" + )] + NotCanonical(usize), + #[error("Record body of {0} bytes is over {MAX_RECORD_BYTES} byte limit")] + Oversized(usize), +} + +#[derive(Debug, Clone, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub struct RecordBody(Vec); + +impl RecordBody { + pub fn new(bytes: Vec) -> Result { + if bytes.len() > MAX_RECORD_BYTES { + return Err(RecordBodyError::Oversized(bytes.len())); + } + let value: Ipld = serde_ipld_dagcbor::from_slice(&bytes) + .map_err(|error| RecordBodyError::NotDagCbor(error.to_string()))?; + let canonical = serde_ipld_dagcbor::to_vec(&value) + .map_err(|error| RecordBodyError::NotDagCbor(error.to_string()))?; + if canonical == bytes { + Ok(Self(bytes)) + } else { + Err(RecordBodyError::NotCanonical(bytes.len())) + } + } + + pub fn as_bytes(&self) -> &[u8] { + &self.0 + } + + pub fn into_bytes(self) -> Vec { + self.0 + } + + pub fn len(&self) -> usize { + self.0.len() + } + + pub fn is_empty(&self) -> bool { + self.0.is_empty() + } + + pub fn cid(&self) -> RecordCid { + RecordCid::of(self) + } +} + +impl fmt::Display for RecordCid { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + fmt::Display::fmt(&self.cid(), f) + } +} + +#[cfg(test)] +mod tests { + use super::{ + MAX_RECORD_BYTES, MST_KEY_MAX_BYTES, RecordAddress, RecordBody, RecordBodyError, RecordCid, + SubjectNumber, + }; + use crate::ids::{RecordCollection, RecordRkey, RepoDid}; + + fn address() -> RecordAddress { + RecordAddress::new( + RecordCollection::new("sh.tangled.repo.issue").unwrap(), + RecordRkey::new("3lubrptx57d22").unwrap(), + ) + } + + #[test] + fn an_address_renders_the_mst_key_and_the_at_uri_of_one_repo() { + let repo = RepoDid::new("did:plc:squid").unwrap(); + assert_eq!( + address().mst_key().as_str(), + "sh.tangled.repo.issue/3lubrptx57d22" + ); + assert_eq!( + address().at_uri(&repo).as_str(), + "at://did:plc:squid/sh.tangled.repo.issue/3lubrptx57d22", + "At-uri client resolves is an address under repository DID" + ); + } + + #[test] + fn a_record_key_parses_only_a_tid() { + assert!( + RecordRkey::new("self").is_err(), + "Knot-minted record key is TID, and `self` is the shape DidRkey takes" + ); + assert!(RecordRkey::new("did:plc:squid").is_err()); + assert!(RecordRkey::new("3lubrptx57d2").is_err()); + } + + #[test] + fn a_subject_number_starts_at_one_and_stops_before_it_repeats_the_last_one() { + assert_eq!(SubjectNumber::FIRST.get(), 1); + assert_eq!(SubjectNumber::FIRST.after(), SubjectNumber::new(2)); + assert!(SubjectNumber::new(7) > SubjectNumber::new(1)); + assert_eq!(SubjectNumber::new(7).unwrap().to_string(), "#7"); + assert_eq!( + SubjectNumber::new(0), + None, + "first subject is #1. zero is sentinel somebody smuggled in" + ); + assert!( + serde_json::from_str::("0").is_err(), + "number off wire goes through the same door, as one built in process" + ); + assert_eq!( + serde_json::from_str::("7").unwrap(), + SubjectNumber::new(7).unwrap() + ); + assert_eq!( + SubjectNumber::new(u32::MAX).unwrap().after(), + None, + "Count runs out instead of handing the last number out twice" + ); + } + + #[test] + fn a_collection_that_would_overrun_an_mst_key_isnt_a_record_collection() { + let room = MST_KEY_MAX_BYTES - super::RECORD_RKEY_BYTES - 1; + let segments: Vec = (0..4).map(|_| "a".repeat(57)).collect(); + let longest = format!("sh.tangled.{}", segments.join(".")); + assert_eq!(longest.len(), room); + let collection = RecordCollection::new(&longest).expect("Longest key still fits"); + assert_eq!( + RecordAddress::new(collection, RecordRkey::new("3lubrptx57d22").unwrap()) + .mst_key() + .as_str() + .len(), + MST_KEY_MAX_BYTES, + "Bound is MST's own maximum key length, and this address is exactly that long" + ); + assert!( + RecordCollection::new(format!("{longest}a")).is_err(), + "objects never end up with one, addresses MST would refuse can't be built" + ); + } + + #[test] + fn a_record_body_parses_before_a_cid_labels_it_dag_cbor() { + let body = RecordBody::new( + serde_ipld_dagcbor::to_vec(&std::collections::BTreeMap::from([("text", "kelp")])) + .unwrap(), + ) + .expect("dag-cbor serializer wrote parses back"); + assert_eq!(body.cid(), RecordCid::of(&body)); + assert!( + matches!( + RecordBody::new(vec![0; MAX_RECORD_BYTES + 1]), + Err(RecordBodyError::Oversized(_)) + ), + "Body over the limit is turned away before it is decoded, whatever its bytes are" + ); + assert!( + RecordBody::new(vec![0xff, 0x00]).is_err(), + "bytes parse before digest runs, label on CID claims them" + ); + assert!( + RecordBody::new([body.as_bytes(), b"trailing"].concat()).is_err(), + "trailing bytes don't decode, dag-cbor is one complete value" + ); + } + + fn door_out_of(bytes: &[u8]) -> &'static str { + match RecordBody::new(bytes.to_vec()) { + Ok(_) => "accepted", + Err(RecordBodyError::NotDagCbor(_)) => "not dag-cbor", + Err(RecordBodyError::NotCanonical(_)) => "not canonical", + Err(RecordBodyError::Oversized(_)) => "oversized", + } + } + + #[test] + fn dag_cbor_that_isnt_canonical_is_no_record_body() { + let cases: &[(&str, &str, &[u8])] = &[ + ("not canonical", "Non-minimal integer width", &[0x18, 0x05]), + ( + "not dag-cbor", + "an indefinite-length array", + &[0x9f, 0x01, 0xff], + ), + ( + "not dag-cbor", + "an indefinite-length string", + &[0x7f, 0x61, 0x61, 0xff], + ), + ( + "not canonical", + "map keys out of order", + &[0xa2, 0x61, 0x62, 0x01, 0x61, 0x61, 0x01], + ), + ( + "not dag-cbor", + "Duplicated map key", + &[0xa2, 0x61, 0x61, 0x01, 0x61, 0x61, 0x02], + ), + ]; + cases.iter().for_each(|(door, why, bytes)| { + assert_eq!( + door_out_of(bytes), + *door, + "would give CID naming bytes nobody else reproduces, that mr cid leaves through {door}" + ); + }); + assert_eq!( + door_out_of(&[0x05]), + "accepted", + "canonical spelling, of the same integer, is the one that stands" + ); + } + + #[test] + fn a_record_cid_with_the_wrong_codec_is_refused() { + let body = RecordBody::new(serde_ipld_dagcbor::to_vec(&"kelp").unwrap()).unwrap(); + let raw = cid::Cid::new_v1(0x55, body.cid().cid().hash().to_owned()); + assert_eq!( + RecordCid::new(raw), + None, + "A raw block is not a record, however well it hashes" + ); + assert_eq!(RecordCid::new(body.cid().cid()), Some(body.cid())); + let encoded = serde_ipld_dagcbor::to_vec(&raw).unwrap(); + assert!( + serde_ipld_dagcbor::from_slice::(&encoded).is_err(), + "Wire goes through the same check the constructor makes" + ); + } +}