diff --git a/blob-stream-volume/src/summarize.rs b/blob-stream-volume/src/summarize.rs index 781abf7..e16ae7a 100644 --- a/blob-stream-volume/src/summarize.rs +++ b/blob-stream-volume/src/summarize.rs @@ -37,7 +37,10 @@ pub fn run(path: &Path) -> Result<(), Error> { } fn print_summary(s: &Snapshot) { - println!("blob-stream-volume summary ({}, schema {})", s.generated_at_iso, s.schema); + println!( + "blob-stream-volume summary ({}, schema {})", + s.generated_at_iso, s.schema + ); println!(); // ---- time window ---- @@ -205,7 +208,10 @@ fn print_summary(s: &Snapshot) { let mut mimes: Vec<(&String, &crate::MimeStats)> = s.per_mime.iter().collect(); mimes.sort_by_key(|(_, b)| std::cmp::Reverse(b.total_size)); println!("mime types (by declared blob bytes)"); - println!(" {:<30} {:>10} {:>14} {:>8}", "mime", "blobs", "total bytes", "share"); + println!( + " {:<30} {:>10} {:>14} {:>8}", + "mime", "blobs", "total bytes", "share" + ); for (mime, ms) in &mimes { let pct = 100.0 * (ms.total_size as f64) / (total_blob_bytes as f64); println!( diff --git a/hubble-pds/src/encoding.rs b/hubble-pds/src/encoding.rs index a46c1d4..83b9c85 100644 --- a/hubble-pds/src/encoding.rs +++ b/hubble-pds/src/encoding.rs @@ -25,8 +25,9 @@ pub enum KnownEncoding { Zstd, } -impl KnownEncoding { - pub fn from_str(s: &str) -> Result { +impl std::str::FromStr for KnownEncoding { + type Err = RequestEncodingError; + fn from_str(s: &str) -> Result { match s.trim().to_ascii_lowercase().as_str() { "br" => Ok(Self::Brotli), "deflate" => Ok(Self::Deflate), @@ -35,7 +36,9 @@ impl KnownEncoding { _ => Err(RequestEncodingError::Unsupported(s.to_string())), } } +} +impl KnownEncoding { /// Read the `Content-Encoding` header (if present) and parse it. /// Returns `Ok(None)` when the header is absent. /// Errors on non-ASCII header bytes or an unrecognized encoding name. @@ -46,7 +49,7 @@ impl KnownEncoding { None => Ok(None), Some(v) => { let s = v.to_str()?; - Ok(Some(Self::from_str(s)?)) + Ok(Some(s.parse()?)) } } } diff --git a/hubble-pds/src/hostname.rs b/hubble-pds/src/hostname.rs index 0ed69cd..06d49f0 100644 --- a/hubble-pds/src/hostname.rs +++ b/hubble-pds/src/hostname.rs @@ -120,7 +120,11 @@ mod tests { #[test] fn is_bsky_recognizes_host_subdomains() { - assert!(Hostname::new("morel.us-east.host.bsky.network").unwrap().is_bsky()); + assert!( + Hostname::new("morel.us-east.host.bsky.network") + .unwrap() + .is_bsky() + ); assert!(!Hostname::new("bsky.network").unwrap().is_bsky()); assert!(!Hostname::new("example.com").unwrap().is_bsky()); } diff --git a/mini/src/check_did.rs b/mini/src/check_did.rs index a94dac2..aafd0c3 100644 --- a/mini/src/check_did.rs +++ b/mini/src/check_did.rs @@ -43,6 +43,7 @@ pub struct HostResult { } #[derive(Debug)] +#[expect(clippy::large_enum_variant, reason = "verified is the normal case")] pub enum PerHost { /// MST verification ran; see `Outcome.matched` for the verdict. Verified(Outcome), diff --git a/mini/src/host_summary.rs b/mini/src/host_summary.rs index 5ff4d09..6f00bce 100644 --- a/mini/src/host_summary.rs +++ b/mini/src/host_summary.rs @@ -60,11 +60,7 @@ impl Summary { /// Sum of all five buckets. Equal to the count of R/ rows for the /// host (every row landed in exactly one bucket). pub fn total_seen(&self) -> u64 { - self.captured - + self.seen_uncaptured - + self.failed - + self.inactive - + self.tombstoned + self.captured + self.seen_uncaptured + self.failed + self.inactive + self.tombstoned } } @@ -183,8 +179,14 @@ mod tests { .unwrap(); let mut batch = WriteBatch::default(); commit::put_into(&mut batch, &db, &pds, "did:plc:cap", &sample_commit()).unwrap(); - repo_stats::put_into(&mut batch, &db, &pds, "did:plc:cap", &sample_stats(100, 400, 5)) - .unwrap(); + repo_stats::put_into( + &mut batch, + &db, + &pds, + "did:plc:cap", + &sample_stats(100, 400, 5), + ) + .unwrap(); db.inner.write(batch).unwrap(); // inactive(deactivated): listrepos_active=false. @@ -231,8 +233,9 @@ mod tests { let (db, _g) = open_temporary(); let pds = h("a.example"); - for (did, w, c, recs) in - [("did:plc:1", 100, 400, 5), ("did:plc:2", 50, 250, 3)].iter().copied() + for (did, w, c, recs) in [("did:plc:1", 100, 400, 5), ("did:plc:2", 50, 250, 3)] + .iter() + .copied() { repo_state::put_fetched(&db, &pds, did, &r(true, false, None, None, false)).unwrap(); let mut batch = WriteBatch::default(); diff --git a/mini/src/main.rs b/mini/src/main.rs index 07ab351..46e1c9a 100644 --- a/mini/src/main.rs +++ b/mini/src/main.rs @@ -11,8 +11,8 @@ mod export_repo; mod find_pdses; mod host_summary; mod pds; -mod rlimits; mod repo; +mod rlimits; mod snapshot_repos; mod storage; mod summarize_stats; diff --git a/mini/src/rlimits.rs b/mini/src/rlimits.rs index a0104e9..ab6fd1b 100644 --- a/mini/src/rlimits.rs +++ b/mini/src/rlimits.rs @@ -36,7 +36,12 @@ pub fn ensure_at_least(target: u64) -> Result<(), std::io::Error> { let new_soft = target.min(hard); if new_soft > soft { Resource::NOFILE.set(new_soft, hard)?; - tracing::info!(from = soft, to = new_soft, hard, "raised RLIMIT_NOFILE soft cap"); + tracing::info!( + from = soft, + to = new_soft, + hard, + "raised RLIMIT_NOFILE soft cap" + ); } if new_soft < target { tracing::warn!( diff --git a/mini/src/snapshot_repos.rs b/mini/src/snapshot_repos.rs index 095a157..6684abe 100644 --- a/mini/src/snapshot_repos.rs +++ b/mini/src/snapshot_repos.rs @@ -751,8 +751,7 @@ fn spawn_progress_ticker(progress: Arc) -> tokio::task::JoinHandle<()> /// spill headroom for the tier-3 fjall instances. Idempotent w.r.t. the /// baseline call — only raises further if `big_repos_limit` needs it. fn ensure_nofile_for_big_repos(big_repos_limit: usize) -> Result<(), Error> { - let target = - rlimits::NOFILE_BASELINE + big_repos_limit as u64 * rlimits::NOFILE_PER_BIG_REPO; + let target = rlimits::NOFILE_BASELINE + big_repos_limit as u64 * rlimits::NOFILE_PER_BIG_REPO; rlimits::ensure_at_least(target).map_err(Error::Rlimit) } diff --git a/mini/src/storage/repo_state.rs b/mini/src/storage/repo_state.rs index 234e387..f9f8937 100644 --- a/mini/src/storage/repo_state.rs +++ b/mini/src/storage/repo_state.rs @@ -422,9 +422,15 @@ mod tests { let mut got = scan_all(&db).unwrap(); got.sort_by(|x, y| (&x.pds, &x.did).cmp(&(&y.pds, &y.did))); assert_eq!(got.len(), 3); - assert_eq!((got[0].pds.as_str(), got[0].did.as_str()), ("a.example", "did:plc:1")); + assert_eq!( + (got[0].pds.as_str(), got[0].did.as_str()), + ("a.example", "did:plc:1") + ); assert_eq!(got[0].state.last_seen_rev, "r1"); - assert_eq!((got[2].pds.as_str(), got[2].did.as_str()), ("b.example", "did:plc:2")); + assert_eq!( + (got[2].pds.as_str(), got[2].did.as_str()), + ("b.example", "did:plc:2") + ); } #[test] @@ -493,7 +499,10 @@ mod tests { // Preserved from prior: assert_eq!(got.last_fetched_at_unix, 1_700_000_000); assert_eq!(got.last_error, Some("car too large".into())); - assert!(got.expects_big, "expects_big must survive phase-1 eager write"); + assert!( + got.expects_big, + "expects_big must survive phase-1 eager write" + ); // Always reset to false on reappearance. assert!(!got.tombstoned); } diff --git a/space-efficiency-check/examples/inspect.rs b/space-efficiency-check/examples/inspect.rs index 6976efd..e33b007 100644 --- a/space-efficiency-check/examples/inspect.rs +++ b/space-efficiency-check/examples/inspect.rs @@ -1,4 +1,4 @@ -use rocksdb::{Options, DB}; +use rocksdb::{DB, Options}; use std::path::PathBuf; fn main() { diff --git a/space-efficiency-check/examples/intern.rs b/space-efficiency-check/examples/intern.rs index 6ae412c..d22994b 100644 --- a/space-efficiency-check/examples/intern.rs +++ b/space-efficiency-check/examples/intern.rs @@ -1,6 +1,6 @@ use clap::Parser; use rocksdb::{ - BlockBasedOptions, CompactOptions, DBCompressionType, Options, WriteBatch, WriteOptions, DB, + BlockBasedOptions, CompactOptions, DB, DBCompressionType, Options, WriteBatch, WriteOptions, }; use std::collections::HashMap; use std::path::PathBuf; diff --git a/space-efficiency-check/examples/other.rs b/space-efficiency-check/examples/other.rs index 0a58fa6..7831cd1 100644 --- a/space-efficiency-check/examples/other.rs +++ b/space-efficiency-check/examples/other.rs @@ -1,5 +1,7 @@ use clap::Parser; -use rocksdb::{BlockBasedOptions, CompactOptions, DBCompressionType, Options, WriteBatch, WriteOptions, DB}; +use rocksdb::{ + BlockBasedOptions, CompactOptions, DB, DBCompressionType, Options, WriteBatch, WriteOptions, +}; use std::path::PathBuf; use std::time::Instant; @@ -72,8 +74,8 @@ fn build_options(c: &BenchConfig, for_flush: bool) -> Options { fn run_flush(source_path: &PathBuf, configs: &[BenchConfig]) { let source_opts = Options::default(); - let source = DB::open_for_read_only(&source_opts, source_path, false) - .expect("failed to open source db"); + let source = + DB::open_for_read_only(&source_opts, source_path, false).expect("failed to open source db"); // pre-read all keys+values into memory so source I/O doesn't affect timing eprintln!("pre-reading source data into memory..."); @@ -101,10 +103,7 @@ fn run_flush(source_path: &PathBuf, configs: &[BenchConfig]) { println!("mode,compression,block_size,entries,data_mib,time_secs,throughput_mib_s"); for config in configs { - let dest_path = source_path - .parent() - .unwrap() - .join("bench-flush-tmp"); + let dest_path = source_path.parent().unwrap().join("bench-flush-tmp"); let _ = std::fs::remove_dir_all(&dest_path); let opts = build_options(config, true); @@ -181,8 +180,8 @@ fn run_scan(db_path: &PathBuf, configs: &[BenchConfig]) { // reopen read-only and scan let read_opts = Options::default(); - let db = DB::open_for_read_only(&read_opts, db_path, false) - .expect("failed to open db for scan"); + let db = + DB::open_for_read_only(&read_opts, db_path, false).expect("failed to open db for scan"); let start = Instant::now(); let mut iter = db.raw_iterator(); diff --git a/space-efficiency-check/examples/sample.rs b/space-efficiency-check/examples/sample.rs index 6484c56..4407a14 100644 --- a/space-efficiency-check/examples/sample.rs +++ b/space-efficiency-check/examples/sample.rs @@ -1,5 +1,5 @@ use clap::Parser; -use rocksdb::{Options, WriteBatch, WriteOptions, DB}; +use rocksdb::{DB, Options, WriteBatch, WriteOptions}; use std::path::PathBuf; use std::time::Instant; @@ -27,8 +27,8 @@ fn main() { // open source read-only to get file list let source_opts = Options::default(); - let source = - DB::open_for_read_only(&source_opts, &args.source, false).expect("failed to open source db"); + let source = DB::open_for_read_only(&source_opts, &args.source, false) + .expect("failed to open source db"); let mut files = source.live_files().expect("failed to get live files"); files.sort_by(|a, b| a.start_key.cmp(&b.start_key)); @@ -93,8 +93,11 @@ fn main() { total_written += 1; if batch_count >= 32_768 { - dest.write_opt(std::mem::replace(&mut batch, WriteBatch::default()), &write_opts) - .expect("failed to write batch"); + dest.write_opt( + std::mem::replace(&mut batch, WriteBatch::default()), + &write_opts, + ) + .expect("failed to write batch"); batch_count = 0; } diff --git a/space-efficiency-check/examples/stats.rs b/space-efficiency-check/examples/stats.rs index 5ce69a7..4a9a153 100644 --- a/space-efficiency-check/examples/stats.rs +++ b/space-efficiency-check/examples/stats.rs @@ -1,5 +1,5 @@ use clap::Parser; -use rocksdb::{Options, DB}; +use rocksdb::{DB, Options}; use std::collections::HashMap; use std::path::PathBuf; use std::time::Instant; @@ -31,7 +31,11 @@ fn main() { total_keys += 1; // value size histogram (log2 buckets) - let bucket = if value.is_empty() { 0 } else { (value.len() as f64).log2().floor() as u32 }; + let bucket = if value.is_empty() { + 0 + } else { + (value.len() as f64).log2().floor() as u32 + }; *value_size_buckets.entry(bucket).or_default() += 1; // extract DID prefix (everything before first '/') @@ -61,7 +65,11 @@ fn main() { // build keys-per-account histogram (log2 buckets) let mut acct_size_buckets: HashMap = HashMap::new(); for &count in account_keys.values() { - let bucket = if count == 0 { 0 } else { (count as f64).log2().floor() as u32 }; + let bucket = if count == 0 { + 0 + } else { + (count as f64).log2().floor() as u32 + }; *acct_size_buckets.entry(bucket).or_default() += 1; } diff --git a/space-efficiency-check/examples/sweep.rs b/space-efficiency-check/examples/sweep.rs index b3e7860..b7f174e 100644 --- a/space-efficiency-check/examples/sweep.rs +++ b/space-efficiency-check/examples/sweep.rs @@ -1,6 +1,6 @@ use clap::Parser; use csv::{ReaderBuilder, WriterBuilder}; -use rocksdb::{BlockBasedOptions, CompactOptions, DBCompressionType, Options, DB}; +use rocksdb::{BlockBasedOptions, CompactOptions, DB, DBCompressionType, Options}; use serde::{Deserialize, Serialize}; use std::collections::HashSet; use std::path::PathBuf; @@ -176,7 +176,9 @@ fn sweep_configs() -> Vec { for &block_kb in &[4, 8, 16, 32, 64, 128] { for &level in &[-111, -4, 0, 1, 3, 6, 9] { for &(dict, train) in &[(0_i32, None), (16_384, Some(16)), (16_384, Some(128))] { - if level == 0 && dict > 0 { continue; } // lz4 has no dictionary + if level == 0 && dict > 0 { + continue; + } // lz4 has no dictionary configs.push(SweepConfig { block_size: block_kb * 1024, zstd_level: level, @@ -252,7 +254,9 @@ fn sweep_configs() -> Vec { let dict = dict_kb * 1024; for &train_kb in &[256, 1024] { let mult = (train_kb / dict_kb) as i32; - if mult < 1 { continue; } // training must be >= dict size + if mult < 1 { + continue; + } // training must be >= dict size configs.push(SweepConfig { block_size: block_kb * 1024, dict_bytes: dict as i32, @@ -269,7 +273,9 @@ fn sweep_configs() -> Vec { let dict = dict_kb * 1024; for &train_kb in &[256, 1024] { let mult = (train_kb / dict_kb) as i32; - if mult < 1 { continue; } + if mult < 1 { + continue; + } configs.push(SweepConfig { block_size: 32 * 1024, zstd_level: level, diff --git a/space-efficiency-check/src/lib.rs b/space-efficiency-check/src/lib.rs index b5119db..6ddcb9e 100644 --- a/space-efficiency-check/src/lib.rs +++ b/space-efficiency-check/src/lib.rs @@ -13,7 +13,7 @@ pub fn encode_account_status(active: bool, rev: &str, cid: &[u8]) -> Vec { 1 + // active bool rev.len() + 1 + // null sep - cid.len() + cid.len(), ); buf.push(if active { 0x01 } else { 0x00 }); buf.extend_from_slice(rev.as_bytes()); diff --git a/space-efficiency-check/src/main.rs b/space-efficiency-check/src/main.rs index da300d0..5f6c45b 100644 --- a/space-efficiency-check/src/main.rs +++ b/space-efficiency-check/src/main.rs @@ -1,5 +1,5 @@ use clap::Parser; -use rocksdb::{DB, CompactOptions, DBCompressionType, Options}; +use rocksdb::{CompactOptions, DB, DBCompressionType, Options}; use std::path::{Path, PathBuf}; use std::sync::{Arc, atomic::Ordering}; @@ -62,10 +62,18 @@ async fn main() -> Result<(), ProcessError> { if args.just_compact { manually_compact(db)?; - return Ok(()) + return Ok(()); } - let stats = run_workers(&args.car_dir, db.clone(), args.workers, args.mem_limit_mb, args.sample, args.intern).await?; + let stats = run_workers( + &args.car_dir, + db.clone(), + args.workers, + args.mem_limit_mb, + args.sample, + args.intern, + ) + .await?; let repos = stats.repos.load(Ordering::Relaxed); let empty = stats.empty_repos.load(Ordering::Relaxed); diff --git a/space-efficiency-check/src/work.rs b/space-efficiency-check/src/work.rs index d8b58cf..7dc7ae9 100644 --- a/space-efficiency-check/src/work.rs +++ b/space-efficiency-check/src/work.rs @@ -47,7 +47,7 @@ pub struct Stats { /// stable within a Rust version, which is enough for "rerun the same /// sample" across back-to-back runs. fn sampled_in(path: &Path, frac: f64) -> bool { - let name = path.file_name().unwrap_or_else(|| path.as_os_str()); + let name = path.file_name().unwrap_or(path.as_os_str()); let mut hasher = std::hash::DefaultHasher::new(); name.hash(&mut hasher); let h = hasher.finish(); @@ -88,10 +88,10 @@ pub async fn run_workers( // skip this DID if it doesn't fall within the sample fraction. // deterministic: same car_dir + same frac -> same selection. - if let Some(frac) = sample { - if !sampled_in(&entry.path(), frac) { - continue - } + if let Some(frac) = sample + && !sampled_in(&entry.path(), frac) + { + continue; } stats.car_bytes.fetch_add(meta.len(), Ordering::Relaxed); @@ -173,7 +173,7 @@ async fn process_car( // for now we're not writing account keys let mut account_key = Vec::with_capacity(1 + did.len()); account_key.push(PREFIX_ACCOUNT); - account_key.extend_from_slice(&did.as_bytes()); + account_key.extend_from_slice(did.as_bytes()); let account_val = encode_account_status(true, &car.commit.rev, &cid); let db = db.clone(); tokio::task::spawn_blocking(move || { @@ -197,7 +197,7 @@ async fn process_car( let mut key = Vec::with_capacity(prefix.len() + 1 + output.key.len()); key.extend_from_slice(&prefix); key.push(b'/'); - key.extend_from_slice(&output.key.as_bytes()); + key.extend_from_slice(output.key.as_bytes()); batch.put(&key, &output.data); } let db = db.clone(); diff --git a/stats-backfill/src/pds.rs b/stats-backfill/src/pds.rs index 30f192d..3c14e88 100644 --- a/stats-backfill/src/pds.rs +++ b/stats-backfill/src/pds.rs @@ -312,4 +312,3 @@ async fn get_a_lil_json(client: &reqwest::Client, mut base: Url, path: &str) -> ProbeResult::OkJson(normalized) } -