diff --git a/calibre-plugin/main.py b/calibre-plugin/main.py index edcd04e..e82ff55 100644 --- a/calibre-plugin/main.py +++ b/calibre-plugin/main.py @@ -69,7 +69,9 @@ def run_upload(gui): if code == 200: uploaded += 1 elif code == 409: - # Server dedupes by owner + content hash: already uploaded. + # Server dedupes by the EPUB's stable book UUID (re-embedding + # metadata changes the bytes each send, so a content hash would + # miss the re-upload): this book is already on the server. skipped += 1 elif code in (401, 403): # Authentication won't recover mid-run — stop and report. diff --git a/src/db/app_db.rs b/src/db/app_db.rs index ed9d564..fc86169 100644 --- a/src/db/app_db.rs +++ b/src/db/app_db.rs @@ -418,6 +418,21 @@ impl AppDb { self.add_column_if_missing("books", "default_sync", "INTEGER NOT NULL DEFAULT 1") .await?; + // Stable per-book identifier from the uploaded EPUB's `urn:uuid:` + // `` (Calibre writes one). Used to dedupe re-uploads of + // the same book: Calibre re-exports change the file bytes (so the + // content hash differs every send) but keep this identifier constant. + self.add_column_if_missing("books", "source_uuid", "TEXT") + .await?; + sqlx::query( + "CREATE INDEX IF NOT EXISTS idx_books_source_uuid \ + ON books(owner_user_id, source_uuid)", + ) + .execute(&self.pool) + .await + .change_context(Error::Database) + .attach("Failed to create source_uuid index")?; + Ok(()) } diff --git a/src/db/book_store.rs b/src/db/book_store.rs index 066901a..ea4cd32 100644 --- a/src/db/book_store.rs +++ b/src/db/book_store.rs @@ -68,6 +68,9 @@ pub struct NewBook { /// of the preferred format. Upload: `book`. pub file_basename: String, pub content_hash: Option, + /// Stable source identifier for re-upload dedup (upload EPUB's `urn:uuid:` + /// ``). `None` for Calibre-ingested rows. + pub source_uuid: Option, pub has_cover: bool, pub owner_user_id: Option, /// Default sync eligibility for the book, applied to any user without an @@ -608,6 +611,25 @@ impl BookStore { Ok(row.map(|r| r.get::("id"))) } + /// Find an uploaded book by owner + stable source identifier (the EPUB's + /// `urn:uuid:` ``). Catches re-uploads whose bytes changed + /// across a Calibre re-export but whose book identity did not. + pub async fn find_by_source_uuid( + &self, + owner_user_id: i64, + source_uuid: &str, + ) -> Result> { + let row = + sqlx::query("SELECT id FROM books WHERE owner_user_id = ? AND source_uuid = ? LIMIT 1") + .bind(owner_user_id) + .bind(source_uuid) + .fetch_optional(&self.pool) + .await + .change_context(Error::Database) + .attach("Failed to look up book by source uuid")?; + Ok(row.map(|r| r.get::("id"))) + } + // ---- Internal ---- /// Build a `SELECT ... FROM books {tail}` over every column. @@ -649,9 +671,9 @@ const INSERT_BOOK_SQL: &str = r#" INSERT INTO books (id, uuid, title, authors, publisher, language, description, series_name, series_index, pubdate, file_format, file_size, - source, locator, original_filename, content_hash, has_cover, + source, locator, original_filename, content_hash, source_uuid, has_cover, owner_user_id, default_sync, timestamp, last_modified) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) "#; /// Bind a [`NewBook`]'s fields onto the shared insert statement. @@ -677,6 +699,7 @@ fn bind_new_book<'q>( .bind(&book.locator) .bind(&book.file_basename) .bind(&book.content_hash) + .bind(&book.source_uuid) .bind(book.has_cover as i64) .bind(book.owner_user_id) .bind(book.default_sync as i64) diff --git a/src/ingest/calibre.rs b/src/ingest/calibre.rs index 02f38fa..9d87463 100644 --- a/src/ingest/calibre.rs +++ b/src/ingest/calibre.rs @@ -88,6 +88,7 @@ fn new_book_from( locator: path.to_string(), file_basename: format.name.clone(), content_hash: None, + source_uuid: None, has_cover, owner_user_id: None, // Calibre books are syncable by default; a user can still opt out diff --git a/src/upload/epub.rs b/src/upload/epub.rs index bf65550..360ccaf 100644 --- a/src/upload/epub.rs +++ b/src/upload/epub.rs @@ -20,6 +20,10 @@ pub struct EpubMeta { pub series_index: Option, pub pubdate: Option>, pub isbn: Option, + /// Stable per-book identifier from a `urn:uuid:…` value + /// (Calibre writes one). Survives re-export even though the file bytes + /// change, so it is the reliable re-upload dedup key. + pub book_uuid: Option, } /// A parsed EPUB: its metadata plus the raw cover image bytes, if one was found. @@ -208,9 +212,16 @@ fn apply_dc_text(meta: &mut EpubMeta, field: &str, text: String) { "language" if meta.language.is_none() => meta.language = Some(text), "description" if meta.description.is_none() => meta.description = Some(text), "date" if meta.pubdate.is_none() => meta.pubdate = parse_date(&text), - "identifier" if meta.isbn.is_none() => { - if let Some(isbn) = extract_isbn(&text) { - meta.isbn = Some(isbn); + "identifier" => { + if meta.book_uuid.is_none() { + if let Some(uuid) = parse_book_uuid(&text) { + meta.book_uuid = Some(uuid); + } + } + if meta.isbn.is_none() { + if let Some(isbn) = extract_isbn(&text) { + meta.isbn = Some(isbn); + } } } _ => {} @@ -308,6 +319,31 @@ fn parse_date(s: &str) -> Option> { None } +/// Extract a book UUID from a `` value. Calibre writes its stable +/// per-book UUID either bare (`2148815d-…`) or `urn:uuid:`-prefixed; both are +/// accepted. Matching by canonical 8-4-4-4-12 hex shape (not by `opf:scheme`, +/// which the OPF parser does not surface here) keeps ISBNs, the small integer +/// `calibre` id, and vendor ids from being mistaken for a UUID. +fn parse_book_uuid(raw: &str) -> Option { + let trimmed = raw.trim(); + let candidate = match trimmed.get(..9) { + Some(prefix) if prefix.eq_ignore_ascii_case("urn:uuid:") => trimmed[9..].trim(), + _ => trimmed, + }; + is_uuid(candidate).then(|| candidate.to_ascii_lowercase()) +} + +/// Whether `s` is a canonical 8-4-4-4-12 hex UUID (case-insensitive). +fn is_uuid(s: &str) -> bool { + let mut groups = [8usize, 4, 4, 4, 12].into_iter(); + let mut parts = s.split('-'); + let matched = parts + .by_ref() + .zip(groups.by_ref()) + .all(|(part, len)| part.len() == len && part.bytes().all(|b| b.is_ascii_hexdigit())); + matched && parts.next().is_none() && groups.next().is_none() +} + /// Pull an ISBN out of a `` value if it looks like one. fn extract_isbn(s: &str) -> Option { let digits: String = s @@ -423,4 +459,51 @@ mod tests { let parsed = parse_epub(&bytes).expect("parse"); assert_eq!(parsed.meta.title.as_deref(), Some("A Study in Scarlet")); } + + #[test] + fn calibre_urn_uuid_identifier_becomes_dedup_key() { + // Calibre writes both a scheme identifier and a urn:uuid one; the latter + // is the stable re-upload dedup key. Case-insensitive prefix, lowercased. + let bytes = epub_with_metadata( + r#"D&D 5e Players Handbook + 42 + URN:UUID:5F8E1A2B-0000-4C3D-9E7F-ABCDEF012345 + 9780786965601"#, + ); + let parsed = parse_epub(&bytes).expect("parse"); + assert_eq!( + parsed.meta.book_uuid.as_deref(), + Some("5f8e1a2b-0000-4c3d-9e7f-abcdef012345") + ); + // ISBN parsing still works alongside the uuid capture. + assert_eq!(parsed.meta.isbn.as_deref(), Some("9780786965601")); + } + + #[test] + fn bare_calibre_uuid_identifier_is_captured() { + // Real Calibre EPUBs store the book UUID *without* a urn:uuid: prefix. + let bytes = epub_with_metadata( + r#"Game of Thrones + 7 + 2148815D-3C21-4625-B85E-852CE2E3F46E"#, + ); + let parsed = parse_epub(&bytes).expect("parse"); + assert_eq!( + parsed.meta.book_uuid.as_deref(), + Some("2148815d-3c21-4625-b85e-852ce2e3f46e") + ); + } + + #[test] + fn non_uuid_identifiers_leave_dedup_key_empty() { + // Neither the small `calibre` integer id nor an ISBN is a UUID. + let bytes = epub_with_metadata( + r#"Plain + 7 + 9780786965601"#, + ); + let parsed = parse_epub(&bytes).expect("parse"); + assert_eq!(parsed.meta.book_uuid, None); + assert_eq!(parsed.meta.isbn.as_deref(), Some("9780786965601")); + } } diff --git a/src/upload/handler.rs b/src/upload/handler.rs index 8fefd64..401d32a 100644 --- a/src/upload/handler.rs +++ b/src/upload/handler.rs @@ -98,6 +98,23 @@ pub async fn api_upload( (StatusCode::BAD_REQUEST, format!("Invalid EPUB: {e}")) })?; + // Dedupe by the EPUB's stable identifier. Calibre re-exports the same book + // with different bytes each send (re-zip, refreshed metadata), so the + // content hash above misses re-uploads; the `urn:uuid:` identifier does not. + if let Some(src_uuid) = parsed.meta.book_uuid.as_deref() { + if let Some(existing) = state + .books + .find_by_source_uuid(user.id, src_uuid) + .await + .map_err(internal)? + { + return Err(( + StatusCode::CONFLICT, + format!("This book was already uploaded (id {existing})"), + )); + } + } + // Persist the file under upload_dir/{uuid}/ before responding. The heavy, // multi-second work (online enrichment, cover transcode, kepubify) is // deferred to a background task so the upload request returns immediately — @@ -158,6 +175,7 @@ pub async fn api_upload( locator: uuid.clone(), file_basename: "book".to_string(), content_hash: Some(hash), + source_uuid: meta.book_uuid.clone(), has_cover: false, owner_user_id: Some(user.id), default_sync, -- 2.51.2 From 65d83e730f1405383b17aa67c0220a1630f76aa7 Mon Sep 17 00:00:00 2001 From: servius Date: Tue, 28 Jul 2026 00:15:39 +0530 Subject: [PATCH 2/2] feat: pre-check uploads by UUID so the plugin skips already-synced books /api/upload can only reject a duplicate after receiving the whole file, so re-syncing an already-uploaded library re-POSTed every book just to get a 409. Add POST /api/upload/check (same ApiUser Bearer/session auth as /api/upload): it takes a batch of book UUIDs and returns which the owner already has, matched case-insensitively against books.source_uuid. Backed by BookStore::known_source_uuids, which loads the owner's stored UUID set once. The Calibre plugin now calls it at the start of a run, keyed by each book's Calibre uuid (db.field_for("uuid", ...), the same value embedded in the OPF), and skips the books already present without reading or transferring their files. The check is best-effort: an older server without the endpoint (or any network error) yields an empty set, so every book is still uploaded and the per-book 409 dedup remains the backstop. --- calibre-plugin/main.py | 36 ++++++++++++++++++++++++++++++++++++ src/db/book_store.rs | 22 ++++++++++++++++++++++ src/upload/handler.rs | 40 +++++++++++++++++++++++++++++++++++++++- src/web.rs | 5 ++++- 4 files changed, 101 insertions(+), 2 deletions(-) diff --git a/calibre-plugin/main.py b/calibre-plugin/main.py index e82ff55..2736596 100644 --- a/calibre-plugin/main.py +++ b/calibre-plugin/main.py @@ -6,6 +6,7 @@ use the Python standard library only (no extra plugin dependencies). """ import io +import json import urllib.error import urllib.request import uuid @@ -51,6 +52,14 @@ def run_upload(gui): no_format = [] failures = [] # list of (title, reason) + # Ask the server which of these books it already has (matched by Calibre's + # per-book uuid, which is embedded in every EPUB's OPF). Skipping them here + # avoids uploading the whole file only to be 409'd: /api/upload can only + # reject a duplicate after receiving it. Best-effort — an older server that + # lacks the endpoint yields an empty set, so every book is still uploaded. + book_uuids = {book_id: (db.field_for("uuid", book_id) or "") for book_id in ids} + known = _check_existing(server, token, [u for u in book_uuids.values() if u]) + for i, book_id in enumerate(ids): if progress.wasCanceled(): break @@ -59,6 +68,11 @@ def run_upload(gui): progress.setLabelText("Uploading: %s" % title) QApplication.processEvents() + if book_uuids[book_id].strip().lower() in known: + # Already on the server — do not read or transfer the file. + skipped += 1 + continue + prepared = _prepare_book(db, book_id) if prepared is None: no_format.append(title) @@ -125,6 +139,28 @@ def _embed_metadata(db, book_id, raw, fmt): return raw +def _check_existing(server, token, uuids, timeout=30): + """Return the lowercased set of book UUIDs the server already has. + + Best-effort: any failure (older server without the endpoint, network error) + returns an empty set, so the caller falls back to uploading everything and + relying on the server's per-book 409 dedup.""" + if not uuids: + return set() + body = json.dumps({"uuids": uuids}).encode("utf-8") + req = urllib.request.Request( + server + "/api/upload/check", data=body, method="POST" + ) + req.add_header("Content-Type", "application/json") + req.add_header("Authorization", "Bearer " + token) + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + payload = json.loads(resp.read().decode("utf-8")) + return {str(u).strip().lower() for u in payload.get("known", [])} + except Exception: + return set() + + def _post_book(server, token, filename, data, timeout=180): """POST one book to ``/api/upload`` as multipart/form-data. diff --git a/src/db/book_store.rs b/src/db/book_store.rs index ea4cd32..662ef3e 100644 --- a/src/db/book_store.rs +++ b/src/db/book_store.rs @@ -1,3 +1,4 @@ +use std::collections::HashSet; use std::path::PathBuf; use chrono::{DateTime, NaiveDateTime, Utc}; @@ -630,6 +631,27 @@ impl BookStore { Ok(row.map(|r| r.get::("id"))) } + /// Every source UUID this owner has already uploaded. Lets the Calibre + /// plugin pre-check a batch of books and skip re-sending the ones already + /// present, rather than POSTing each full file only to receive a 409. The + /// stored values are already lowercased by the EPUB parser, so callers + /// compare against a lowercased UUID. + pub async fn known_source_uuids(&self, owner_user_id: i64) -> Result> { + let rows = sqlx::query( + "SELECT source_uuid FROM books \ + WHERE owner_user_id = ? AND source_uuid IS NOT NULL", + ) + .bind(owner_user_id) + .fetch_all(&self.pool) + .await + .change_context(Error::Database) + .attach("Failed to list source uuids")?; + Ok(rows + .iter() + .map(|r| r.get::("source_uuid")) + .collect()) + } + // ---- Internal ---- /// Build a `SELECT ... FROM books {tail}` over every column. diff --git a/src/upload/handler.rs b/src/upload/handler.rs index 401d32a..db36b40 100644 --- a/src/upload/handler.rs +++ b/src/upload/handler.rs @@ -4,7 +4,7 @@ use axum::extract::{Multipart, Path, State}; use axum::http::StatusCode; use axum::Json; use chrono::Utc; -use serde::Serialize; +use serde::{Deserialize, Serialize}; use sha2::{Digest, Sha256}; use tracing::error; use uuid::Uuid; @@ -27,6 +27,44 @@ pub struct UploadedJson { pub title: String, } +/// Body of `POST /api/upload/check` — the book UUIDs the client is about to +/// upload (Calibre's per-book uuid, the same value embedded in each OPF). +#[derive(Deserialize)] +pub struct UploadCheckReq { + pub uuids: Vec, +} + +/// JSON returned by `POST /api/upload/check` — the subset of the requested +/// UUIDs already present for this owner (i.e. the ones that would 409). +#[derive(Serialize)] +pub struct UploadCheckResp { + pub known: Vec, +} + +/// POST /api/upload/check — Report which of the given book UUIDs this owner has +/// already uploaded, so the Calibre plugin can skip re-sending them. `/api/upload` +/// must receive the whole file before it can 409 a duplicate, so a re-sync of an +/// already-uploaded library wastes bandwidth on every book; this cheap pre-check +/// avoids the transfer. UUIDs are matched case-insensitively, matching the +/// normalization the EPUB parser applies to `source_uuid`. +pub async fn api_upload_check( + user: ApiUser, + State(state): State, + Json(req): Json, +) -> Result, (StatusCode, String)> { + let owned = state + .books + .known_source_uuids(user.id) + .await + .map_err(internal)?; + let known = req + .uuids + .into_iter() + .filter(|u| owned.contains(u.trim().to_ascii_lowercase().as_str())) + .collect(); + Ok(Json(UploadCheckResp { known })) +} + /// POST /api/upload — Accept an EPUB upload, extract metadata + cover, and add /// it to the catalog. `multipart/form-data` with a single `file` field. pub async fn api_upload( diff --git a/src/web.rs b/src/web.rs index a0dd4cb..05291f5 100644 --- a/src/web.rs +++ b/src/web.rs @@ -22,7 +22,7 @@ struct WebAssets; use crate::auth::{AuthSession, Credentials}; use crate::kobo::auth::AppState; -use crate::upload::handler::{api_delete_upload, api_upload}; +use crate::upload::handler::{api_delete_upload, api_upload, api_upload_check}; /// Session keys holding the in-flight OIDC login secrets between redirect and /// callback. @@ -157,6 +157,9 @@ pub fn router(max_upload_size: usize) -> Router { "/api/upload", post(api_upload).layer(DefaultBodyLimit::max(max_upload_size)), ) + // Cheap pre-check so the Calibre plugin can skip already-uploaded books + // instead of POSTing each full file only to be 409'd. + .route("/api/upload/check", post(api_upload_check)) .route("/api/upload/{id}", delete(api_delete_upload)); let public = Router::new()