diff --git a/crates/codex-api/src/docs.rs b/crates/codex-api/src/docs.rs index a868a5ac6..164294f4c 100644 --- a/crates/codex-api/src/docs.rs +++ b/crates/codex-api/src/docs.rs @@ -372,6 +372,11 @@ The following paths are exempt from rate limiting: v1::handlers::get_orphaned_reading_history, v1::handlers::purge_orphaned_reading_history, + // Reading progress export/import: carrying reading state across a + // library reorganisation or an instance move + v1::handlers::export_reading_progress, + v1::handlers::import_reading_progress, + // Reading progress endpoints v1::handlers::update_reading_progress, v1::handlers::get_reading_progress, @@ -1067,6 +1072,27 @@ The following paths are exempt from rate limiting: v1::dto::ReadCompletionDto, v1::dto::ReadHistoryResponse, + // Reading progress export/import DTOs + v1::dto::ExportReadingProgressQuery, + v1::dto::ReadingProgressExportDocument, + v1::dto::ExportSeriesDto, + v1::dto::ExportBookDto, + v1::dto::ExportExternalIdDto, + v1::dto::ExportProgressDto, + v1::dto::ExportCompletionDto, + v1::dto::ExportSessionDto, + v1::dto::ImportReadingProgressRequest, + v1::dto::ImportReadingProgressResponse, + v1::dto::HashMode, + v1::dto::ConflictPolicy, + v1::dto::SeriesDisposition, + v1::dto::BookDisposition, + v1::dto::FieldOutcome, + v1::dto::WriteCounts, + v1::dto::ImportSeriesReport, + v1::dto::ImportBookReport, + v1::dto::ImportSummary, + // Want to read (per-user queue) DTOs v1::dto::WantToReadEntryDto, v1::dto::WantToReadListResponse, @@ -1330,6 +1356,7 @@ The following paths are exempt from rate limiting: (name = "User Reader Settings", description = "Per-user, per-series reader overrides"), (name = "Filter Presets", description = "Saved filter combinations for list pages and the advanced search page"), (name = "Reading Progress", description = "Reading progress tracking"), + (name = "Reading Progress Transfer", description = "Exporting and importing reading state across a library reorganisation or an instance move"), // Background Jobs (name = "Task Queue", description = "Background job queue management"), diff --git a/crates/codex-api/src/routes/v1/dto/mod.rs b/crates/codex-api/src/routes/v1/dto/mod.rs index 344265e4f..4f71eb4f4 100644 --- a/crates/codex-api/src/routes/v1/dto/mod.rs +++ b/crates/codex-api/src/routes/v1/dto/mod.rs @@ -25,6 +25,7 @@ pub mod pdf_cache; pub mod plugin_storage; pub mod plugins; pub mod read_progress; +pub mod reading_progress_transfer; pub mod reading_sessions; pub mod reading_stats; pub mod readlist; @@ -68,6 +69,7 @@ pub use pdf_cache::*; pub use plugin_storage::*; pub use plugins::*; pub use read_progress::*; +pub use reading_progress_transfer::*; pub use reading_sessions::*; pub use reading_stats::*; pub use readlist::*; diff --git a/crates/codex-api/src/routes/v1/dto/reading_progress_transfer.rs b/crates/codex-api/src/routes/v1/dto/reading_progress_transfer.rs new file mode 100644 index 000000000..b7506386c --- /dev/null +++ b/crates/codex-api/src/routes/v1/dto/reading_progress_transfer.rs @@ -0,0 +1,34 @@ +//! DTOs for exporting and importing one user's reading state. +//! +//! The document, request, and report shapes are defined in +//! `codex_services::reading_transfer::model` (re-exported below) rather than +//! here, because the service that produces and consumes them owns the shape; +//! this module only adds the query parameters for the export endpoint, which +//! are handler-only concerns. + +pub use codex_services::reading_transfer::model::*; + +use serde::Deserialize; +use utoipa::ToSchema; + +fn default_true() -> bool { + true +} + +/// Query parameters for `GET /api/v1/reading-progress/export`. +#[derive(Debug, Clone, Deserialize, ToSchema)] +pub struct ExportReadingProgressQuery { + /// Sessions are opt-out: they are the only source of every reading + /// statistic, so leaving them out is easy to do by accident and hard to + /// notice until the numbers are gone. + #[serde(default = "default_true")] + pub include_sessions: bool, +} + +impl Default for ExportReadingProgressQuery { + fn default() -> Self { + Self { + include_sessions: true, + } + } +} diff --git a/crates/codex-api/src/routes/v1/handlers/mod.rs b/crates/codex-api/src/routes/v1/handlers/mod.rs index 49b90fd54..6bb6491ca 100644 --- a/crates/codex-api/src/routes/v1/handlers/mod.rs +++ b/crates/codex-api/src/routes/v1/handlers/mod.rs @@ -88,6 +88,7 @@ pub mod plugin_storage; pub mod plugin_web_links; pub mod plugins; pub mod read_progress; +pub mod reading_progress_transfer; pub mod reading_sessions; pub mod reading_stats; pub mod readlists; @@ -121,6 +122,7 @@ pub use libraries::*; pub use metrics::*; pub use pages::*; pub use read_progress::*; +pub use reading_progress_transfer::*; pub use reading_sessions::*; pub use reading_stats::*; pub use readlists::*; diff --git a/crates/codex-api/src/routes/v1/handlers/reading_progress_transfer.rs b/crates/codex-api/src/routes/v1/handlers/reading_progress_transfer.rs new file mode 100644 index 000000000..79e6c711e --- /dev/null +++ b/crates/codex-api/src/routes/v1/handlers/reading_progress_transfer.rs @@ -0,0 +1,148 @@ +//! Export and import of one user's reading state, for carrying it across a +//! library reorganisation or a move to a different Codex instance. +//! +//! Export streams a JSON document naming the user's `read_progress`, +//! `read_completions`, `reading_sessions`, and `user_series_ratings` rows +//! against stable keys (external ids, a series-relative path, file name, +//! hashes) rather than database ids, which a rescan under a new root always +//! replaces. Import resolves those keys back to real ids in the current +//! library and writes the state under an explicit conflict policy, one +//! transaction per series, with a dry run that reports the identical shape +//! without writing anything. + +use axum::{ + Json, + extract::{Query, State}, + http::{HeaderValue, header}, + response::{IntoResponse, Response}, +}; +use chrono::Utc; +use std::sync::Arc; + +use super::super::dto::{ + ExportReadingProgressQuery, ImportReadingProgressRequest, ImportReadingProgressResponse, +}; +use crate::{AppState, error::ApiError, extractors::AuthContext, permissions::Permission}; +use codex_services::reading_transfer::import::{ImportError, ImportOptions}; + +/// Export the authenticated user's reading progress +/// +/// Produces the whole `read_progress` / `read_completions` / `reading_sessions` +/// / `user_series_ratings` state for the caller, keyed by external ids and +/// series-relative paths instead of database ids so it can be matched back +/// against a differently organised library (or a different Codex instance +/// entirely) by `POST /api/v1/reading-progress/import`. +/// +/// Take this export **before** deleting an old library during a split: the +/// underlying rows survive a hard delete as orphans, but nothing except this +/// file can say which book an orphan used to belong to. +#[utoipa::path( + get, + path = "/api/v1/reading-progress/export", + params( + ("include_sessions" = Option, Query, description = "Include the reading-session log (default: true). Sessions are the only source of every reading statistic, so this is opt-out rather than opt-in.") + ), + responses( + (status = 200, description = "The export document", body = codex_services::reading_transfer::model::ReadingProgressExportDocument), + (status = 401, description = "Unauthorized"), + (status = 403, description = "Forbidden"), + ), + security( + ("jwt_bearer" = []), + ("api_key" = []) + ), + tag = "Reading Progress Transfer" +)] +pub async fn export_reading_progress( + State(state): State>, + auth: AuthContext, + Query(query): Query, +) -> Result { + auth.require_permission(&Permission::ProgressRead)?; + + let document = + codex_services::export_reading_progress(&state.db, auth.user_id, query.include_sessions) + .await + .map_err(|e| ApiError::Internal(format!("Failed to export reading progress: {e}")))?; + + let filename = format!( + "codex-reading-progress-{}.json", + Utc::now().format("%Y-%m-%d") + ); + let disposition = crate::ranged_file::content_disposition_attachment(&filename); + + let mut response = Json(document).into_response(); + if let Ok(value) = HeaderValue::from_str(&disposition) { + response + .headers_mut() + .insert(header::CONTENT_DISPOSITION, value); + } + Ok(response) +} + +/// Import reading progress from an export document +/// +/// Resolves the file's series and books against the current library through +/// the same access-group / sharing-tag visibility that an ordinary read +/// respects: a book the importing user cannot see resolves as `unmatched`, +/// never as a permission error that would confirm it exists, and nothing is +/// ever written against it. +/// +/// Matching never guesses: any step (external id, path, file name, or, under +/// `hash_mode = "match"`, hash) that finds more than one candidate reports +/// `ambiguous` and writes nothing for that series or book. `dry_run: true` +/// returns the identical response shape without writing anything, which is +/// what makes it safe to preview before committing. +/// +/// Each series is applied in its own transaction; the response's per-series +/// `committed` field says which ones actually landed. `read_completions` and +/// `reading_sessions` reuse their exported ids on insert, so importing the +/// same file twice leaves row counts unchanged rather than duplicating +/// history. +#[utoipa::path( + post, + path = "/api/v1/reading-progress/import", + request_body = ImportReadingProgressRequest, + responses( + (status = 200, description = "Import processed (or, for a dry run, previewed)", body = ImportReadingProgressResponse), + (status = 400, description = "Unknown export format, a version newer than this server supports, or a value the normal write paths reject"), + (status = 413, description = "The file is larger than the import limit"), + (status = 401, description = "Unauthorized"), + (status = 403, description = "Forbidden"), + ), + security( + ("jwt_bearer" = []), + ("api_key" = []) + ), + tag = "Reading Progress Transfer" +)] +pub async fn import_reading_progress( + State(state): State>, + auth: AuthContext, + Json(request): Json, +) -> Result, ApiError> { + auth.require_permission(&Permission::ProgressWrite)?; + + let options = ImportOptions { + dry_run: request.dry_run, + hash_mode: request.hash_mode, + source_preference: request.source_preference, + conflict_policy: request.conflict_policy, + reattach_sessions: request.reattach_sessions, + accept_stem_matches: request.accept_stem_matches, + }; + + let response = + codex_services::import_reading_progress(&state.db, auth.user_id, &request.file, &options) + .await + .map_err(|err| match err { + ImportError::UnknownFormat(_) + | ImportError::UnsupportedVersion(_) + | ImportError::InvalidValue(_) => ApiError::BadRequest(err.to_string()), + ImportError::Database(source) => { + ApiError::Internal(format!("Failed to import reading progress: {source}")) + } + })?; + + Ok(Json(response)) +} diff --git a/crates/codex-api/src/routes/v1/handlers/reading_stats.rs b/crates/codex-api/src/routes/v1/handlers/reading_stats.rs index f79ff0c29..e30253e0e 100644 --- a/crates/codex-api/src/routes/v1/handlers/reading_stats.rs +++ b/crates/codex-api/src/routes/v1/handlers/reading_stats.rs @@ -232,8 +232,10 @@ pub async fn get_orphaned_reading_history( /// the series and format breakdowns show it as one "removed from library" row. /// This discards those rows for the caller, and only for the caller. /// -/// Irreversible. Only history already detached from any book is touched; -/// attributed reading is never affected. +/// Irreversible, and it forecloses the other way out: importing a reading +/// progress export taken before the delete puts those sessions back on their +/// books once the files are scanned again. Only history already detached from +/// any book is touched; attributed reading is never affected. #[utoipa::path( delete, path = "/api/v1/reading-stats/orphaned", diff --git a/crates/codex-api/src/routes/v1/routes/books.rs b/crates/codex-api/src/routes/v1/routes/books.rs index b83210204..cf39a8d2a 100644 --- a/crates/codex-api/src/routes/v1/routes/books.rs +++ b/crates/codex-api/src/routes/v1/routes/books.rs @@ -7,10 +7,16 @@ use super::super::handlers; use crate::extractors::AppState; use axum::{ Router, + extract::DefaultBodyLimit, routing::{delete, get, patch, post, put}, }; use std::sync::Arc; +/// Largest reading-progress import accepted. A measured 5,000-book history +/// with one session per book is about 4 MB, so this leaves room for heavy +/// re-readers and several devices without accepting unbounded bodies. +const MAX_READING_PROGRESS_IMPORT_BYTES: usize = 64 * 1024 * 1024; + /// Create book routes /// /// All routes are protected (authentication required). @@ -193,6 +199,20 @@ pub fn routes(_state: Arc) -> Router> { get(handlers::get_orphaned_reading_history) .delete(handlers::purge_orphaned_reading_history), ) + // Carrying reading state across a library reorganisation or a move + // to a different Codex instance. + .route( + "/reading-progress/export", + get(handlers::export_reading_progress), + ) + // A whole reading history in one body: axum's 2 MB default stops at + // roughly 2,500 books with their sessions, well short of a large + // library, so this route alone gets a higher ceiling. + .route( + "/reading-progress/import", + post(handlers::import_reading_progress) + .layer(DefaultBodyLimit::max(MAX_READING_PROGRESS_IMPORT_BYTES)), + ) // Mark as read/unread routes .route("/books/{book_id}/read", post(handlers::mark_book_as_read)) .route( diff --git a/crates/codex-db/src/repositories/reading_stats.rs b/crates/codex-db/src/repositories/reading_stats.rs index ef392d478..1117feccb 100644 --- a/crates/codex-db/src/repositories/reading_stats.rs +++ b/crates/codex-db/src/repositories/reading_stats.rs @@ -738,7 +738,9 @@ impl ReadingStatsRepository { /// and nothing else. /// /// Those rows keep counting towards every total until the reader decides - /// otherwise; this is that decision, so it is never run on a schedule. + /// otherwise; this is that decision, so it is never run on a schedule. A + /// reading-progress import can reattach these rows to their books, and a + /// purge makes that impossible. /// /// Sessions and completions go together in one transaction so the two /// logs cannot disagree about whether the history exists. diff --git a/crates/codex-services/src/lib.rs b/crates/codex-services/src/lib.rs index a32478c50..93f4c6cdf 100644 --- a/crates/codex-services/src/lib.rs +++ b/crates/codex-services/src/lib.rs @@ -31,6 +31,7 @@ pub mod plugin_file_storage; pub mod plugin_metrics; pub mod rate_limiter; pub mod read_progress; +pub mod reading_transfer; pub mod refresh_token; pub mod release; pub mod scheduler_handle; @@ -58,6 +59,8 @@ pub use pdf_handle_cache::{HandleCacheEntrySnapshot, HandleCacheSnapshot, PdfHan pub use pdf_handle_cache_subscriber::PdfHandleCacheSubscriber; pub use rate_limiter::RateLimiterService; pub use read_progress::ReadProgressService; +pub use reading_transfer::export::export_reading_progress; +pub use reading_transfer::import::{ImportError, ImportOptions, import_reading_progress}; #[allow(unused_imports)] pub use refresh_token::{IssuedRefreshToken, RefreshTokenError, RefreshTokenService}; pub use settings::SettingsService; diff --git a/crates/codex-services/src/reading_transfer/export.rs b/crates/codex-services/src/reading_transfer/export.rs new file mode 100644 index 000000000..601d70c7b --- /dev/null +++ b/crates/codex-services/src/reading_transfer/export.rs @@ -0,0 +1,761 @@ +//! Assembling one user's reading state into the portable export document. +//! +//! Reads `read_progress`, `read_completions`, `reading_sessions`, and +//! `user_series_ratings` for exactly the requesting user, groups them by +//! book and then by series, and serialises everything against stable keys +//! (external ids, a series-relative path, file name, hashes) instead of the +//! database ids that a rescan under a new root would replace. +//! +//! No visibility filtering: this is the user's own data, not a view of the +//! library, so a book later hidden behind a sharing tag is still exported. +//! Import is where the visibility filter matters, because that is where a +//! crafted file could otherwise be used to write state onto a book the +//! importing user cannot see. + +use anyhow::Result; +use chrono::Utc; +use sea_orm::{ColumnTrait, DatabaseConnection, EntityTrait, QueryFilter, QueryOrder}; +use std::collections::{HashMap, HashSet}; +use uuid::Uuid; + +use codex_db::entities::{books, read_completions, read_progress, reading_sessions}; +use codex_db::repositories::{ + LibraryRepository, ReadProgressRepository, SeriesExternalIdRepository, SeriesRepository, + UserSeriesRatingRepository, +}; + +use super::model::{ + ExportBookDto, ExportCompletionDto, ExportExternalIdDto, ExportProgressDto, ExportSeriesDto, + ExportSessionDto, READING_PROGRESS_FORMAT, READING_PROGRESS_VERSION, + ReadingProgressExportDocument, +}; +use super::series_relative_book_path; + +/// Every completion for `user_id` whose book has not been hard-deleted. +/// +/// Already-orphaned completions (`book_id IS NULL`) are skipped for the same +/// reason as sessions: nothing in the file could say which book to put them +/// back on. +async fn completions_for_user( + db: &DatabaseConnection, + user_id: Uuid, +) -> Result> { + let rows = read_completions::Entity::find() + .filter(read_completions::Column::UserId.eq(user_id)) + .filter(read_completions::Column::BookId.is_not_null()) + .order_by_asc(read_completions::Column::StartedAt) + .all(db) + .await?; + Ok(rows) +} + +/// The books a user's history points at, **including soft-deleted ones**. +/// +/// `BookRepository::get_by_ids` hides books the scanner has marked deleted, +/// which is right for browsing and wrong here. Splitting a library usually +/// moves the files first, the next scan of the old library marks every moved +/// book deleted without purging it, and that is exactly when the reader takes +/// the export. Hiding them would export nothing for the books the file exists +/// to carry across. +/// +/// Batched so a long history stays under the engines' bind-parameter limits. +async fn books_including_soft_deleted( + db: &DatabaseConnection, + ids: &[Uuid], +) -> Result> { + const BATCH: usize = 1_000; + let mut found = Vec::with_capacity(ids.len()); + for chunk in ids.chunks(BATCH) { + found.extend( + books::Entity::find() + .filter(books::Column::Id.is_in(chunk.to_vec())) + .all(db) + .await?, + ); + } + Ok(found) +} + +/// Every session for `user_id` whose book has not been hard-deleted. +async fn sessions_for_user( + db: &DatabaseConnection, + user_id: Uuid, +) -> Result> { + let rows = reading_sessions::Entity::find() + .filter(reading_sessions::Column::UserId.eq(user_id)) + .filter(reading_sessions::Column::BookId.is_not_null()) + .order_by_asc(reading_sessions::Column::ClientEndedAt) + .all(db) + .await?; + Ok(rows) +} + +/// Assemble the export document for one user. +/// +/// `include_sessions = false` omits the `sessions` key entirely on every book +/// rather than emitting empty arrays, so a client can tell "not exported" +/// apart from "exported, and there were none". +pub async fn export_reading_progress( + db: &DatabaseConnection, + user_id: Uuid, + include_sessions: bool, +) -> Result { + let progress_rows = ReadProgressRepository::get_by_user(db, user_id).await?; + let completion_rows = completions_for_user(db, user_id).await?; + let session_rows = if include_sessions { + sessions_for_user(db, user_id).await? + } else { + Vec::new() + }; + let ratings = UserSeriesRatingRepository::get_all_for_user(db, user_id).await?; + + let mut book_id_set: HashSet = HashSet::new(); + for p in &progress_rows { + book_id_set.insert(p.book_id); + } + for c in &completion_rows { + if let Some(id) = c.book_id { + book_id_set.insert(id); + } + } + for s in &session_rows { + if let Some(id) = s.book_id { + book_id_set.insert(id); + } + } + let book_ids: Vec = book_id_set.into_iter().collect(); + let books = books_including_soft_deleted(db, &book_ids).await?; + + let mut series_id_set: HashSet = books.iter().map(|b| b.series_id).collect(); + for r in &ratings { + series_id_set.insert(r.series_id); + } + let series_ids: Vec = series_id_set.into_iter().collect(); + + let mut series_rows = SeriesRepository::get_by_ids(db, &series_ids).await?; + series_rows.sort_by(|a, b| a.path.cmp(&b.path).then_with(|| a.id.cmp(&b.id))); + + let library_ids: Vec = series_rows + .iter() + .map(|s| s.library_id) + .collect::>() + .into_iter() + .collect(); + let library_path_by_id: HashMap = LibraryRepository::get_by_ids(db, &library_ids) + .await? + .into_iter() + .map(|(id, lib)| (id, lib.path)) + .collect(); + + let external_ids_by_series = + SeriesExternalIdRepository::get_for_series_ids(db, &series_ids).await?; + let ratings_by_series: HashMap = + ratings.into_iter().map(|r| (r.series_id, r)).collect(); + + let mut books_by_series: HashMap> = HashMap::new(); + for b in &books { + books_by_series.entry(b.series_id).or_default().push(b); + } + + let progress_by_book: HashMap = + progress_rows.iter().map(|p| (p.book_id, p)).collect(); + + let mut completions_by_book: HashMap> = HashMap::new(); + for c in &completion_rows { + if let Some(book_id) = c.book_id { + completions_by_book.entry(book_id).or_default().push(c); + } + } + + let mut sessions_by_book: HashMap> = HashMap::new(); + for s in &session_rows { + if let Some(book_id) = s.book_id { + sessions_by_book.entry(book_id).or_default().push(s); + } + } + + let mut series_docs = Vec::with_capacity(series_rows.len()); + + for series in &series_rows { + let library_path = library_path_by_id + .get(&series.library_id) + .cloned() + .unwrap_or_default(); + + let external_ids = external_ids_by_series + .get(&series.id) + .map(|ids| { + ids.iter() + .map(|e| ExportExternalIdDto { + source: e.source.clone(), + id: e.external_id.clone(), + }) + .collect() + }) + .unwrap_or_default(); + + let rating_row = ratings_by_series.get(&series.id); + + let mut book_docs = Vec::new(); + if let Some(series_books) = books_by_series.get(&series.id) { + let mut sorted_books: Vec<&books::Model> = series_books.to_vec(); + sorted_books.sort_by(|a, b| a.path.cmp(&b.path)); + + for book in sorted_books { + let relative_path = + series_relative_book_path(&library_path, &series.path, &book.path); + + let progress = progress_by_book.get(&book.id).map(|p| ExportProgressDto { + current_page: p.current_page, + progress_percentage: p.progress_percentage, + completed: p.completed, + started_at: p.started_at, + updated_at: p.updated_at, + completed_at: p.completed_at, + r2_progression: p.r2_progression.clone(), + }); + + let completions: Vec = completions_by_book + .get(&book.id) + .map(|list| { + list.iter() + .map(|c| ExportCompletionDto { + id: c.id, + started_at: c.started_at, + completed_at: c.completed_at, + }) + .collect() + }) + .unwrap_or_default(); + + let sessions: Option> = if include_sessions { + Some( + sessions_by_book + .get(&book.id) + .map(|list| { + list.iter() + .map(|s| ExportSessionDto { + id: s.id, + device_id: s.device_id.clone(), + device_name: s.device_name.clone(), + pass: s.pass, + kind: s.kind.clone(), + to_page: s.to_page, + to_percentage: s.to_percentage, + active_duration_ms: s.active_duration_ms, + duration_source: s.duration_source.clone(), + pages_read: s.pages_read, + client_started_at: s.client_started_at, + client_ended_at: s.client_ended_at, + server_recorded_at: s.server_recorded_at, + }) + .collect() + }) + .unwrap_or_default(), + ) + } else { + None + }; + + let has_sessions = sessions.as_ref().is_some_and(|s| !s.is_empty()); + if progress.is_none() && completions.is_empty() && !has_sessions { + // Nothing to say about this book for this user. + continue; + } + + book_docs.push(ExportBookDto { + path: relative_path, + file_name: book.file_name.clone(), + file_hash: book.file_hash.clone(), + partial_hash: book.partial_hash.clone(), + progress, + completions, + sessions, + }); + } + } + + if rating_row.is_none() && book_docs.is_empty() { + // Nothing recorded against this series for this user. + continue; + } + + series_docs.push(ExportSeriesDto { + external_ids, + // Always `/`, whatever the server stores, so the file matches on + // an instance running on another platform. + library_relative_path: series.path.replace('\\', "/"), + name: series.name.clone(), + rating: rating_row.map(|r| r.rating), + notes: rating_row.and_then(|r| r.notes.clone()), + rating_updated_at: rating_row.map(|r| r.updated_at), + books: book_docs, + }); + } + + Ok(ReadingProgressExportDocument { + format: READING_PROGRESS_FORMAT.to_string(), + version: READING_PROGRESS_VERSION, + exported_at: Utc::now(), + includes_sessions: include_sessions, + series: series_docs, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use chrono::Duration; + use codex_db::ScanningStrategy; + use codex_db::entities::reading_sessions::SessionKind; + use codex_db::repositories::{ + BookRepository, LibraryRepository, NewSession, ReadCompletionRepository, SeriesRepository, + UserRepository, + }; + use codex_db::test_helpers::create_test_db; + + async fn make_user(db: &DatabaseConnection) -> Uuid { + use chrono::Utc; + use codex_db::entities::users; + let now = Utc::now(); + let model = users::Model { + id: Uuid::new_v4(), + username: format!("u-{}", Uuid::new_v4()), + email: format!("{}@test.com", Uuid::new_v4()), + password_hash: "hash".to_string(), + role: "reader".to_string(), + is_active: true, + email_verified: true, + permissions: serde_json::json!([]), + created_at: now, + updated_at: now, + last_login_at: None, + }; + UserRepository::create(db, &model).await.unwrap().id + } + + #[tokio::test] + async fn exports_a_nested_volume_folder_and_empty_hash() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + + let library = + LibraryRepository::create(conn, "Lib", "/library/root", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(conn, library.id, "Naruto", None) + .await + .unwrap(); + SeriesRepository::update_path(conn, series.id, "shonen/Naruto".to_string()) + .await + .unwrap(); + + let book = books::Model { + id: Uuid::new_v4(), + series_id: series.id, + library_id: library.id, + path: "/library/root/shonen/Naruto/Vol 01/v01.cbz".to_string(), + file_name: "v01.cbz".to_string(), + file_size: 1024, + file_hash: String::new(), + partial_hash: String::new(), + format: "cbz".to_string(), + page_count: 20, + deleted: false, + analyzed: false, + analysis_error: None, + analysis_errors: None, + modified_at: Utc::now(), + created_at: Utc::now(), + updated_at: Utc::now(), + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }; + let book = BookRepository::create(conn, &book, None).await.unwrap(); + + ReadProgressRepository::upsert(conn, user, book.id, 5, false) + .await + .unwrap(); + + let doc = export_reading_progress(conn, user, true).await.unwrap(); + + assert_eq!(doc.format, READING_PROGRESS_FORMAT); + assert_eq!(doc.series.len(), 1); + let series_doc = &doc.series[0]; + assert_eq!(series_doc.library_relative_path, "shonen/Naruto"); + assert_eq!(series_doc.books.len(), 1); + let book_doc = &series_doc.books[0]; + assert_eq!(book_doc.path, "Vol 01/v01.cbz"); + assert_eq!(book_doc.file_hash, ""); + assert!(book_doc.progress.is_some()); + } + + /// Splitting a library usually moves the files first, and the next scan of + /// the old library marks every moved book deleted without purging it. That + /// is exactly when a reader takes the export, so soft-deleted books must be + /// in it: leaving them out would export nothing for the very books the + /// file exists to carry across. + #[tokio::test] + async fn exports_books_the_scanner_has_soft_deleted() { + use sea_orm::{ActiveModelTrait, Set}; + + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + + let library = LibraryRepository::create(conn, "Lib", "/manga", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(conn, library.id, "Naruto", None) + .await + .unwrap(); + SeriesRepository::update_path(conn, series.id, "shonen/Naruto".to_string()) + .await + .unwrap(); + let book = books::Model { + id: Uuid::new_v4(), + series_id: series.id, + library_id: library.id, + path: "/manga/shonen/Naruto/v01.cbz".to_string(), + file_name: "v01.cbz".to_string(), + file_size: 10, + file_hash: "h".to_string(), + partial_hash: "p".to_string(), + format: "cbz".to_string(), + page_count: 20, + deleted: false, + analyzed: true, + analysis_error: None, + analysis_errors: None, + modified_at: Utc::now(), + created_at: Utc::now(), + updated_at: Utc::now(), + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }; + let book = BookRepository::create(conn, &book, None).await.unwrap(); + ReadProgressRepository::upsert(conn, user, book.id, 12, false) + .await + .unwrap(); + + let mut missing: books::ActiveModel = book.clone().into(); + missing.deleted = Set(true); + missing.update(conn).await.unwrap(); + + let doc = export_reading_progress(conn, user, true).await.unwrap(); + + assert_eq!( + doc.series.len(), + 1, + "the soft-deleted book's series is exported" + ); + let book_doc = &doc.series[0].books[0]; + assert_eq!(book_doc.path, "v01.cbz"); + assert_eq!(book_doc.progress.as_ref().unwrap().current_page, 12); + } + + #[tokio::test] + async fn omits_sessions_key_when_not_requested() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(conn, library.id, "Series", None) + .await + .unwrap(); + let book = books::Model { + id: Uuid::new_v4(), + series_id: series.id, + library_id: library.id, + path: "/lib/book.cbz".to_string(), + file_name: "book.cbz".to_string(), + file_size: 10, + file_hash: "h".to_string(), + partial_hash: "p".to_string(), + format: "cbz".to_string(), + page_count: 1, + deleted: false, + analyzed: true, + analysis_error: None, + analysis_errors: None, + modified_at: Utc::now(), + created_at: Utc::now(), + updated_at: Utc::now(), + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }; + let book = BookRepository::create(conn, &book, None).await.unwrap(); + + let now = Utc::now(); + let session = NewSession::from_client( + Uuid::new_v4(), + user, + book.id, + "device", + None, + SessionKind::Progress, + Some(1000), + Some(1), + now - Duration::minutes(5), + now, + ) + .with_page(1); + ReadProgressRepository::record_session(conn, session) + .await + .unwrap(); + ReadCompletionRepository::record(conn, user, book.id, now - Duration::minutes(5), now) + .await + .unwrap(); + + let doc_without = export_reading_progress(conn, user, false).await.unwrap(); + assert!(!doc_without.includes_sessions); + let book_doc = &doc_without.series[0].books[0]; + assert!(book_doc.sessions.is_none()); + // Completions and progress are unaffected by include_sessions. + assert_eq!(book_doc.completions.len(), 1); + assert!(book_doc.progress.is_some()); + + let doc_with = export_reading_progress(conn, user, true).await.unwrap(); + assert!(doc_with.includes_sessions); + let book_doc = &doc_with.series[0].books[0]; + assert!(book_doc.sessions.is_some()); + assert_eq!(book_doc.sessions.as_ref().unwrap().len(), 1); + // Sessions never carry r2_progression. + assert!(book_doc.progress.as_ref().unwrap().current_page >= 0); + } + + /// Confirms the plan's size budget: a 5,000-book read history should stay + /// under ~5 MB uncompressed. Each book here gets full progress, one + /// completion, and one session (one read-through) plus a per-series + /// rating, which measured 3.85 MB. Doubling the session count per book + /// (a re-read, or two devices each syncing their own session) measured + /// 5.56 MB, over the estimate: sessions are the dominant cost, so a + /// heavier history than "read once" can exceed it. + /// + /// Seeding 5,000 books plus their progress, completions, and sessions + /// through the ORM is slow enough to skip in the default run; measure it + /// on demand with: + /// `cargo test -p codex-services --lib reading_transfer::export::tests::export_of_5000_books_stays_under_5mb -- --ignored --nocapture` + #[tokio::test] + #[ignore] + async fn export_of_5000_books_stays_under_5mb() { + use codex_db::entities::read_progress; + use sea_orm::{EntityTrait, Set}; + + const SERIES_COUNT: usize = 50; + const BOOKS_PER_SERIES: usize = 100; + + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + + let library = + LibraryRepository::create(conn, "Big Library", "/big", ScanningStrategy::Default) + .await + .unwrap(); + + let mut all_books = Vec::with_capacity(SERIES_COUNT * BOOKS_PER_SERIES); + let mut progress_rows = Vec::with_capacity(SERIES_COUNT * BOOKS_PER_SERIES); + let mut completion_rows = Vec::new(); + let mut session_rows = Vec::new(); + let now = Utc::now(); + + for s in 0..SERIES_COUNT { + let series = + SeriesRepository::create(conn, library.id, &format!("Series {s:04}"), None) + .await + .unwrap(); + SeriesRepository::update_path(conn, series.id, format!("series-{s:04}")) + .await + .unwrap(); + UserSeriesRatingRepository::create( + conn, + user, + series.id, + 80, + Some("solid run".to_string()), + ) + .await + .unwrap(); + + for b in 0..BOOKS_PER_SERIES { + let book_id = Uuid::new_v4(); + all_books.push(books::Model { + id: book_id, + series_id: series.id, + library_id: library.id, + path: format!("/big/series-{s:04}/v{b:03}.cbz"), + file_name: format!("v{b:03}.cbz"), + file_size: 12_345, + file_hash: format!("hash-{s:04}-{b:03}"), + partial_hash: format!("partial-{s:04}-{b:03}"), + format: "cbz".to_string(), + page_count: 24, + deleted: false, + analyzed: true, + analysis_error: None, + analysis_errors: None, + modified_at: now, + created_at: now, + updated_at: now, + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }); + + progress_rows.push(read_progress::ActiveModel { + id: Set(Uuid::new_v4()), + user_id: Set(user), + book_id: Set(book_id), + current_page: Set(24), + progress_percentage: Set(None), + completed: Set(true), + started_at: Set(now - Duration::days(1)), + updated_at: Set(now), + completed_at: Set(Some(now)), + r2_progression: Set(None), + }); + + completion_rows.push(read_completions::ActiveModel { + id: Set(Uuid::new_v4()), + user_id: Set(user), + book_id: Set(Some(book_id)), + started_at: Set(now - Duration::days(1)), + completed_at: Set(now), + }); + + for pass in 0..1 { + session_rows.push(reading_sessions::ActiveModel { + id: Set(Uuid::new_v4()), + user_id: Set(user), + book_id: Set(Some(book_id)), + device_id: Set("device-1".to_string()), + device_name: Set(Some("Test Device".to_string())), + pass: Set(pass), + kind: Set("progress".to_string()), + to_page: Set(Some(24)), + to_percentage: Set(None), + r2_progression: Set(None), + active_duration_ms: Set(Some(600_000)), + duration_source: Set("measured".to_string()), + pages_read: Set(Some(24)), + client_started_at: Set(now - Duration::minutes(10)), + client_ended_at: Set(now), + server_recorded_at: Set(now), + }); + } + } + } + + for chunk in all_books.chunks(500) { + BookRepository::create_batch(conn, chunk).await.unwrap(); + } + for chunk in progress_rows.chunks(500) { + read_progress::Entity::insert_many(chunk.to_vec()) + .exec(conn) + .await + .unwrap(); + } + for chunk in completion_rows.chunks(500) { + read_completions::Entity::insert_many(chunk.to_vec()) + .exec(conn) + .await + .unwrap(); + } + for chunk in session_rows.chunks(500) { + reading_sessions::Entity::insert_many(chunk.to_vec()) + .exec(conn) + .await + .unwrap(); + } + + let doc = export_reading_progress(conn, user, true).await.unwrap(); + assert_eq!(doc.series.len(), SERIES_COUNT); + + let json = serde_json::to_vec(&doc).unwrap(); + let bytes = json.len(); + println!( + "5,000-book export: {} bytes ({:.2} MB)", + bytes, + bytes as f64 / 1_048_576.0 + ); + assert!( + bytes < 5 * 1_048_576, + "export of 5,000 books should stay under ~5 MB uncompressed, was {bytes} bytes" + ); + } + + #[tokio::test] + async fn export_contains_no_other_users_rows() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user_a = make_user(conn).await; + let user_b = make_user(conn).await; + + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(conn, library.id, "Series", None) + .await + .unwrap(); + let book = books::Model { + id: Uuid::new_v4(), + series_id: series.id, + library_id: library.id, + path: "/lib/book.cbz".to_string(), + file_name: "book.cbz".to_string(), + file_size: 10, + file_hash: "h".to_string(), + partial_hash: "p".to_string(), + format: "cbz".to_string(), + page_count: 1, + deleted: false, + analyzed: true, + analysis_error: None, + analysis_errors: None, + modified_at: Utc::now(), + created_at: Utc::now(), + updated_at: Utc::now(), + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }; + let book = BookRepository::create(conn, &book, None).await.unwrap(); + + ReadProgressRepository::upsert(conn, user_a, book.id, 3, false) + .await + .unwrap(); + ReadProgressRepository::upsert(conn, user_b, book.id, 9, false) + .await + .unwrap(); + + let doc = export_reading_progress(conn, user_a, true).await.unwrap(); + assert_eq!(doc.series.len(), 1); + assert_eq!(doc.series[0].books.len(), 1); + assert_eq!( + doc.series[0].books[0] + .progress + .as_ref() + .unwrap() + .current_page, + 3 + ); + } +} diff --git a/crates/codex-services/src/reading_transfer/import.rs b/crates/codex-services/src/reading_transfer/import.rs new file mode 100644 index 000000000..d164ac7a3 --- /dev/null +++ b/crates/codex-services/src/reading_transfer/import.rs @@ -0,0 +1,1534 @@ +//! Applying a matched export document under an explicit conflict policy. +//! +//! One transaction per series: a partial failure part-way through a large +//! import must not leave some of a series' books updated and others not, but +//! it also must not cost every other series in the file its own progress just +//! because one series' data was bad. The response reports which series +//! actually committed. +//! +//! A dry run walks the exact same decision logic as a real import (matching, +//! conflict resolution, orphan reattachment) and simply never opens a +//! transaction to apply it, which is what guarantees the response shape is +//! identical either way and that nothing is written. + +use anyhow::Result; +use chrono::{DateTime, Utc}; +use sea_orm::sea_query::Expr; +use sea_orm::{ + ActiveModelTrait, ColumnTrait, DatabaseConnection, DatabaseTransaction, EntityTrait, + QueryFilter, Set, TransactionTrait, +}; +use std::collections::{HashMap, HashSet}; +use std::fmt; +use uuid::Uuid; + +use codex_db::entities::{ + books, read_completions, read_progress, reading_sessions, user_series_ratings, +}; +use codex_db::repositories::{ + BookRepository, LibraryRepository, SeriesRepository, UserSeriesRatingRepository, +}; + +use crate::content_filter::ContentFilter; + +use super::matching::{self, BookCandidate, BookMatch, SeriesMatch}; +use super::model::{ + BookDisposition, ConflictPolicy, ExportBookDto, ExportCompletionDto, ExportSeriesDto, + ExportSessionDto, FieldOutcome, HashMode, ImportBookReport, ImportReadingProgressResponse, + ImportSeriesReport, ImportSummary, READING_PROGRESS_FORMAT, READING_PROGRESS_VERSION, + ReadingProgressExportDocument, SeriesDisposition, WriteCounts, +}; + +/// Everything the caller controls about how an import behaves. +#[derive(Debug, Clone)] +pub struct ImportOptions { + pub dry_run: bool, + pub hash_mode: HashMode, + pub source_preference: Vec, + pub conflict_policy: ConflictPolicy, + pub reattach_sessions: bool, + pub accept_stem_matches: bool, +} + +/// A rejection worth a 400, versus every other failure which is a 500. +#[derive(Debug)] +pub enum ImportError { + UnknownFormat(String), + UnsupportedVersion(i32), + /// A value the normal write paths would never accept. Named precisely so + /// the user can find it in the file. + InvalidValue(String), + Database(anyhow::Error), +} + +impl fmt::Display for ImportError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + ImportError::UnknownFormat(format) => write!( + f, + "unknown export format '{format}', expected '{READING_PROGRESS_FORMAT}'" + ), + ImportError::UnsupportedVersion(version) => write!( + f, + "export version {version} is newer than this server supports (max {READING_PROGRESS_VERSION})" + ), + ImportError::InvalidValue(message) => write!(f, "{message}"), + ImportError::Database(err) => write!(f, "{err}"), + } + } +} + +impl std::error::Error for ImportError {} + +fn validate_document(document: &ReadingProgressExportDocument) -> Result<(), ImportError> { + if document.format != READING_PROGRESS_FORMAT { + return Err(ImportError::UnknownFormat(document.format.clone())); + } + if document.version > READING_PROGRESS_VERSION { + return Err(ImportError::UnsupportedVersion(document.version)); + } + // Ratings feed a per-series average every user sees, so a file must not + // be a way around the range the rating endpoint enforces: an out-of-range + // value would skew that average for everyone, and a large one overflows + // the sum behind it. + for series in &document.series { + if let Some(rating) = series.rating + && !(1..=100).contains(&rating) + { + return Err(ImportError::InvalidValue(format!( + "series '{}' has rating {rating}; ratings must be between 1 and 100", + series.name + ))); + } + } + Ok(()) +} + +/// A session's reading time can never exceed the span it covers. The normal +/// write path clamps to that span for the same reason: a larger figure is a +/// bug or a fabrication, and one huge value is enough to overflow the +/// statistics sums for the reader. +fn clamped_duration(doc: &ExportSessionDto) -> Option { + let span = (doc.client_ended_at - doc.client_started_at).num_milliseconds(); + doc.active_duration_ms + .map(|reported| reported.clamp(0, span.max(0))) +} + +// --------------------------------------------------------------------------- +// Decisions: pure data describing what would happen, computed once and used +// both to build the report and (for a real import) to drive the writes. +// --------------------------------------------------------------------------- + +#[derive(Debug, Clone)] +struct ProgressValues { + current_page: i32, + progress_percentage: Option, + completed: bool, + started_at: DateTime, + updated_at: DateTime, + completed_at: Option>, + r2_progression: Option, +} + +#[derive(Debug, Clone)] +enum ProgressDecision { + /// The id is minted when planning, so a later entry for the same book can + /// be planned as an update of this row before it exists. + Insert(Uuid, ProgressValues), + Update(Uuid, ProgressValues), + Skip, +} + +impl ProgressDecision { + fn outcome(&self) -> FieldOutcome { + match self { + ProgressDecision::Insert(_, _) => FieldOutcome::Inserted, + ProgressDecision::Update(_, _) => FieldOutcome::Updated, + ProgressDecision::Skip => FieldOutcome::Skipped, + } + } +} + +/// How far into a book a position is, for the `furthest` conflict policy. A +/// completed read always wins: it is further than any partial position could +/// be, regardless of which position field the format happens to use. +fn progress_score(completed: bool, percentage: Option, page: i32) -> f64 { + if completed { + f64::INFINITY + } else if let Some(p) = percentage { + p + } else { + page as f64 + } +} + +fn decide_progress( + existing: Option<&read_progress::Model>, + imported: &super::model::ExportProgressDto, + policy: ConflictPolicy, +) -> ProgressDecision { + let values = ProgressValues { + current_page: imported.current_page, + progress_percentage: imported.progress_percentage, + completed: imported.completed, + started_at: imported.started_at, + updated_at: imported.updated_at, + completed_at: imported.completed_at, + r2_progression: imported.r2_progression.clone(), + }; + + let Some(existing) = existing else { + return ProgressDecision::Insert(Uuid::new_v4(), values); + }; + + match policy { + ConflictPolicy::SkipExisting => ProgressDecision::Skip, + ConflictPolicy::Overwrite => ProgressDecision::Update(existing.id, values), + ConflictPolicy::Newest => { + if imported.updated_at > existing.updated_at { + ProgressDecision::Update(existing.id, values) + } else { + ProgressDecision::Skip + } + } + ConflictPolicy::Furthest => { + let existing_score = progress_score( + existing.completed, + existing.progress_percentage, + existing.current_page, + ); + let imported_score = progress_score( + imported.completed, + imported.progress_percentage, + imported.current_page, + ); + if imported_score > existing_score { + ProgressDecision::Update(existing.id, values) + } else { + ProgressDecision::Skip + } + } + } +} + +#[derive(Debug, Clone)] +enum RatingDecision { + NoOp, + Insert { + id: Uuid, + rating: i32, + notes: Option, + updated_at: DateTime, + }, + Update { + existing: user_series_ratings::Model, + rating: i32, + notes: Option, + updated_at: DateTime, + }, + Skip, +} + +impl RatingDecision { + fn outcome(&self) -> Option { + match self { + RatingDecision::NoOp => None, + RatingDecision::Insert { .. } => Some(FieldOutcome::Inserted), + RatingDecision::Update { .. } => Some(FieldOutcome::Updated), + RatingDecision::Skip => Some(FieldOutcome::Skipped), + } + } +} + +/// Decide a series rating under the conflict policy. `newest` compares the +/// file's `rating_updated_at` with the existing row's `updated_at`; see the +/// comments inside for `furthest` and for a file without the timestamp. +fn decide_rating( + existing: Option<&user_series_ratings::Model>, + series_doc: &ExportSeriesDto, + policy: ConflictPolicy, +) -> RatingDecision { + let Some(rating) = series_doc.rating else { + return RatingDecision::NoOp; + }; + let notes = series_doc.notes.clone(); + // The rating keeps the time it was actually set, so a second import of the + // same file compares equal and is skipped rather than rewritten. + let updated_at = series_doc.rating_updated_at.unwrap_or_else(Utc::now); + + match existing { + None => RatingDecision::Insert { + id: Uuid::new_v4(), + rating, + notes, + updated_at, + }, + Some(existing) => { + let replace = match policy { + ConflictPolicy::SkipExisting => false, + ConflictPolicy::Overwrite => true, + // A rating has no position to be "further" along, so furthest + // reads as newest. A file without the timestamp cannot show it + // is newer, and an old export must not silently undo a rating + // changed since, so it loses. + ConflictPolicy::Newest | ConflictPolicy::Furthest => series_doc + .rating_updated_at + .is_some_and(|imported| imported > existing.updated_at), + }; + if replace { + RatingDecision::Update { + existing: existing.clone(), + rating, + notes, + updated_at, + } + } else { + RatingDecision::Skip + } + } + } +} + +/// What happens to one append-only row (`read_completions` / `reading_sessions`) +/// whose primary key is the id reused from the export. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum RowDecision { + /// No row with this id exists yet. + Insert, + /// The row exists and is the importing user's own, but it is not on a + /// live book: its book was hard-deleted (`book_id IS NULL`), or it still + /// points at a book the scanner has marked deleted because the file moved. + /// Move it onto the matched book. + Reattach, + /// Already on a live book (re-importing the same file), listed twice in + /// the file, or belongs to someone else (a UUID collision that should + /// never happen, handled by leaving it alone). + Skip, +} + +/// An existing history row, as much as a decision needs. +#[derive(Debug, Clone, Copy)] +struct ExistingRow { + user_id: Uuid, + book_id: Option, +} + +/// What this import has already planned to write, so each entry is decided +/// against the state the earlier ones will leave rather than against the +/// database as it was before the import started. +/// +/// Without it, two entries that land on the same book (the scanner keeps a +/// moved file's old row soft-deleted, and the export carries both) are each +/// planned against an empty table: both plan an insert, the second hits the +/// unique index, and the whole series rolls back. The dry run, deciding the +/// same way, would have promised both. +#[derive(Debug, Clone, Default)] +struct PlannedState { + /// The progress row each target book will hold, by book id. + progress: HashMap, + /// The rating each target series will hold, by series id. + ratings: HashMap, + /// Completion and session ids already claimed by an earlier entry. + rows: HashSet, +} + +/// Bind-parameter-safe batch size for `IN (...)` lookups. +const LOOKUP_BATCH: usize = 1_000; + +async fn existing_completions( + db: &DatabaseConnection, + ids: &[Uuid], +) -> Result> { + let mut found = HashMap::new(); + for chunk in ids.chunks(LOOKUP_BATCH) { + for row in read_completions::Entity::find() + .filter(read_completions::Column::Id.is_in(chunk.to_vec())) + .all(db) + .await? + { + found.insert( + row.id, + ExistingRow { + user_id: row.user_id, + book_id: row.book_id, + }, + ); + } + } + Ok(found) +} + +async fn existing_sessions( + db: &DatabaseConnection, + ids: &[Uuid], +) -> Result> { + let mut found = HashMap::new(); + for chunk in ids.chunks(LOOKUP_BATCH) { + for row in reading_sessions::Entity::find() + .filter(reading_sessions::Column::Id.is_in(chunk.to_vec())) + .all(db) + .await? + { + found.insert( + row.id, + ExistingRow { + user_id: row.user_id, + book_id: row.book_id, + }, + ); + } + } + Ok(found) +} + +/// Which of these books the scanner has marked deleted. +async fn soft_deleted_books(db: &DatabaseConnection, ids: &[Uuid]) -> Result> { + let mut found = HashSet::new(); + for chunk in ids.chunks(LOOKUP_BATCH) { + for book in books::Entity::find() + .filter(books::Column::Id.is_in(chunk.to_vec())) + .filter(books::Column::Deleted.eq(true)) + .all(db) + .await? + { + found.insert(book.id); + } + } + Ok(found) +} + +async fn progress_for_books( + db: &DatabaseConnection, + user_id: Uuid, + book_ids: &[Uuid], +) -> Result> { + let mut found = HashMap::new(); + for chunk in book_ids.chunks(LOOKUP_BATCH) { + for row in read_progress::Entity::find() + .filter(read_progress::Column::UserId.eq(user_id)) + .filter(read_progress::Column::BookId.is_in(chunk.to_vec())) + .all(db) + .await? + { + found.insert(row.book_id, row); + } + } + Ok(found) +} + +/// Everything already in the database that a series' decisions depend on, +/// fetched in a handful of batched queries rather than one per row. +struct SeriesLookups { + progress: HashMap, + completions: HashMap, + sessions: HashMap, + soft_deleted: HashSet, +} + +fn decide_row( + id: Uuid, + existing: Option<&ExistingRow>, + target_book: Uuid, + user_id: Uuid, + soft_deleted: &HashSet, + reattach: bool, + planned: &mut PlannedState, +) -> RowDecision { + if !planned.rows.insert(id) { + return RowDecision::Skip; + } + match existing { + None => RowDecision::Insert, + Some(row) if row.user_id == user_id => { + let off_a_live_book = match row.book_id { + None => true, + Some(book) => book != target_book && soft_deleted.contains(&book), + }; + if off_a_live_book && reattach { + RowDecision::Reattach + } else { + RowDecision::Skip + } + } + Some(_) => RowDecision::Skip, + } +} + +/// Everything decided for one exported book, before any write happens. +struct BookPlan<'a> { + book_doc: &'a ExportBookDto, + disposition: BookDisposition, + matched_book_id: Option, + applied: bool, + progress: Option, + completions: Vec<(&'a ExportCompletionDto, RowDecision)>, + sessions: Vec<(&'a ExportSessionDto, RowDecision)>, +} + +fn build_book_plan<'a>( + user_id: Uuid, + book_doc: &'a ExportBookDto, + candidates: &[BookCandidate], + lookups: &SeriesLookups, + planned: &mut PlannedState, + options: &ImportOptions, +) -> BookPlan<'a> { + let book_match = matching::resolve_book(candidates, book_doc, options.hash_mode); + + let (disposition, matched_book_id, applied) = match book_match { + BookMatch::Matched(id) => (BookDisposition::Matched, Some(id), true), + BookMatch::StemMatch(id) => ( + BookDisposition::StemMatch, + Some(id), + options.accept_stem_matches, + ), + BookMatch::Ambiguous => (BookDisposition::Ambiguous, None, false), + BookMatch::Unmatched => (BookDisposition::Unmatched, None, false), + BookMatch::HashMismatch => (BookDisposition::HashMismatch, None, false), + }; + + let Some(book_id) = matched_book_id.filter(|_| applied) else { + return BookPlan { + book_doc, + disposition, + matched_book_id, + applied: false, + progress: None, + completions: vec![], + sessions: vec![], + }; + }; + + let progress = book_doc.progress.as_ref().map(|imported| { + let existing = planned + .progress + .get(&book_id) + .or_else(|| lookups.progress.get(&book_id)) + .cloned(); + let decision = decide_progress(existing.as_ref(), imported, options.conflict_policy); + let planned_row = match &decision { + ProgressDecision::Insert(id, values) | ProgressDecision::Update(id, values) => { + Some(progress_model(*id, user_id, book_id, values)) + } + ProgressDecision::Skip => existing, + }; + if let Some(row) = planned_row { + planned.progress.insert(book_id, row); + } + decision + }); + + let completions = book_doc + .completions + .iter() + .map(|completion| { + let decision = decide_row( + completion.id, + lookups.completions.get(&completion.id), + book_id, + user_id, + &lookups.soft_deleted, + options.reattach_sessions, + planned, + ); + (completion, decision) + }) + .collect(); + + let sessions = book_doc + .sessions + .iter() + .flatten() + .map(|session| { + let decision = decide_row( + session.id, + lookups.sessions.get(&session.id), + book_id, + user_id, + &lookups.soft_deleted, + options.reattach_sessions, + planned, + ); + (session, decision) + }) + .collect(); + + BookPlan { + book_doc, + disposition, + matched_book_id, + applied: true, + progress, + completions, + sessions, + } +} + +fn progress_model( + id: Uuid, + user_id: Uuid, + book_id: Uuid, + values: &ProgressValues, +) -> read_progress::Model { + read_progress::Model { + id, + user_id, + book_id, + current_page: values.current_page, + progress_percentage: values.progress_percentage, + completed: values.completed, + started_at: values.started_at, + updated_at: values.updated_at, + completed_at: values.completed_at, + r2_progression: values.r2_progression.clone(), + } +} + +fn book_report_from_plan(plan: &BookPlan<'_>) -> ImportBookReport { + let mut completions = WriteCounts::default(); + for (_, decision) in &plan.completions { + match decision { + RowDecision::Insert => completions.inserted += 1, + RowDecision::Reattach => completions.reattached += 1, + RowDecision::Skip => completions.skipped += 1, + } + } + let mut sessions = WriteCounts::default(); + for (_, decision) in &plan.sessions { + match decision { + RowDecision::Insert => sessions.inserted += 1, + RowDecision::Reattach => sessions.reattached += 1, + RowDecision::Skip => sessions.skipped += 1, + } + } + + ImportBookReport { + path: plan.book_doc.path.clone(), + file_name: plan.book_doc.file_name.clone(), + disposition: plan.disposition, + matched_book_id: plan.matched_book_id, + applied: plan.applied, + progress: plan.progress.as_ref().map(ProgressDecision::outcome), + completions, + sessions, + } +} + +/// Whether the series rating write counts as a real write for the summary, +/// following the same `count_writes` gate as [`tally_book_report`]. +fn tally_rating_write(summary: &mut ImportSummary, rating_decision: &RatingDecision) { + if matches!( + rating_decision.outcome(), + Some(FieldOutcome::Inserted) | Some(FieldOutcome::Updated) + ) { + summary.ratings_written += 1; + } +} + +/// Fold one book's report into the running summary. `count_writes` is false +/// for a series that was matched but whose transaction failed to commit: the +/// disposition counts (how many books matched, were ambiguous, etc.) are +/// still meaningful, but nothing was actually written. +fn tally_book_report(summary: &mut ImportSummary, report: &ImportBookReport, count_writes: bool) { + summary.books_total += 1; + match report.disposition { + BookDisposition::Matched => summary.books_matched += 1, + BookDisposition::StemMatch => summary.books_stem_matched += 1, + BookDisposition::Ambiguous => summary.books_ambiguous += 1, + BookDisposition::Unmatched => summary.books_unmatched += 1, + BookDisposition::HashMismatch => summary.books_hash_mismatch += 1, + } + if !count_writes { + return; + } + if matches!( + report.progress, + Some(FieldOutcome::Inserted) | Some(FieldOutcome::Updated) + ) { + summary.progress_written += 1; + } + summary.completions_inserted += report.completions.inserted; + summary.completions_reattached += report.completions.reattached; + summary.sessions_inserted += report.sessions.inserted; + summary.sessions_reattached += report.sessions.reattached; +} + +// --------------------------------------------------------------------------- +// Execution: only reached for a real (non-dry-run) import. +// --------------------------------------------------------------------------- + +async fn apply_progress( + txn: &DatabaseTransaction, + user_id: Uuid, + book_id: Uuid, + decision: &ProgressDecision, +) -> Result<()> { + match decision { + ProgressDecision::Skip => Ok(()), + ProgressDecision::Insert(id, values) => { + read_progress::ActiveModel { + id: Set(*id), + user_id: Set(user_id), + book_id: Set(book_id), + current_page: Set(values.current_page), + progress_percentage: Set(values.progress_percentage), + completed: Set(values.completed), + started_at: Set(values.started_at), + updated_at: Set(values.updated_at), + completed_at: Set(values.completed_at), + r2_progression: Set(values.r2_progression.clone()), + } + .insert(txn) + .await?; + Ok(()) + } + ProgressDecision::Update(id, values) => { + read_progress::ActiveModel { + id: Set(*id), + user_id: Set(user_id), + book_id: Set(book_id), + current_page: Set(values.current_page), + progress_percentage: Set(values.progress_percentage), + completed: Set(values.completed), + started_at: Set(values.started_at), + updated_at: Set(values.updated_at), + completed_at: Set(values.completed_at), + r2_progression: Set(values.r2_progression.clone()), + } + .update(txn) + .await?; + Ok(()) + } + } +} + +async fn apply_completion( + txn: &DatabaseTransaction, + user_id: Uuid, + book_id: Uuid, + doc: &ExportCompletionDto, + decision: &RowDecision, +) -> Result<()> { + match decision { + RowDecision::Skip => Ok(()), + RowDecision::Insert => { + read_completions::ActiveModel { + id: Set(doc.id), + user_id: Set(user_id), + book_id: Set(Some(book_id)), + started_at: Set(doc.started_at), + completed_at: Set(doc.completed_at), + } + .insert(txn) + .await?; + Ok(()) + } + RowDecision::Reattach => { + read_completions::Entity::update_many() + .col_expr(read_completions::Column::BookId, Expr::value(book_id)) + .filter(read_completions::Column::Id.eq(doc.id)) + .filter(read_completions::Column::UserId.eq(user_id)) + .exec(txn) + .await?; + Ok(()) + } + } +} + +async fn apply_session( + txn: &DatabaseTransaction, + user_id: Uuid, + book_id: Uuid, + doc: &ExportSessionDto, + decision: &RowDecision, +) -> Result<()> { + match decision { + RowDecision::Skip => Ok(()), + RowDecision::Insert => { + reading_sessions::ActiveModel { + id: Set(doc.id), + user_id: Set(user_id), + book_id: Set(Some(book_id)), + device_id: Set(doc.device_id.clone()), + device_name: Set(doc.device_name.clone()), + pass: Set(doc.pass.max(1)), + kind: Set(doc.kind.clone()), + to_page: Set(doc.to_page), + to_percentage: Set(doc.to_percentage), + // The export never carries a historical locator; only the + // live `read_progress` row keeps one. + r2_progression: Set(None), + active_duration_ms: Set(clamped_duration(doc)), + duration_source: Set(doc.duration_source.clone()), + pages_read: Set(doc.pages_read.map(|pages| pages.max(0))), + client_started_at: Set(doc.client_started_at), + client_ended_at: Set(doc.client_ended_at), + server_recorded_at: Set(doc.server_recorded_at), + } + .insert(txn) + .await?; + Ok(()) + } + RowDecision::Reattach => { + reading_sessions::Entity::update_many() + .col_expr(reading_sessions::Column::BookId, Expr::value(book_id)) + .filter(reading_sessions::Column::Id.eq(doc.id)) + .filter(reading_sessions::Column::UserId.eq(user_id)) + .exec(txn) + .await?; + Ok(()) + } + } +} + +async fn apply_rating( + txn: &DatabaseTransaction, + user_id: Uuid, + series_id: Uuid, + decision: &RatingDecision, +) -> Result<()> { + match decision { + RatingDecision::NoOp | RatingDecision::Skip => Ok(()), + RatingDecision::Insert { + id, + rating, + notes, + updated_at, + } => { + user_series_ratings::ActiveModel { + id: Set(*id), + user_id: Set(user_id), + series_id: Set(series_id), + rating: Set(*rating), + notes: Set(notes.clone()), + created_at: Set(Utc::now()), + updated_at: Set(*updated_at), + } + .insert(txn) + .await?; + Ok(()) + } + RatingDecision::Update { + existing, + rating, + notes, + updated_at, + } => { + let mut active: user_series_ratings::ActiveModel = existing.clone().into(); + active.rating = Set(*rating); + active.notes = Set(notes.clone()); + active.updated_at = Set(*updated_at); + active.update(txn).await?; + Ok(()) + } + } +} + +async fn apply_series( + db: &DatabaseConnection, + user_id: Uuid, + series_id: Uuid, + plans: &[BookPlan<'_>], + rating_decision: &RatingDecision, +) -> Result<()> { + let txn = db.begin().await?; + + for plan in plans { + if !plan.applied { + continue; + } + let book_id = plan + .matched_book_id + .expect("applied implies a matched book id"); + + if let Some(progress_decision) = &plan.progress { + apply_progress(&txn, user_id, book_id, progress_decision).await?; + } + for (completion_doc, decision) in &plan.completions { + apply_completion(&txn, user_id, book_id, completion_doc, decision).await?; + } + for (session_doc, decision) in &plan.sessions { + apply_session(&txn, user_id, book_id, session_doc, decision).await?; + } + } + + apply_rating(&txn, user_id, series_id, rating_decision).await?; + + txn.commit().await?; + Ok(()) +} + +// --------------------------------------------------------------------------- +// Series-level orchestration +// --------------------------------------------------------------------------- + +async fn plan_series<'a>( + db: &DatabaseConnection, + user_id: Uuid, + series_id: Uuid, + series_doc: &'a ExportSeriesDto, + planned: &mut PlannedState, + options: &ImportOptions, +) -> Result<(Vec>, RatingDecision)> { + let series_row = SeriesRepository::get_by_id(db, series_id) + .await? + .ok_or_else(|| anyhow::anyhow!("matched series {series_id} vanished during import"))?; + let library_row = LibraryRepository::get_by_id(db, series_row.library_id) + .await? + .ok_or_else(|| { + anyhow::anyhow!("library {} vanished during import", series_row.library_id) + })?; + + let book_models = BookRepository::list_by_series(db, series_id, false).await?; + let candidates: Vec = book_models + .iter() + .map(|b| BookCandidate::from_model(b, &library_row.path, &series_row.path)) + .collect(); + + let book_ids: Vec = candidates.iter().map(|c| c.id).collect(); + let completion_ids: Vec = series_doc + .books + .iter() + .flat_map(|b| b.completions.iter().map(|c| c.id)) + .collect(); + let session_ids: Vec = series_doc + .books + .iter() + .flat_map(|b| b.sessions.iter().flatten().map(|s| s.id)) + .collect(); + let completions = existing_completions(db, &completion_ids).await?; + let sessions = existing_sessions(db, &session_ids).await?; + let referenced: Vec = completions + .values() + .chain(sessions.values()) + .filter_map(|row| row.book_id) + .collect::>() + .into_iter() + .collect(); + let lookups = SeriesLookups { + progress: progress_for_books(db, user_id, &book_ids).await?, + soft_deleted: soft_deleted_books(db, &referenced).await?, + completions, + sessions, + }; + + let plans = series_doc + .books + .iter() + .map(|book_doc| build_book_plan(user_id, book_doc, &candidates, &lookups, planned, options)) + .collect(); + + let existing_rating = match planned.ratings.get(&series_id) { + Some(row) => Some(row.clone()), + None => UserSeriesRatingRepository::get_by_user_and_series(db, user_id, series_id).await?, + }; + let rating_decision = decide_rating( + existing_rating.as_ref(), + series_doc, + options.conflict_policy, + ); + let planned_rating = match &rating_decision { + RatingDecision::Insert { + id, + rating, + notes, + updated_at, + } => Some(user_series_ratings::Model { + id: *id, + user_id, + series_id, + rating: *rating, + notes: notes.clone(), + created_at: *updated_at, + updated_at: *updated_at, + }), + RatingDecision::Update { + existing, + rating, + notes, + updated_at, + } => Some(user_series_ratings::Model { + rating: *rating, + notes: notes.clone(), + updated_at: *updated_at, + ..existing.clone() + }), + RatingDecision::NoOp | RatingDecision::Skip => existing_rating, + }; + if let Some(row) = planned_rating { + planned.ratings.insert(series_id, row); + } + + Ok((plans, rating_decision)) +} + +fn unresolved_book_reports( + series_doc: &ExportSeriesDto, + disposition: BookDisposition, +) -> Vec { + series_doc + .books + .iter() + .map(|b| ImportBookReport { + path: b.path.clone(), + file_name: b.file_name.clone(), + disposition, + matched_book_id: None, + applied: false, + progress: None, + completions: WriteCounts::default(), + sessions: WriteCounts::default(), + }) + .collect() +} + +/// The report for a series that could not be planned at all. +fn failed_series_report( + summary: &mut ImportSummary, + series_doc: &ExportSeriesDto, + disposition: SeriesDisposition, + matched_series_id: Option, + err: anyhow::Error, +) -> ImportSeriesReport { + let books = unresolved_book_reports(series_doc, BookDisposition::Unmatched); + for report in &books { + tally_book_report(summary, report, false); + } + ImportSeriesReport { + library_relative_path: series_doc.library_relative_path.clone(), + name: series_doc.name.clone(), + disposition, + matched_series_id, + attempted: matched_series_id.is_some(), + committed: false, + error: Some(err.to_string()), + rating: None, + books, + } +} + +/// Apply (or dry-run) a matched export document for one user. +pub async fn import_reading_progress( + db: &DatabaseConnection, + user_id: Uuid, + document: &ReadingProgressExportDocument, + options: &ImportOptions, +) -> Result { + validate_document(document)?; + + let content_filter = ContentFilter::for_user(db, user_id) + .await + .map_err(ImportError::Database)?; + + let mut notices = Vec::new(); + if options.reattach_sessions && !document.includes_sessions { + notices.push( + "reattach_sessions has no effect: the imported file does not include sessions." + .to_string(), + ); + } + + let mut summary = ImportSummary::default(); + let mut series_reports = Vec::with_capacity(document.series.len()); + let mut planned = PlannedState::default(); + + for series_doc in &document.series { + summary.series_total += 1; + + // A failure in one series is reported on that series and the import + // carries on: earlier series may already have committed, and the + // report is the only place the reader learns which did. + let series_match = match matching::resolve_series( + db, + &content_filter, + series_doc, + &options.source_preference, + ) + .await + { + Ok(found) => found, + Err(err) => { + summary.series_unmatched += 1; + series_reports.push(failed_series_report( + &mut summary, + series_doc, + SeriesDisposition::Unmatched, + None, + err, + )); + continue; + } + }; + + let series_id = match series_match { + SeriesMatch::Matched(series_id) => series_id, + SeriesMatch::Ambiguous | SeriesMatch::Unmatched => { + let (disposition, book_disposition) = if series_match == SeriesMatch::Ambiguous { + summary.series_ambiguous += 1; + (SeriesDisposition::Ambiguous, BookDisposition::Ambiguous) + } else { + summary.series_unmatched += 1; + (SeriesDisposition::Unmatched, BookDisposition::Unmatched) + }; + let books = unresolved_book_reports(series_doc, book_disposition); + for report in &books { + tally_book_report(&mut summary, report, false); + } + series_reports.push(ImportSeriesReport { + library_relative_path: series_doc.library_relative_path.clone(), + name: series_doc.name.clone(), + disposition, + matched_series_id: None, + attempted: false, + committed: false, + error: None, + rating: None, + books, + }); + continue; + } + }; + summary.series_matched += 1; + + // Restored if this series does not land, so later series are not + // planned against writes that never happened. + let before_series = planned.clone(); + + let (plans, rating_decision) = + match plan_series(db, user_id, series_id, series_doc, &mut planned, options).await { + Ok(planned_series) => planned_series, + Err(err) => { + planned = before_series; + series_reports.push(failed_series_report( + &mut summary, + series_doc, + SeriesDisposition::Matched, + Some(series_id), + err, + )); + continue; + } + }; + + let outcome = if options.dry_run { + Ok(false) + } else { + apply_series(db, user_id, series_id, &plans, &rating_decision) + .await + .map(|()| true) + }; + + let books: Vec = plans.iter().map(book_report_from_plan).collect(); + match outcome { + Ok(committed) => { + if committed { + summary.series_committed += 1; + } + for report in &books { + tally_book_report(&mut summary, report, true); + } + tally_rating_write(&mut summary, &rating_decision); + series_reports.push(ImportSeriesReport { + library_relative_path: series_doc.library_relative_path.clone(), + name: series_doc.name.clone(), + disposition: SeriesDisposition::Matched, + matched_series_id: Some(series_id), + attempted: true, + committed, + error: None, + rating: rating_decision.outcome(), + books, + }); + } + Err(err) => { + planned = before_series; + for report in &books { + tally_book_report(&mut summary, report, false); + } + series_reports.push(ImportSeriesReport { + library_relative_path: series_doc.library_relative_path.clone(), + name: series_doc.name.clone(), + disposition: SeriesDisposition::Matched, + matched_series_id: Some(series_id), + attempted: true, + committed: false, + error: Some(err.to_string()), + rating: None, + books, + }); + } + } + } + + Ok(ImportReadingProgressResponse { + dry_run: options.dry_run, + sessions_in_file: document.includes_sessions, + notices, + summary, + series: series_reports, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::reading_transfer::model::ExportProgressDto; + + fn progress_row( + current_page: i32, + completed: bool, + updated_at: DateTime, + ) -> read_progress::Model { + read_progress::Model { + id: Uuid::new_v4(), + user_id: Uuid::new_v4(), + book_id: Uuid::new_v4(), + current_page, + progress_percentage: None, + completed, + started_at: updated_at, + updated_at, + completed_at: if completed { Some(updated_at) } else { None }, + r2_progression: None, + } + } + + fn imported_progress( + current_page: i32, + completed: bool, + updated_at: DateTime, + ) -> ExportProgressDto { + ExportProgressDto { + current_page, + progress_percentage: None, + completed, + started_at: updated_at, + updated_at, + completed_at: if completed { Some(updated_at) } else { None }, + r2_progression: None, + } + } + + fn t(minutes: i64) -> DateTime { + Utc.with_ymd_and_hms(2026, 1, 1, 0, 0, 0).unwrap() + chrono::Duration::minutes(minutes) + } + + use chrono::TimeZone; + + #[test] + fn progress_inserts_when_nothing_exists() { + let imported = imported_progress(10, false, t(0)); + let decision = decide_progress(None, &imported, ConflictPolicy::Newest); + assert!(matches!(decision, ProgressDecision::Insert(_, _))); + } + + #[test] + fn progress_skip_existing_leaves_existing_row_alone() { + let existing = progress_row(5, false, t(0)); + let imported = imported_progress(50, false, t(10)); + let decision = decide_progress(Some(&existing), &imported, ConflictPolicy::SkipExisting); + assert!(matches!(decision, ProgressDecision::Skip)); + } + + #[test] + fn progress_overwrite_always_replaces() { + let existing = progress_row(50, false, t(10)); + let imported = imported_progress(1, false, t(0)); + let decision = decide_progress(Some(&existing), &imported, ConflictPolicy::Overwrite); + assert!(matches!(decision, ProgressDecision::Update(_, _))); + } + + #[test] + fn progress_newest_picks_the_later_updated_at() { + let existing = progress_row(5, false, t(0)); + let newer_import = imported_progress(3, false, t(10)); + assert!(matches!( + decide_progress(Some(&existing), &newer_import, ConflictPolicy::Newest), + ProgressDecision::Update(_, _) + )); + + let older_import = imported_progress(999, false, t(-10)); + assert!(matches!( + decide_progress(Some(&existing), &older_import, ConflictPolicy::Newest), + ProgressDecision::Skip + )); + } + + #[test] + fn progress_furthest_prefers_the_higher_position() { + let existing = progress_row(10, false, t(0)); + let further_import = imported_progress(50, false, t(-100)); // older, but further + assert!(matches!( + decide_progress(Some(&existing), &further_import, ConflictPolicy::Furthest), + ProgressDecision::Update(_, _) + )); + + let behind_import = imported_progress(2, false, t(100)); // newer, but behind + assert!(matches!( + decide_progress(Some(&existing), &behind_import, ConflictPolicy::Furthest), + ProgressDecision::Skip + )); + } + + #[test] + fn progress_furthest_treats_completed_as_further_than_any_partial_position() { + let existing = progress_row(999, false, t(0)); + let completed_import = imported_progress(1, true, t(-100)); + assert!(matches!( + decide_progress(Some(&existing), &completed_import, ConflictPolicy::Furthest), + ProgressDecision::Update(_, _) + )); + } + + fn rating_row(rating: i32) -> user_series_ratings::Model { + user_series_ratings::Model { + id: Uuid::new_v4(), + user_id: Uuid::new_v4(), + series_id: Uuid::new_v4(), + rating, + notes: None, + created_at: t(0), + updated_at: t(0), + } + } + + fn series_with_rating(rating: Option) -> ExportSeriesDto { + rated_at(rating, None) + } + + fn rated_at(rating: Option, when: Option>) -> ExportSeriesDto { + ExportSeriesDto { + external_ids: vec![], + library_relative_path: "s".to_string(), + name: "S".to_string(), + rating, + notes: None, + rating_updated_at: when, + books: vec![], + } + } + + #[test] + fn rating_no_op_when_file_carries_none() { + let doc = series_with_rating(None); + assert!(matches!( + decide_rating(None, &doc, ConflictPolicy::Overwrite), + RatingDecision::NoOp + )); + } + + #[test] + fn rating_inserts_when_nothing_exists() { + let doc = series_with_rating(Some(80)); + assert!(matches!( + decide_rating(None, &doc, ConflictPolicy::Newest), + RatingDecision::Insert { .. } + )); + } + + #[test] + fn rating_skip_existing_leaves_existing_alone() { + let existing = rating_row(40); + let doc = series_with_rating(Some(90)); + assert!(matches!( + decide_rating(Some(&existing), &doc, ConflictPolicy::SkipExisting), + RatingDecision::Skip + )); + } + + #[test] + fn rating_overwrite_always_replaces() { + let existing = rating_row(40); + let doc = series_with_rating(Some(90)); + assert!(matches!( + decide_rating(Some(&existing), &doc, ConflictPolicy::Overwrite), + RatingDecision::Update { .. } + )); + } + + /// The default policy must not let an old export undo a rating the reader + /// changed after taking it. + #[test] + fn rating_newest_replaces_only_a_strictly_older_rating() { + let existing = rating_row(40); // updated at t(0) + for policy in [ConflictPolicy::Newest, ConflictPolicy::Furthest] { + assert!(matches!( + decide_rating(Some(&existing), &rated_at(Some(90), Some(t(10))), policy), + RatingDecision::Update { .. } + )); + assert!(matches!( + decide_rating(Some(&existing), &rated_at(Some(90), Some(t(-10))), policy), + RatingDecision::Skip + )); + assert!( + matches!( + decide_rating(Some(&existing), &rated_at(Some(90), Some(t(0))), policy), + RatingDecision::Skip + ), + "an equal timestamp is the same rating re-imported" + ); + assert!( + matches!( + decide_rating(Some(&existing), &series_with_rating(Some(90)), policy), + RatingDecision::Skip + ), + "without a timestamp the file cannot show it is newer" + ); + } + } + + #[test] + fn rejects_unknown_format() { + let doc = ReadingProgressExportDocument { + format: "something-else".to_string(), + version: 1, + exported_at: t(0), + includes_sessions: true, + series: vec![], + }; + assert!(matches!( + validate_document(&doc), + Err(ImportError::UnknownFormat(_)) + )); + } + + #[test] + fn rejects_a_version_newer_than_supported() { + let doc = ReadingProgressExportDocument { + format: READING_PROGRESS_FORMAT.to_string(), + version: READING_PROGRESS_VERSION + 1, + exported_at: t(0), + includes_sessions: true, + series: vec![], + }; + assert!(matches!( + validate_document(&doc), + Err(ImportError::UnsupportedVersion(_)) + )); + } + + #[test] + fn accepts_the_current_format_and_version() { + let doc = ReadingProgressExportDocument { + format: READING_PROGRESS_FORMAT.to_string(), + version: READING_PROGRESS_VERSION, + exported_at: t(0), + includes_sessions: true, + series: vec![], + }; + assert!(validate_document(&doc).is_ok()); + } + + // ------------------------------------------------------------------ + // Row decisions: insert / reattach / skip for completions and sessions. + // ------------------------------------------------------------------ + + fn row(user_id: Uuid, book_id: Option) -> ExistingRow { + ExistingRow { user_id, book_id } + } + + #[test] + fn row_inserts_when_absent() { + let mut planned = PlannedState::default(); + let decision = decide_row( + Uuid::new_v4(), + None, + Uuid::new_v4(), + Uuid::new_v4(), + &HashSet::new(), + true, + &mut planned, + ); + assert_eq!(decision, RowDecision::Insert); + } + + #[test] + fn row_reattaches_an_orphan_only_when_the_flag_is_on() { + let user = Uuid::new_v4(); + let target = Uuid::new_v4(); + let orphan = row(user, None); + for (flag, expected) in [(true, RowDecision::Reattach), (false, RowDecision::Skip)] { + let mut planned = PlannedState::default(); + let decision = decide_row( + Uuid::new_v4(), + Some(&orphan), + target, + user, + &HashSet::new(), + flag, + &mut planned, + ); + assert_eq!(decision, expected); + } + } + + /// A moved file leaves its old book soft-deleted with the history still + /// on it; importing onto the new book moves that history across. + #[test] + fn row_on_a_soft_deleted_book_moves_to_the_matched_book() { + let user = Uuid::new_v4(); + let old_book = Uuid::new_v4(); + let mut planned = PlannedState::default(); + let decision = decide_row( + Uuid::new_v4(), + Some(&row(user, Some(old_book))), + Uuid::new_v4(), + user, + &HashSet::from([old_book]), + true, + &mut planned, + ); + assert_eq!(decision, RowDecision::Reattach); + } + + #[test] + fn row_on_a_live_book_is_left_alone() { + let user = Uuid::new_v4(); + let live = Uuid::new_v4(); + let mut planned = PlannedState::default(); + let decision = decide_row( + Uuid::new_v4(), + Some(&row(user, Some(live))), + Uuid::new_v4(), + user, + &HashSet::new(), + true, + &mut planned, + ); + assert_eq!(decision, RowDecision::Skip); + } + + #[test] + fn another_users_row_is_never_touched() { + let mut planned = PlannedState::default(); + let decision = decide_row( + Uuid::new_v4(), + Some(&row(Uuid::new_v4(), None)), + Uuid::new_v4(), + Uuid::new_v4(), + &HashSet::new(), + true, + &mut planned, + ); + assert_eq!(decision, RowDecision::Skip); + } + + #[test] + fn an_id_listed_twice_is_written_once() { + let id = Uuid::new_v4(); + let mut planned = PlannedState::default(); + let args = |planned: &mut PlannedState| { + decide_row( + id, + None, + Uuid::new_v4(), + Uuid::new_v4(), + &HashSet::new(), + true, + planned, + ) + }; + assert_eq!(args(&mut planned), RowDecision::Insert); + assert_eq!(args(&mut planned), RowDecision::Skip); + } +} diff --git a/crates/codex-services/src/reading_transfer/matching.rs b/crates/codex-services/src/reading_transfer/matching.rs new file mode 100644 index 000000000..88e5f64cc --- /dev/null +++ b/crates/codex-services/src/reading_transfer/matching.rs @@ -0,0 +1,746 @@ +//! Resolving an export document's series and books against the current +//! library. +//! +//! Every step here follows the same rule: gather every candidate, keep only +//! the ones visible to the importing user, and accept the step's result only +//! when exactly one visible candidate remains. Multiple candidates are +//! reported as ambiguous rather than guessed, and a candidate the user cannot +//! see is simply not a candidate at all, so a series or book hidden behind a +//! sharing tag resolves as unmatched rather than as a permission error that +//! would confirm it exists. + +use anyhow::Result; +use sea_orm::{ColumnTrait, DatabaseConnection, EntityTrait, QueryFilter}; +use std::collections::HashSet; +use uuid::Uuid; + +use codex_db::entities::{books, series, series_external_ids}; +use codex_db::repositories::SeriesRepository; + +use crate::content_filter::ContentFilter; + +use super::model::{ExportBookDto, ExportSeriesDto, HashMode}; +use super::{file_stem, series_relative_book_path}; + +/// The outcome of resolving one exported series against the current library. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SeriesMatch { + Matched(Uuid), + Ambiguous, + Unmatched, +} + +/// The outcome of resolving one exported book against its matched series. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BookMatch { + Matched(Uuid), + StemMatch(Uuid), + Ambiguous, + Unmatched, + HashMismatch, +} + +/// A book in the current library, scoped to one series, as seen by the +/// matcher. Deliberately narrower than `books::Model`: the matcher only needs +/// enough to compare against the export. +#[derive(Debug, Clone)] +pub struct BookCandidate { + pub id: Uuid, + pub relative_path: String, + pub file_name: String, + pub file_hash: String, + pub partial_hash: String, +} + +impl BookCandidate { + pub fn from_model(model: &books::Model, library_path: &str, series_path: &str) -> Self { + Self { + id: model.id, + relative_path: series_relative_book_path(library_path, series_path, &model.path), + file_name: model.file_name.clone(), + file_hash: model.file_hash.clone(), + partial_hash: model.partial_hash.clone(), + } + } +} + +/// Keep only the ids visible to the user, deduplicated. Order does not +/// matter: callers only ever ask how many are left. +fn visible_ids(content_filter: &ContentFilter, ids: Vec) -> Vec { + let mut seen = HashSet::new(); + ids.into_iter() + .filter(|id| content_filter.is_series_visible(*id) && seen.insert(*id)) + .collect() +} + +/// Visible candidates that still have at least one book on disk. +/// +/// Splitting a library on the same instance leaves the old library's series +/// in place until someone deletes it, with every book soft-deleted by the +/// rescan that noticed the files had gone. That leftover series matches the +/// export exactly by path, and matching it would stop the search there with +/// no book to write onto, never reaching the new series. A series with nothing +/// on disk is not somewhere reading state can land, so it is not a candidate. +async fn live_candidates( + db: &DatabaseConnection, + content_filter: &ContentFilter, + ids: Vec, +) -> Result> { + let visible = visible_ids(content_filter, ids); + if visible.is_empty() { + return Ok(visible); + } + let live: HashSet = books::Entity::find() + .filter(books::Column::SeriesId.is_in(visible.clone())) + .filter(books::Column::Deleted.eq(false)) + .all(db) + .await? + .into_iter() + .map(|b| b.series_id) + .collect(); + Ok(visible.into_iter().filter(|id| live.contains(id)).collect()) +} + +async fn series_ids_by_external_id( + db: &DatabaseConnection, + source: &str, + external_id: &str, +) -> Result> { + let rows = series_external_ids::Entity::find() + .filter(series_external_ids::Column::Source.eq(source)) + .filter(series_external_ids::Column::ExternalId.eq(external_id)) + .all(db) + .await?; + Ok(rows.into_iter().map(|r| r.series_id).collect()) +} + +/// Match by `series.path` across every library, not just the one the export +/// came from: a library split re-roots series under a different library id, +/// so scoping this to one library would defeat the entire feature. +/// +/// Stored series paths use the server's own separator, while exports always +/// use `/`, so a Windows-stored path is compared in both spellings. +async fn series_ids_by_path(db: &DatabaseConnection, path: &str) -> Result> { + let spellings = vec![path.to_string(), path.replace('/', "\\")]; + let rows = series::Entity::find() + .filter(series::Column::Path.is_in(spellings)) + .all(db) + .await?; + Ok(rows.into_iter().map(|r| r.id).collect()) +} + +/// Match by `series.normalized_name` across every library, for the same +/// reason as [`series_ids_by_path`]. +async fn series_ids_by_normalized_name( + db: &DatabaseConnection, + normalized_name: &str, +) -> Result> { + let rows = series::Entity::find() + .filter(series::Column::NormalizedName.eq(normalized_name)) + .all(db) + .await?; + Ok(rows.into_iter().map(|r| r.id).collect()) +} + +/// Resolve an exported series against the current library. +/// +/// Tries, in order: each source in `source_preference` for which the export +/// carries an external id (skipping a source the export does not have, and +/// moving to the next preferred source when a tried one yields no visible +/// candidate), then `series.path`, then `series.normalized_name`. Stops at +/// the first step that produces any visible candidate. +pub async fn resolve_series( + db: &DatabaseConnection, + content_filter: &ContentFilter, + exported: &ExportSeriesDto, + source_preference: &[String], +) -> Result { + for source in source_preference { + let Some(external) = exported.external_ids.iter().find(|e| &e.source == source) else { + continue; + }; + + let candidates = series_ids_by_external_id(db, source, &external.id).await?; + match live_candidates(db, content_filter, candidates) + .await? + .as_slice() + { + [] => continue, + [only] => return Ok(SeriesMatch::Matched(*only)), + _ => return Ok(SeriesMatch::Ambiguous), + } + } + + let by_path = series_ids_by_path(db, &exported.library_relative_path).await?; + match live_candidates(db, content_filter, by_path) + .await? + .as_slice() + { + [] => {} + [only] => return Ok(SeriesMatch::Matched(*only)), + _ => return Ok(SeriesMatch::Ambiguous), + } + + let normalized = SeriesRepository::normalize_name(&exported.name); + let by_name = series_ids_by_normalized_name(db, &normalized).await?; + match live_candidates(db, content_filter, by_name) + .await? + .as_slice() + { + [] => Ok(SeriesMatch::Unmatched), + [only] => Ok(SeriesMatch::Matched(*only)), + _ => Ok(SeriesMatch::Ambiguous), + } +} + +/// One matching step's outcome: `None` means "no usable candidate, try the +/// next step"; `Some` means the search is over, one way or another. +fn decide_step( + matches: Vec<&BookCandidate>, + exported: &ExportBookDto, + hash_mode: HashMode, + is_stem_step: bool, +) -> Option { + match matches.len() { + 0 => None, + 1 => { + let candidate = matches[0]; + // Never on the stem step: it exists for a `.cbr` repacked to + // `.cbz`, and a repack always changes the hash, so checking there + // would turn every real repack into a mismatch. + let hashes_disagree = !is_stem_step + && hash_mode != HashMode::Off + && !exported.file_hash.is_empty() + && !candidate.file_hash.is_empty() + && exported.file_hash != candidate.file_hash; + if hashes_disagree { + Some(BookMatch::HashMismatch) + } else if is_stem_step { + Some(BookMatch::StemMatch(candidate.id)) + } else { + Some(BookMatch::Matched(candidate.id)) + } + } + _ => Some(BookMatch::Ambiguous), + } +} + +/// Resolve an exported book against the books already known to belong to its +/// matched series. +/// +/// Tries, in order: series-relative path, file name, filename stem (survives +/// a `.cbr` repacked to `.cbz`), and finally, only under `hash_mode = match`, +/// `file_hash` / `partial_hash` (rescues a bulk rename). Empty hash values +/// are never used as a matching key or a mismatch signal: the column is only +/// populated during analysis, so an unanalyzed book legitimately has `""`. +pub fn resolve_book( + candidates: &[BookCandidate], + exported: &ExportBookDto, + hash_mode: HashMode, +) -> BookMatch { + let by_path: Vec<&BookCandidate> = candidates + .iter() + .filter(|c| c.relative_path == exported.path) + .collect(); + if let Some(result) = decide_step(by_path, exported, hash_mode, false) { + return result; + } + + let by_name: Vec<&BookCandidate> = candidates + .iter() + .filter(|c| c.file_name == exported.file_name) + .collect(); + if let Some(result) = decide_step(by_name, exported, hash_mode, false) { + return result; + } + + let stem = file_stem(&exported.file_name); + let by_stem: Vec<&BookCandidate> = candidates + .iter() + .filter(|c| file_stem(&c.file_name) == stem) + .collect(); + if let Some(result) = decide_step(by_stem, exported, hash_mode, true) { + return result; + } + + if hash_mode == HashMode::Match { + let hash_usable = !exported.file_hash.is_empty(); + let partial_usable = !exported.partial_hash.is_empty(); + let by_hash: Vec<&BookCandidate> = candidates + .iter() + .filter(|c| { + (hash_usable && !c.file_hash.is_empty() && c.file_hash == exported.file_hash) + || (partial_usable + && !c.partial_hash.is_empty() + && c.partial_hash == exported.partial_hash) + }) + .collect(); + match by_hash.len() { + 0 => {} + 1 => return BookMatch::Matched(by_hash[0].id), + _ => return BookMatch::Ambiguous, + } + } + + BookMatch::Unmatched +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::content_filter::ContentFilter; + + fn candidate( + id: Uuid, + relative_path: &str, + file_name: &str, + file_hash: &str, + partial_hash: &str, + ) -> BookCandidate { + BookCandidate { + id, + relative_path: relative_path.to_string(), + file_name: file_name.to_string(), + file_hash: file_hash.to_string(), + partial_hash: partial_hash.to_string(), + } + } + + fn exported_book( + path: &str, + file_name: &str, + file_hash: &str, + partial_hash: &str, + ) -> ExportBookDto { + ExportBookDto { + path: path.to_string(), + file_name: file_name.to_string(), + file_hash: file_hash.to_string(), + partial_hash: partial_hash.to_string(), + progress: None, + completions: vec![], + sessions: None, + } + } + + #[test] + fn matches_by_exact_relative_path() { + let id = Uuid::new_v4(); + let candidates = vec![candidate(id, "Vol 01/v01.cbz", "v01.cbz", "", "")]; + let exported = exported_book("Vol 01/v01.cbz", "v01.cbz", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Matched(id) + ); + } + + #[test] + fn falls_back_to_file_name_when_path_moved() { + let id = Uuid::new_v4(); + // The volume folder was flattened away, but the file kept its name. + let candidates = vec![candidate(id, "v01.cbz", "v01.cbz", "", "")]; + let exported = exported_book("Vol 01/v01.cbz", "v01.cbz", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Matched(id) + ); + } + + #[test] + fn falls_back_to_stem_for_a_repack() { + let id = Uuid::new_v4(); + let candidates = vec![candidate(id, "v01.cbz", "v01.cbz", "", "")]; + let exported = exported_book("v01.cbr", "v01.cbr", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::StemMatch(id) + ); + } + + #[test] + fn two_files_sharing_a_stem_are_ambiguous() { + let candidates = vec![ + candidate(Uuid::new_v4(), "v01.cbz", "v01.cbz", "", ""), + candidate(Uuid::new_v4(), "v01.cbr", "v01.cbr", "", ""), + ]; + let exported = exported_book("elsewhere/v01.epub", "v01.epub", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Ambiguous + ); + } + + #[test] + fn verify_mode_rejects_a_hash_disagreement() { + let id = Uuid::new_v4(); + let candidates = vec![candidate(id, "v01.cbz", "v01.cbz", "hash-b", "")]; + let exported = exported_book("v01.cbz", "v01.cbz", "hash-a", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::HashMismatch + ); + } + + #[test] + fn off_mode_ignores_a_hash_disagreement() { + let id = Uuid::new_v4(); + let candidates = vec![candidate(id, "v01.cbz", "v01.cbz", "hash-b", "")]; + let exported = exported_book("v01.cbz", "v01.cbz", "hash-a", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Off), + BookMatch::Matched(id) + ); + } + + #[test] + fn empty_hashes_never_trigger_a_mismatch() { + let id = Uuid::new_v4(); + // Neither side has been analyzed. + let candidates = vec![candidate(id, "v01.cbz", "v01.cbz", "", "")]; + let exported = exported_book("v01.cbz", "v01.cbz", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Matched(id) + ); + } + + #[test] + fn match_mode_rescues_a_bulk_rename_by_hash() { + let id = Uuid::new_v4(); + let candidates = vec![candidate( + id, + "renamed/totally-different.cbz", + "totally-different.cbz", + "hash-a", + "", + )]; + let exported = exported_book("v01.cbz", "v01.cbz", "hash-a", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Match), + BookMatch::Matched(id) + ); + } + + #[test] + fn verify_mode_does_not_use_hash_as_a_matching_key() { + // Same scenario as the rescue above, but under `verify`: path and + // name both fail, and hash matching is a `match`-mode-only step. + let id = Uuid::new_v4(); + let candidates = vec![candidate( + id, + "renamed/totally-different.cbz", + "totally-different.cbz", + "hash-a", + "", + )]; + let exported = exported_book("v01.cbz", "v01.cbz", "hash-a", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Unmatched + ); + } + + #[test] + fn no_candidates_is_unmatched() { + let exported = exported_book("v01.cbz", "v01.cbz", "", ""); + assert_eq!( + resolve_book(&[], &exported, HashMode::Verify), + BookMatch::Unmatched + ); + } + + #[test] + fn ambiguous_by_path_does_not_fall_through_to_name() { + // Two candidates share the exact relative path (should not happen, + // but must not be resolved by silently trying the next step). + let candidates = vec![ + candidate(Uuid::new_v4(), "v01.cbz", "a.cbz", "", ""), + candidate(Uuid::new_v4(), "v01.cbz", "b.cbz", "", ""), + ]; + let exported = exported_book("v01.cbz", "a.cbz", "", ""); + assert_eq!( + resolve_book(&candidates, &exported, HashMode::Verify), + BookMatch::Ambiguous + ); + } + + // ------------------------------------------------------------------ + // Series resolution: exercises the DB queries and the visibility gate. + // ------------------------------------------------------------------ + + use codex_db::ScanningStrategy; + use codex_db::entities::user_sharing_tags::AccessMode; + use codex_db::repositories::{ + LibraryRepository, SeriesExternalIdRepository, SeriesRepository, SharingTagRepository, + UserRepository, + }; + use codex_db::test_helpers::create_test_db; + + async fn make_user(db: &DatabaseConnection) -> Uuid { + use chrono::Utc; + use codex_db::entities::users; + let now = Utc::now(); + let model = users::Model { + id: Uuid::new_v4(), + username: format!("u-{}", Uuid::new_v4()), + email: format!("{}@test.com", Uuid::new_v4()), + password_hash: "hash".to_string(), + role: "reader".to_string(), + is_active: true, + email_verified: true, + permissions: serde_json::json!([]), + created_at: now, + updated_at: now, + last_login_at: None, + }; + UserRepository::create(db, &model).await.unwrap().id + } + + fn doc_series(name: &str, path: &str, external: Option<(&str, &str)>) -> ExportSeriesDto { + ExportSeriesDto { + external_ids: external + .map(|(source, id)| { + vec![super::super::model::ExportExternalIdDto { + source: source.to_string(), + id: id.to_string(), + }] + }) + .unwrap_or_default(), + library_relative_path: path.to_string(), + name: name.to_string(), + rating: None, + notes: None, + rating_updated_at: None, + books: vec![], + } + } + + /// A series with one book on disk: a series with nothing on disk is not a + /// place reading state can land, so the matcher ignores it. + async fn live_series( + conn: &DatabaseConnection, + library_id: Uuid, + name: &str, + ) -> codex_db::entities::series::Model { + let series = SeriesRepository::create(conn, library_id, name, None) + .await + .unwrap(); + add_book(conn, &series, false).await; + series + } + + async fn add_book( + conn: &DatabaseConnection, + series: &codex_db::entities::series::Model, + deleted: bool, + ) { + use codex_db::repositories::BookRepository; + use sea_orm::{ActiveModelTrait, Set}; + let now = chrono::Utc::now(); + let book = books::Model { + id: Uuid::new_v4(), + series_id: series.id, + library_id: series.library_id, + path: format!("/lib/{}.cbz", Uuid::new_v4()), + file_name: "b.cbz".to_string(), + file_size: 1, + file_hash: String::new(), + partial_hash: String::new(), + format: "cbz".to_string(), + page_count: 1, + deleted: false, + analyzed: false, + analysis_error: None, + analysis_errors: None, + modified_at: now, + created_at: now, + updated_at: now, + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + }; + let book = BookRepository::create(conn, &book, None).await.unwrap(); + if deleted { + let mut gone: books::ActiveModel = book.into(); + gone.deleted = Set(true); + gone.update(conn).await.unwrap(); + } + } + + /// The same-instance split: the old library's series is still there with + /// every book soft-deleted, and matches the export exactly by path. It + /// must not stop the search, or the new series is never reached. + #[tokio::test] + async fn a_series_with_nothing_on_disk_is_not_a_candidate() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let old_library = + LibraryRepository::create(conn, "Old", "/manga", ScanningStrategy::Default) + .await + .unwrap(); + let new_library = + LibraryRepository::create(conn, "New", "/shonen", ScanningStrategy::Default) + .await + .unwrap(); + let leftover = SeriesRepository::create(conn, old_library.id, "Naruto", None) + .await + .unwrap(); + SeriesRepository::update_path(conn, leftover.id, "shonen/Naruto".to_string()) + .await + .unwrap(); + add_book(conn, &leftover, true).await; + let moved = live_series(conn, new_library.id, "Naruto").await; + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + let exported = doc_series("Naruto", "shonen/Naruto", None); + let result = resolve_series(conn, &filter, &exported, &[]).await.unwrap(); + assert_eq!(result, SeriesMatch::Matched(moved.id)); + } + + #[tokio::test] + async fn resolves_by_path_when_no_external_id_matches() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = live_series(conn, library.id, "Naruto").await; + SeriesRepository::update_path(conn, series.id, "shonen/Naruto".to_string()) + .await + .unwrap(); + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + let doc = doc_series("Naruto", "shonen/Naruto", None); + let result = resolve_series(conn, &filter, &doc, &[]).await.unwrap(); + assert_eq!(result, SeriesMatch::Matched(series.id)); + } + + #[tokio::test] + async fn falls_back_to_normalized_name() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = live_series(conn, library.id, "One Piece").await; + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + // A path that does not exist anywhere; only the name matches. + let doc = doc_series("One Piece", "moved/somewhere/else", None); + let result = resolve_series(conn, &filter, &doc, &[]).await.unwrap(); + assert_eq!(result, SeriesMatch::Matched(series.id)); + } + + #[tokio::test] + async fn preference_order_decides_the_winner() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series_a = live_series(conn, library.id, "Series A").await; + let series_b = live_series(conn, library.id, "Series B").await; + SeriesExternalIdRepository::create( + conn, + series_a.id, + "plugin:mangabaka", + "111", + None, + None, + ) + .await + .unwrap(); + SeriesExternalIdRepository::create(conn, series_b.id, "plugin:anilist", "222", None, None) + .await + .unwrap(); + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + let mut doc = doc_series("Whatever", "does/not/exist", None); + doc.external_ids = vec![ + super::super::model::ExportExternalIdDto { + source: "plugin:mangabaka".to_string(), + id: "111".to_string(), + }, + super::super::model::ExportExternalIdDto { + source: "plugin:anilist".to_string(), + id: "222".to_string(), + }, + ]; + + let winner_a = resolve_series( + conn, + &filter, + &doc, + &["plugin:mangabaka".to_string(), "plugin:anilist".to_string()], + ) + .await + .unwrap(); + assert_eq!(winner_a, SeriesMatch::Matched(series_a.id)); + + let winner_b = resolve_series( + conn, + &filter, + &doc, + &["plugin:anilist".to_string(), "plugin:mangabaka".to_string()], + ) + .await + .unwrap(); + assert_eq!(winner_b, SeriesMatch::Matched(series_b.id)); + } + + #[tokio::test] + async fn multiple_candidates_are_ambiguous() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + // Two libraries can each contain a series with the same normalized + // name after a split; nothing may guess between them. + let library2 = LibraryRepository::create(conn, "Lib2", "/lib2", ScanningStrategy::Default) + .await + .unwrap(); + live_series(conn, library.id, "Duplicate").await; + live_series(conn, library2.id, "Duplicate").await; + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + let doc = doc_series("Duplicate", "nowhere/matching", None); + let result = resolve_series(conn, &filter, &doc, &[]).await.unwrap(); + assert_eq!(result, SeriesMatch::Ambiguous); + } + + #[tokio::test] + async fn a_series_the_user_cannot_see_resolves_as_unmatched() { + let (db, _tmp) = create_test_db().await; + let conn = db.sea_orm_connection(); + let user = make_user(conn).await; + let library = LibraryRepository::create(conn, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = live_series(conn, library.id, "Hidden").await; + + let tag = SharingTagRepository::create(conn, "restricted", None) + .await + .unwrap(); + SharingTagRepository::add_tag_to_series(conn, series.id, tag.id) + .await + .unwrap(); + // A personal deny grant is enough on its own: deny always wins, + // regardless of whitelist mode. + SharingTagRepository::set_user_grant(conn, user, tag.id, AccessMode::Deny) + .await + .unwrap(); + + let filter = ContentFilter::for_user(conn, user).await.unwrap(); + let doc = doc_series("Hidden", &series.path, None); + let result = resolve_series(conn, &filter, &doc, &[]).await.unwrap(); + assert_eq!(result, SeriesMatch::Unmatched); + } +} diff --git a/crates/codex-services/src/reading_transfer/mod.rs b/crates/codex-services/src/reading_transfer/mod.rs new file mode 100644 index 000000000..ec9c5ff24 --- /dev/null +++ b/crates/codex-services/src/reading_transfer/mod.rs @@ -0,0 +1,126 @@ +//! Export and import of one user's reading state across a library +//! reorganisation or an instance move. +//! +//! Every piece of state this feature touches (`read_progress`, +//! `read_completions`, `reading_sessions`, `user_series_ratings`) is keyed on +//! `books.id` / `series.id`, and both are minted fresh whenever a library is +//! rescanned under a new root. Export serialises the state against stable +//! identifiers (external ids, relative paths, file names, hashes) instead; +//! import resolves those back to real ids in the current library and writes +//! the state under an explicit conflict policy. +//! +//! Submodules: +//! - [`export`] assembles the document from the four tables. +//! - [`matching`] resolves a document's series and books against the current +//! library, never guessing: any step with more than one candidate is +//! reported as ambiguous rather than picking one. +//! - [`import`] applies a matched document, one transaction per series, and +//! can run as a dry run that writes nothing but produces the same report. + +pub mod export; +pub mod import; +pub mod matching; +pub mod model; + +use std::path::Path; + +/// Compute a book's path relative to its series folder. +/// +/// `books.path` is absolute; `series.path` is relative to the library root. +/// Stripping both prefixes is what makes the result stable across a library +/// re-root: the series folder can move to a different library root entirely +/// and this value does not change, which is the whole point of exporting it +/// instead of the absolute path. +/// +/// Falls back to the book's file name when the book path is not inside the +/// library folder its row names. That happens for a soft-deleted book after +/// its library's root was changed, and exporting the absolute path instead +/// would publish a server filesystem path that can never match anyway; the +/// file-name step still can. +pub fn series_relative_book_path(library_path: &str, series_path: &str, book_path: &str) -> String { + let Ok(library_relative) = Path::new(book_path).strip_prefix(library_path) else { + return Path::new(book_path) + .file_name() + .map(|name| name.to_string_lossy().into_owned()) + .unwrap_or_else(|| book_path.to_string()); + }; + + let series_relative = if series_path.is_empty() { + library_relative + } else { + library_relative + .strip_prefix(series_path) + .unwrap_or(library_relative) + }; + + // Normalise to forward slashes so the exported path is stable regardless + // of the platform the server runs on. + series_relative + .components() + .map(|c| c.as_os_str().to_string_lossy().into_owned()) + .collect::>() + .join("/") +} + +/// The filename stem used for the "survives a `.cbr` repacked to `.cbz`" +/// matching step: everything before the last `.`, or the whole name when +/// there is no extension. +pub fn file_stem(file_name: &str) -> &str { + match file_name.rsplit_once('.') { + Some((stem, _ext)) if !stem.is_empty() => stem, + _ => file_name, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn strips_library_and_series_prefix() { + assert_eq!( + series_relative_book_path( + "/library/root", + "shonen/Naruto", + "/library/root/shonen/Naruto/Vol 01/v01.cbz" + ), + "Vol 01/v01.cbz" + ); + } + + #[test] + fn handles_series_at_library_root() { + assert_eq!( + series_relative_book_path("/library/root", "", "/library/root/v01.cbz"), + "v01.cbz" + ); + } + + #[test] + fn falls_back_to_the_file_name_when_prefixes_do_not_match() { + // A soft-deleted book under a library whose root has since changed. + // Never an absolute server path in the file. + assert_eq!( + series_relative_book_path("/other/root", "shonen/Naruto", "/library/root/v01.cbz"), + "v01.cbz" + ); + } + + #[test] + fn stem_strips_extension() { + assert_eq!(file_stem("v01.cbz"), "v01"); + assert_eq!(file_stem("v01.cbr"), "v01"); + } + + #[test] + fn stem_of_extensionless_name_is_the_name() { + assert_eq!(file_stem("README"), "README"); + } + + #[test] + fn stem_of_dotfile_is_the_whole_name() { + // rsplit_once yields an empty stem for a leading dot; that is not a + // usable stem, so the whole name is kept instead. + assert_eq!(file_stem(".gitignore"), ".gitignore"); + } +} diff --git a/crates/codex-services/src/reading_transfer/model.rs b/crates/codex-services/src/reading_transfer/model.rs new file mode 100644 index 000000000..6fcc4ca52 --- /dev/null +++ b/crates/codex-services/src/reading_transfer/model.rs @@ -0,0 +1,345 @@ +//! The reading-progress export/import document and the import request and +//! report shapes. +//! +//! These types are the wire format for `GET /api/v1/reading-progress/export` +//! and `POST /api/v1/reading-progress/import`, defined here (rather than in +//! `codex-api`'s DTO module, where every other request/response type lives) +//! because [`export`](super::export) produces them and +//! [`import`](super::import) both consumes and produces them directly; the +//! API layer only serializes what the service already built. `codex-api` +//! re-exports this module under its `dto` namespace for OpenAPI registration. +//! +//! The document itself is a portable file meant to survive a library +//! reorganisation or a move to a different Codex instance, so its field names +//! are deliberately **not** camelCased like the rest of the API's DTOs: it is +//! a versioned interchange format rather than a shape a frontend deserializes, +//! and plain `snake_case` matches what anyone writing a compatible tool would +//! expect from the documented shape. + +use chrono::{DateTime, Utc}; +use serde::{Deserialize, Serialize}; +use utoipa::ToSchema; +use uuid::Uuid; + +/// The only document shape this server currently writes, and the highest +/// `version` it knows how to read. +pub const READING_PROGRESS_FORMAT: &str = "codex-reading-progress"; +pub const READING_PROGRESS_VERSION: i32 = 1; + +/// One external identifier attached to a series (a plugin match, a ComicInfo +/// value, or a manual entry). +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportExternalIdDto { + /// `plugin:`, `comicinfo`, `epub`, or `manual`. + #[schema(example = "plugin:mangabaka")] + pub source: String, + #[schema(example = "12345")] + pub id: String, +} + +/// The live resume position for one book. Retains `r2_progression`: it is the +/// only place the EPUB locator survives, since sessions strip it. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportProgressDto { + pub current_page: i32, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub progress_percentage: Option, + pub completed: bool, + pub started_at: DateTime, + pub updated_at: DateTime, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub completed_at: Option>, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub r2_progression: Option, +} + +/// One finished read-through. Keeps its original id so re-importing the same +/// file is a no-op rather than a duplicate. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportCompletionDto { + pub id: Uuid, + pub started_at: DateTime, + pub completed_at: DateTime, +} + +/// One row from the reading-session log. `r2_progression` is deliberately +/// absent: nothing reads a session's historical locator, and it is the only +/// non-scalar column on the row. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportSessionDto { + pub id: Uuid, + pub device_id: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub device_name: Option, + pub pass: i32, + /// `"progress"`, `"completed"`, or `"reset"`. + #[schema(example = "progress")] + pub kind: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub to_page: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub to_percentage: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub active_duration_ms: Option, + /// `"measured"`, `"inferred"`, or `"unknown"`. + #[schema(example = "measured")] + pub duration_source: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub pages_read: Option, + pub client_started_at: DateTime, + pub client_ended_at: DateTime, + pub server_recorded_at: DateTime, +} + +/// One book inside a series, keyed for matching by path, name, and hash +/// rather than by id: the whole point of the file is that ids on the far side +/// are expected to be different. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportBookDto { + /// Relative to the series folder, so a series move does not invalidate it. + #[schema(example = "Vol 01/v01.cbz")] + pub path: String, + #[schema(example = "v01.cbz")] + pub file_name: String, + /// Empty when the book was never analyzed; never treated as a value to + /// match on in that case. + #[serde(default)] + pub file_hash: String, + #[serde(default)] + pub partial_hash: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub progress: Option, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub completions: Vec, + /// Omitted entirely (not an empty array) when the export was taken with + /// `include_sessions=false`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub sessions: Option>, +} + +/// One series and everything the exporting user recorded against its books. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ExportSeriesDto { + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub external_ids: Vec, + /// The series path as stored, relative to the library root. + #[schema(example = "shonen/Naruto")] + pub library_relative_path: String, + #[schema(example = "Naruto")] + pub name: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub rating: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub notes: Option, + /// When the rating was last changed. The `newest` conflict policy needs it + /// to tell a stale rating from a fresh one; a file without it never + /// overwrites an existing rating except under `overwrite`. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub rating_updated_at: Option>, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub books: Vec, +} + +/// The whole export: one user's reading state, self-describing enough to be +/// matched back against a differently organised library. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ReadingProgressExportDocument { + #[schema(example = "codex-reading-progress")] + pub format: String, + #[schema(example = 1)] + pub version: i32, + pub exported_at: DateTime, + pub includes_sessions: bool, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub series: Vec, +} + +/// How aggressively hashes are used to match a book. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, ToSchema, Default)] +#[serde(rename_all = "snake_case")] +pub enum HashMode { + /// Hashes are ignored entirely. + Off, + /// Match by path/name, then reject a pair whose `file_hash` disagrees. + #[default] + Verify, + /// Additionally use `file_hash` / `partial_hash` as a matching key when + /// the path and name steps find nothing, which rescues a bulk rename. + Match, +} + +/// How a conflict between an imported value and an existing row is resolved. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, ToSchema, Default)] +#[serde(rename_all = "snake_case")] +pub enum ConflictPolicy { + /// Whichever side has the later `updated_at` wins. + #[default] + Newest, + /// Whichever side is further into the book wins; a completed read beats + /// any partial one. + Furthest, + /// An existing row is left untouched; only a missing row is written. + SkipExisting, + /// The imported value always wins. + Overwrite, +} + +fn default_reattach_sessions() -> bool { + true +} + +/// `POST /api/v1/reading-progress/import` request body. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ImportReadingProgressRequest { + /// Compute and report the outcome without writing anything. + #[serde(default)] + pub dry_run: bool, + #[serde(default)] + pub hash_mode: HashMode, + /// External-id sources to try, in order, before falling back to path and + /// then normalized name. An empty list skips straight to path matching. + #[serde(default)] + pub source_preference: Vec, + #[serde(default)] + pub conflict_policy: ConflictPolicy, + /// When a session or completion in the file already exists as the + /// importer's own row but is not on a live book (its book was hard-deleted, + /// leaving `book_id` null, or the scanner marked it deleted after the file + /// moved), move it onto the matched book instead of skipping it. + #[serde(default = "default_reattach_sessions")] + pub reattach_sessions: bool, + /// A stem match (`v01.cbr` renamed to `v01.cbz`) is reported either way, + /// but only written when this is set: two files can share a stem, and + /// applying it silently risks writing progress onto the wrong one. + #[serde(default)] + pub accept_stem_matches: bool, + pub file: ReadingProgressExportDocument, +} + +/// Why a series in the import file could not be resolved to exactly one +/// series in the current library. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, ToSchema)] +#[serde(rename_all = "snake_case")] +pub enum SeriesDisposition { + Matched, + /// More than one series matched at the same step. Never guessed. + Ambiguous, + /// No series matched, or the only ones that did are not visible to the + /// importing user. + Unmatched, +} + +/// Why a book in the import file could not be resolved to exactly one book +/// in its matched series. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, ToSchema)] +#[serde(rename_all = "snake_case")] +pub enum BookDisposition { + Matched, + /// Matched only by filename stem (e.g. a `.cbr` repacked to `.cbz`). + /// Applied only when `accept_stem_matches` is set. + StemMatch, + Ambiguous, + Unmatched, + /// A single candidate was found by path or name, but its `file_hash` + /// disagreed with the export under `hash_mode = "verify"` or `"match"`. + HashMismatch, +} + +/// What happened to one scalar field write (progress or rating) under the +/// active conflict policy. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, ToSchema)] +#[serde(rename_all = "snake_case")] +pub enum FieldOutcome { + Inserted, + Updated, + /// An existing row won the conflict policy and was left as-is. + Skipped, +} + +/// Insert/reattach/skip counts for an append-only table (completions or +/// sessions) within one book. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize, ToSchema)] +pub struct WriteCounts { + pub inserted: u32, + /// Adopted an orphaned row (`book_id IS NULL`) rather than inserting a + /// new one. + pub reattached: u32, + /// Already present with the same book attached; re-importing is a no-op. + pub skipped: u32, +} + +/// The outcome for one book in the import file. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ImportBookReport { + pub path: String, + pub file_name: String, + pub disposition: BookDisposition, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub matched_book_id: Option, + /// Whether any writes were attempted for this book. False for every + /// disposition except `matched`, and except `stem_match` when + /// `accept_stem_matches` is off. + pub applied: bool, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub progress: Option, + pub completions: WriteCounts, + pub sessions: WriteCounts, +} + +/// The outcome for one series in the import file. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ImportSeriesReport { + pub library_relative_path: String, + pub name: String, + pub disposition: SeriesDisposition, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub matched_series_id: Option, + /// Whether book/rating processing was attempted at all (only when + /// `disposition == matched`). + pub attempted: bool, + /// Whether this series' transaction was committed. Always `false` in a + /// dry run, and `false` if `attempted` but a write failed. + pub committed: bool, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub error: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub rating: Option, + pub books: Vec, +} + +/// Totals across every series in the file, for a one-line summary. +#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, ToSchema)] +pub struct ImportSummary { + pub series_total: u32, + pub series_matched: u32, + pub series_ambiguous: u32, + pub series_unmatched: u32, + pub series_committed: u32, + pub books_total: u32, + pub books_matched: u32, + pub books_stem_matched: u32, + pub books_ambiguous: u32, + pub books_unmatched: u32, + pub books_hash_mismatch: u32, + pub progress_written: u32, + pub ratings_written: u32, + pub completions_inserted: u32, + pub completions_reattached: u32, + pub sessions_inserted: u32, + pub sessions_reattached: u32, +} + +/// The response for both a real import and a dry run: the shape is identical +/// either way, so a client cannot tell from the response alone whether +/// anything was written. Only `dry_run` (and the DB) says that. +#[derive(Debug, Clone, Serialize, Deserialize, ToSchema)] +pub struct ImportReadingProgressResponse { + pub dry_run: bool, + pub sessions_in_file: bool, + /// Informational notes about the request, e.g. `reattach_sessions` having + /// no effect because the file carries no sessions. + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub notices: Vec, + pub summary: ImportSummary, + pub series: Vec, +} diff --git a/docs/api/openapi.json b/docs/api/openapi.json index 157908ff7..d23780206 100644 --- a/docs/api/openapi.json +++ b/docs/api/openapi.json @@ -10237,6 +10237,105 @@ ] } }, + "/api/v1/reading-progress/export": { + "get": { + "tags": [ + "Reading Progress Transfer" + ], + "summary": "Export the authenticated user's reading progress", + "description": "Produces the whole `read_progress` / `read_completions` / `reading_sessions`\n/ `user_series_ratings` state for the caller, keyed by external ids and\nseries-relative paths instead of database ids so it can be matched back\nagainst a differently organised library (or a different Codex instance\nentirely) by `POST /api/v1/reading-progress/import`.\n\nTake this export **before** deleting an old library during a split: the\nunderlying rows survive a hard delete as orphans, but nothing except this\nfile can say which book an orphan used to belong to.", + "operationId": "export_reading_progress", + "parameters": [ + { + "name": "include_sessions", + "in": "query", + "description": "Include the reading-session log (default: true). Sessions are the only source of every reading statistic, so this is opt-out rather than opt-in.", + "required": false, + "schema": { + "type": "boolean" + } + } + ], + "responses": { + "200": { + "description": "The export document", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ReadingProgressExportDocument" + } + } + } + }, + "401": { + "description": "Unauthorized" + }, + "403": { + "description": "Forbidden" + } + }, + "security": [ + { + "jwt_bearer": [] + }, + { + "api_key": [] + } + ] + } + }, + "/api/v1/reading-progress/import": { + "post": { + "tags": [ + "Reading Progress Transfer" + ], + "summary": "Import reading progress from an export document", + "description": "Resolves the file's series and books against the current library through\nthe same access-group / sharing-tag visibility that an ordinary read\nrespects: a book the importing user cannot see resolves as `unmatched`,\nnever as a permission error that would confirm it exists, and nothing is\never written against it.\n\nMatching never guesses: any step (external id, path, file name, or, under\n`hash_mode = \"match\"`, hash) that finds more than one candidate reports\n`ambiguous` and writes nothing for that series or book. `dry_run: true`\nreturns the identical response shape without writing anything, which is\nwhat makes it safe to preview before committing.\n\nEach series is applied in its own transaction; the response's per-series\n`committed` field says which ones actually landed. `read_completions` and\n`reading_sessions` reuse their exported ids on insert, so importing the\nsame file twice leaves row counts unchanged rather than duplicating\nhistory.", + "operationId": "import_reading_progress", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ImportReadingProgressRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Import processed (or, for a dry run, previewed)", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ImportReadingProgressResponse" + } + } + } + }, + "400": { + "description": "Unknown export format, a version newer than this server supports, or a value the normal write paths reject" + }, + "401": { + "description": "Unauthorized" + }, + "403": { + "description": "Forbidden" + }, + "413": { + "description": "The file is larger than the import limit" + } + }, + "security": [ + { + "jwt_bearer": [] + }, + { + "api_key": [] + } + ] + } + }, "/api/v1/reading-sessions": { "post": { "tags": [ @@ -10474,7 +10573,7 @@ "Reading Statistics" ], "summary": "Delete the caller's reading history for books that no longer exist", - "description": "When a book is deleted from the server its reading sessions and finished\nread-throughs are kept, so the time still counts towards every statistic;\nthe series and format breakdowns show it as one \"removed from library\" row.\nThis discards those rows for the caller, and only for the caller.\n\nIrreversible. Only history already detached from any book is touched;\nattributed reading is never affected.", + "description": "When a book is deleted from the server its reading sessions and finished\nread-throughs are kept, so the time still counts towards every statistic;\nthe series and format breakdowns show it as one \"removed from library\" row.\nThis discards those rows for the caller, and only for the caller.\n\nIrreversible, and it forecloses the other way out: importing a reading\nprogress export taken before the delete puts those sessions back on their\nbooks once the files are scanned again. Only history already detached from\nany book is touched; attributed reading is never affected.", "operationId": "purge_orphaned_reading_history", "responses": { "200": { @@ -24698,6 +24797,17 @@ } } }, + "BookDisposition": { + "type": "string", + "description": "Why a book in the import file could not be resolved to exactly one book\nin its matched series.", + "enum": [ + "matched", + "stem_match", + "ambiguous", + "unmatched", + "hash_mismatch" + ] + }, "BookDto": { "type": "object", "description": "Book data transfer object", @@ -28315,6 +28425,16 @@ } } }, + "ConflictPolicy": { + "type": "string", + "description": "How a conflict between an imported value and an existing row is resolved.", + "enum": [ + "newest", + "furthest", + "skip_existing", + "overwrite" + ] + }, "Contributor": { "type": "object", "description": "Contributor information (author, artist, etc.)", @@ -30538,6 +30658,93 @@ } } }, + "ExportBookDto": { + "type": "object", + "description": "One book inside a series, keyed for matching by path, name, and hash\nrather than by id: the whole point of the file is that ids on the far side\nare expected to be different.", + "required": [ + "path", + "file_name" + ], + "properties": { + "completions": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportCompletionDto" + } + }, + "file_hash": { + "type": "string", + "description": "Empty when the book was never analyzed; never treated as a value to\nmatch on in that case." + }, + "file_name": { + "type": "string", + "example": "v01.cbz" + }, + "partial_hash": { + "type": "string" + }, + "path": { + "type": "string", + "description": "Relative to the series folder, so a series move does not invalidate it.", + "example": "Vol 01/v01.cbz" + }, + "progress": { + "$ref": "#/components/schemas/ExportProgressDto" + }, + "sessions": { + "type": [ + "array", + "null" + ], + "items": { + "$ref": "#/components/schemas/ExportSessionDto" + }, + "description": "Omitted entirely (not an empty array) when the export was taken with\n`include_sessions=false`." + } + } + }, + "ExportCompletionDto": { + "type": "object", + "description": "One finished read-through. Keeps its original id so re-importing the same\nfile is a no-op rather than a duplicate.", + "required": [ + "id", + "started_at", + "completed_at" + ], + "properties": { + "completed_at": { + "type": "string", + "format": "date-time" + }, + "id": { + "type": "string", + "format": "uuid" + }, + "started_at": { + "type": "string", + "format": "date-time" + } + } + }, + "ExportExternalIdDto": { + "type": "object", + "description": "One external identifier attached to a series (a plugin match, a ComicInfo\nvalue, or a manual entry).", + "required": [ + "source", + "id" + ], + "properties": { + "id": { + "type": "string", + "example": "12345" + }, + "source": { + "type": "string", + "description": "`plugin:`, `comicinfo`, `epub`, or `manual`.", + "example": "plugin:mangabaka" + } + } + }, "ExportFieldCatalogResponse": { "type": "object", "description": "Response for the field catalog", @@ -30619,6 +30826,198 @@ } } }, + "ExportProgressDto": { + "type": "object", + "description": "The live resume position for one book. Retains `r2_progression`: it is the\nonly place the EPUB locator survives, since sessions strip it.", + "required": [ + "current_page", + "completed", + "started_at", + "updated_at" + ], + "properties": { + "completed": { + "type": "boolean" + }, + "completed_at": { + "type": [ + "string", + "null" + ], + "format": "date-time" + }, + "current_page": { + "type": "integer", + "format": "int32" + }, + "progress_percentage": { + "type": [ + "number", + "null" + ], + "format": "double" + }, + "r2_progression": { + "type": [ + "string", + "null" + ] + }, + "started_at": { + "type": "string", + "format": "date-time" + }, + "updated_at": { + "type": "string", + "format": "date-time" + } + } + }, + "ExportReadingProgressQuery": { + "type": "object", + "description": "Query parameters for `GET /api/v1/reading-progress/export`.", + "properties": { + "include_sessions": { + "type": "boolean", + "description": "Sessions are opt-out: they are the only source of every reading\nstatistic, so leaving them out is easy to do by accident and hard to\nnotice until the numbers are gone." + } + } + }, + "ExportSeriesDto": { + "type": "object", + "description": "One series and everything the exporting user recorded against its books.", + "required": [ + "library_relative_path", + "name" + ], + "properties": { + "books": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportBookDto" + } + }, + "external_ids": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportExternalIdDto" + } + }, + "library_relative_path": { + "type": "string", + "description": "The series path as stored, relative to the library root.", + "example": "shonen/Naruto" + }, + "name": { + "type": "string", + "example": "Naruto" + }, + "notes": { + "type": [ + "string", + "null" + ] + }, + "rating": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "rating_updated_at": { + "type": [ + "string", + "null" + ], + "format": "date-time", + "description": "When the rating was last changed. The `newest` conflict policy needs it\nto tell a stale rating from a fresh one; a file without it never\noverwrites an existing rating except under `overwrite`." + } + } + }, + "ExportSessionDto": { + "type": "object", + "description": "One row from the reading-session log. `r2_progression` is deliberately\nabsent: nothing reads a session's historical locator, and it is the only\nnon-scalar column on the row.", + "required": [ + "id", + "device_id", + "pass", + "kind", + "duration_source", + "client_started_at", + "client_ended_at", + "server_recorded_at" + ], + "properties": { + "active_duration_ms": { + "type": [ + "integer", + "null" + ], + "format": "int64" + }, + "client_ended_at": { + "type": "string", + "format": "date-time" + }, + "client_started_at": { + "type": "string", + "format": "date-time" + }, + "device_id": { + "type": "string" + }, + "device_name": { + "type": [ + "string", + "null" + ] + }, + "duration_source": { + "type": "string", + "description": "`\"measured\"`, `\"inferred\"`, or `\"unknown\"`.", + "example": "measured" + }, + "id": { + "type": "string", + "format": "uuid" + }, + "kind": { + "type": "string", + "description": "`\"progress\"`, `\"completed\"`, or `\"reset\"`.", + "example": "progress" + }, + "pages_read": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "pass": { + "type": "integer", + "format": "int32" + }, + "server_recorded_at": { + "type": "string", + "format": "date-time" + }, + "to_page": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "to_percentage": { + "type": [ + "number", + "null" + ], + "format": "double" + } + } + }, "ExternalIdContextDto": { "type": "object", "description": "External ID context for template evaluation.\n\nRepresents an external ID from a metadata provider (plugin, comicinfo, etc.)\nin a simplified format suitable for template access.", @@ -31111,6 +31510,15 @@ ], "description": "Operators for string and equality comparisons" }, + "FieldOutcome": { + "type": "string", + "description": "What happened to one scalar field write (progress or rating) under the\nactive conflict policy.", + "enum": [ + "inserted", + "updated", + "skipped" + ] + }, "FileSystemEntry": { "type": "object", "required": [ @@ -32045,6 +32453,15 @@ } } }, + "HashMode": { + "type": "string", + "description": "How aggressively hashes are used to match a book.", + "enum": [ + "off", + "verify", + "match" + ] + }, "ImageLink": { "type": "object", "description": "Image link with optional dimensions\n\nUsed for cover images and thumbnails in publications.", @@ -32079,6 +32496,283 @@ } } }, + "ImportBookReport": { + "type": "object", + "description": "The outcome for one book in the import file.", + "required": [ + "path", + "file_name", + "disposition", + "applied", + "completions", + "sessions" + ], + "properties": { + "applied": { + "type": "boolean", + "description": "Whether any writes were attempted for this book. False for every\ndisposition except `matched`, and except `stem_match` when\n`accept_stem_matches` is off." + }, + "completions": { + "$ref": "#/components/schemas/WriteCounts" + }, + "disposition": { + "$ref": "#/components/schemas/BookDisposition" + }, + "file_name": { + "type": "string" + }, + "matched_book_id": { + "type": [ + "string", + "null" + ], + "format": "uuid" + }, + "path": { + "type": "string" + }, + "progress": { + "$ref": "#/components/schemas/FieldOutcome" + }, + "sessions": { + "$ref": "#/components/schemas/WriteCounts" + } + } + }, + "ImportReadingProgressRequest": { + "type": "object", + "description": "`POST /api/v1/reading-progress/import` request body.", + "required": [ + "file" + ], + "properties": { + "accept_stem_matches": { + "type": "boolean", + "description": "A stem match (`v01.cbr` renamed to `v01.cbz`) is reported either way,\nbut only written when this is set: two files can share a stem, and\napplying it silently risks writing progress onto the wrong one." + }, + "conflict_policy": { + "$ref": "#/components/schemas/ConflictPolicy" + }, + "dry_run": { + "type": "boolean", + "description": "Compute and report the outcome without writing anything." + }, + "file": { + "$ref": "#/components/schemas/ReadingProgressExportDocument" + }, + "hash_mode": { + "$ref": "#/components/schemas/HashMode" + }, + "reattach_sessions": { + "type": "boolean", + "description": "When a session or completion in the file already exists as the\nimporter's own row but is not on a live book (its book was hard-deleted,\nleaving `book_id` null, or the scanner marked it deleted after the file\nmoved), move it onto the matched book instead of skipping it." + }, + "source_preference": { + "type": "array", + "items": { + "type": "string" + }, + "description": "External-id sources to try, in order, before falling back to path and\nthen normalized name. An empty list skips straight to path matching." + } + } + }, + "ImportReadingProgressResponse": { + "type": "object", + "description": "The response for both a real import and a dry run: the shape is identical\neither way, so a client cannot tell from the response alone whether\nanything was written. Only `dry_run` (and the DB) says that.", + "required": [ + "dry_run", + "sessions_in_file", + "summary", + "series" + ], + "properties": { + "dry_run": { + "type": "boolean" + }, + "notices": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Informational notes about the request, e.g. `reattach_sessions` having\nno effect because the file carries no sessions." + }, + "series": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ImportSeriesReport" + } + }, + "sessions_in_file": { + "type": "boolean" + }, + "summary": { + "$ref": "#/components/schemas/ImportSummary" + } + } + }, + "ImportSeriesReport": { + "type": "object", + "description": "The outcome for one series in the import file.", + "required": [ + "library_relative_path", + "name", + "disposition", + "attempted", + "committed", + "books" + ], + "properties": { + "attempted": { + "type": "boolean", + "description": "Whether book/rating processing was attempted at all (only when\n`disposition == matched`)." + }, + "books": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ImportBookReport" + } + }, + "committed": { + "type": "boolean", + "description": "Whether this series' transaction was committed. Always `false` in a\ndry run, and `false` if `attempted` but a write failed." + }, + "disposition": { + "$ref": "#/components/schemas/SeriesDisposition" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "library_relative_path": { + "type": "string" + }, + "matched_series_id": { + "type": [ + "string", + "null" + ], + "format": "uuid" + }, + "name": { + "type": "string" + }, + "rating": { + "$ref": "#/components/schemas/FieldOutcome" + } + } + }, + "ImportSummary": { + "type": "object", + "description": "Totals across every series in the file, for a one-line summary.", + "required": [ + "series_total", + "series_matched", + "series_ambiguous", + "series_unmatched", + "series_committed", + "books_total", + "books_matched", + "books_stem_matched", + "books_ambiguous", + "books_unmatched", + "books_hash_mismatch", + "progress_written", + "ratings_written", + "completions_inserted", + "completions_reattached", + "sessions_inserted", + "sessions_reattached" + ], + "properties": { + "books_ambiguous": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_hash_mismatch": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_stem_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_total": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_unmatched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "completions_inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "completions_reattached": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "progress_written": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "ratings_written": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_ambiguous": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_committed": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_total": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_unmatched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "sessions_inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "sessions_reattached": { + "type": "integer", + "format": "int32", + "minimum": 0 + } + } + }, "InheritedFrom": { "type": "string", "description": "Which layer supplied a value the user is inheriting.", @@ -39320,6 +40014,40 @@ } } }, + "ReadingProgressExportDocument": { + "type": "object", + "description": "The whole export: one user's reading state, self-describing enough to be\nmatched back against a differently organised library.", + "required": [ + "format", + "version", + "exported_at", + "includes_sessions" + ], + "properties": { + "exported_at": { + "type": "string", + "format": "date-time" + }, + "format": { + "type": "string", + "example": "codex-reading-progress" + }, + "includes_sessions": { + "type": "boolean" + }, + "series": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportSeriesDto" + } + }, + "version": { + "type": "integer", + "format": "int32", + "example": 1 + } + } + }, "ReadingSessionDto": { "type": "object", "description": "One reading session measured by a client.", @@ -41948,6 +42676,15 @@ } } }, + "SeriesDisposition": { + "type": "string", + "description": "Why a series in the import file could not be resolved to exactly one\nseries in the current library.", + "enum": [ + "matched", + "ambiguous", + "unmatched" + ] + }, "SeriesDto": { "type": "object", "description": "Series data transfer object", @@ -47381,6 +48118,34 @@ "description": "URL template with config placeholders already substituted. The only\nremaining placeholder is `{externalId}`, filled client-side." } } + }, + "WriteCounts": { + "type": "object", + "description": "Insert/reattach/skip counts for an append-only table (completions or\nsessions) within one book.", + "required": [ + "inserted", + "reattached", + "skipped" + ], + "properties": { + "inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "reattached": { + "type": "integer", + "format": "int32", + "description": "Adopted an orphaned row (`book_id IS NULL`) rather than inserting a\nnew one.", + "minimum": 0 + }, + "skipped": { + "type": "integer", + "format": "int32", + "description": "Already present with the same book attached; re-importing is a no-op.", + "minimum": 0 + } + } } }, "securitySchemes": { @@ -47474,6 +48239,10 @@ "name": "Reading Progress", "description": "Reading progress tracking" }, + { + "name": "Reading Progress Transfer", + "description": "Exporting and importing reading state across a library reorganisation or an instance move" + }, { "name": "Task Queue", "description": "Background job queue management" diff --git a/docs/docs/backup-migration/export-import-copy.md b/docs/docs/backup-migration/export-import-copy.md index 098c53c9d..b3a7eb1fb 100644 --- a/docs/docs/backup-migration/export-import-copy.md +++ b/docs/docs/backup-migration/export-import-copy.md @@ -21,9 +21,11 @@ PostgreSQL's native `uuid`/`jsonb`, so a byte-level copy would corrupt data. The `export`/`import`/`copy` commands translate these correctly. ::: -:::note Not the same as "Data Exports" +:::note Not the same as "Data Exports" or reading-progress transfer This is database-level backup/transfer. The user-facing [Data Exports](../exports) -feature (exporting a series to JSON/CSV) is unrelated. +feature (exporting a series to JSON/CSV) is unrelated, and so is +[carrying one user's reading progress across a library split](./reading-progress-transfer.md), +which moves far less data through the web API rather than the shell. ::: ## `export` diff --git a/docs/docs/backup-migration/reading-progress-transfer.md b/docs/docs/backup-migration/reading-progress-transfer.md new file mode 100644 index 000000000..361d6a655 --- /dev/null +++ b/docs/docs/backup-migration/reading-progress-transfer.md @@ -0,0 +1,135 @@ +--- +--- + +# Carrying Reading Progress Across a Library Split + +Reorganising a library, whether splitting one root into several, moving files +to a new path, or moving your whole collection to a different Codex instance, +mints new series and book ids. Nothing about your reading history follows +automatically: `read_progress`, `read_completions`, `reading_sessions`, and +your series ratings are all keyed on the ids the old scan created. + +`GET /api/v1/reading-progress/export` and `POST /api/v1/reading-progress/import` +exist to carry that state across the move. Export writes one JSON file for your +own reading state; import matches it back against whatever the library looks +like afterwards and writes the state onto the new ids. + +:::danger Export before you delete anything +Take the export **before** you delete the old library or root. Sessions and +completions survive a hard-deleted book as orphaned rows (their `book_id` is +cleared, not the row), so a failed or late import is a retry, not a loss; but +nothing except an export taken beforehand can say **which book** an orphan used +to belong to. Delete first, and that attribution is gone permanently, even +though the row itself still exists. +::: + +:::note Not the same as the instance-level backup +[`codex export` / `codex import`](./export-import-copy.md) move the *entire* +database between instances and require shell access to the server. This +feature moves *one user's own reading state* through the web API and is meant +for exactly this workflow: splitting or re-rooting a library without asking +every reader to re-mark their progress by hand. +::: + +## The workflow + +1. **Export.** From Settings → Reading Progress, download the export (or call + `GET /api/v1/reading-progress/export` directly). This is a snapshot of your + own `read_progress`, `read_completions`, `reading_sessions`, and series + ratings, keyed by external ids, each series' path, and each book's + filename and hash rather than by database id. +2. **Reorganise.** Split the library, move files, rescan under new roots, + move to a new instance, whatever the move is. Delete the old + library/root only after the export from step 1 is safely saved somewhere. + On the same instance you can import before or after deleting it: a series + whose files have all moved away is never chosen as a match, and history + still sitting on the old, soft-deleted books is moved onto the new ones. +3. **Import.** Upload the file from Settings → Reading Progress (or call + `POST /api/v1/reading-progress/import`). Run it once as a **dry run** first: + it returns the identical report shape without writing anything, so you + can check the match quality before committing. +4. **Apply.** Re-run with `dry_run: false` (the Settings page gates this + behind a successful preview). Progress, completions, sessions, and ratings + land on the new books and series. + +Importing the same file again later is safe: `read_completions` and +`reading_sessions` keep their original ids, so a repeat import is a no-op +rather than a duplicate, and `read_progress` / ratings are upserted under +whichever conflict policy you choose. + +## Matching + +Each series in the file is resolved against the current library, in order, +stopping at the first step that finds anything: + +1. An external id (a plugin match, ComicInfo, or a manual entry), tried in the + order given by `source_preference`. +2. The series' path, relative to its library root. +3. The series' normalized name. + +A series with no books left on disk (for example the old library's copy after +its files moved away) is never a candidate at any step. + +Books are then resolved within that series, in order: their path relative to +the series folder, their file name, and finally their filename stem (which +survives a `.cbr` repacked to `.cbz`). A repack always changes the file's +hash, so the stem step never checks hashes. Under `hash_mode: verify`, a path +or name match whose hash differs is reported as `hash_mismatch`; that includes +a file a tagging tool has rewritten since the export, so use `off` if you +re-tagged your collection between export and import. + +When two entries in the file land on the same book (a file that moved inside +its series appears once under its old path and once under its new one), the +second is decided against the first under the conflict policy, exactly as if +the first were already in the database. + +**Nothing is ever guessed.** If a step finds more than one candidate, that +series or book is reported as `ambiguous` and nothing is written for it. A +stem match is reported as `stem_match` and is only applied if you turn on +`accept_stem_matches`: two files can share a stem, so applying it silently +risks writing progress onto the wrong one. + +A series or book you cannot see (denied by a sharing tag, or outside your +access groups) resolves as `unmatched`, the same as one that genuinely is not +in the library. It never surfaces as a permission error, which would confirm +that the content exists. + +## Request flags + +| Flag | Default | Effect | +|---|---|---| +| `dry_run` | `false` | Report the outcome without writing anything | +| `hash_mode` | `verify` | `off` ignores hashes; `verify` rejects a path/name match whose `file_hash` disagrees; `match` additionally uses `file_hash`/`partial_hash` to find a book when path and name both fail (rescues a bulk rename) | +| `source_preference` | `[]` | External-id sources to try, in order, before falling back to path and name | +| `conflict_policy` | `newest` | How to resolve a book/rating that already has a value on this side: `newest` (later `updated_at` wins), `furthest` (further into the book wins; a finished read always beats a partial one), `skip_existing`, or `overwrite`. A rating has no position, so `furthest` behaves like `newest` for ratings, and a file without a rating timestamp never replaces an existing rating except under `overwrite` | +| `reattach_sessions` | `true` | When a session or completion in the file already exists as your own row but is not on a live book (its book was deleted, or the scanner marked it deleted after the file moved), move it onto the matched book instead of skipping it. A no-op, reported as such, when the file carries no sessions | +| `accept_stem_matches` | `false` | Apply a book match found only by filename stem | + +`GET /api/v1/reading-progress/export` takes one query parameter, +`include_sessions` (default `true`). Turn it off only if you specifically want +a smaller file: sessions are the only source of every reading statistic, so +leaving them out is easy to do by accident and easy not to notice until the +numbers are gone. + +## The response + +The response is the same shape whether or not `dry_run` is set: counts, plus a +per-series and per-book breakdown of what matched, what did not, and what was +(or would be) written. Each series is applied in its own transaction, so a bad +series does not cost every other series in the file its progress; the +per-series `committed` field says which ones actually landed. + +Ratings must be between 1 and 100, the same range the rating endpoint +enforces; a file with any other value is rejected with a 400 naming the +series. Imports up to 64 MB are accepted, which is room for tens of thousands +of books with their sessions. + +## Limitations + +- Metadata, collections, read lists, and covers are not carried; this moves + reading state only. See [Data Exports](../exports) for a metadata export, and + [`codex export`](./export-import-copy.md) for a full instance backup. +- A series with no external ids whose name and path both changed will not + match. Give it an external id (or a manual one) before the move if you can. +- Cross-instance import depends on both instances recognizing the same + external id sources. diff --git a/docs/docs/reading-progress.md b/docs/docs/reading-progress.md index d0d7235e2..229fe2be4 100644 --- a/docs/docs/reading-progress.md +++ b/docs/docs/reading-progress.md @@ -144,8 +144,14 @@ removed books** notice above the series panel, whatever dates you are viewing. If you would rather that reading stopped counting, use **Delete** on that notice. The confirmation states how many sittings, how much reading time and how many finished read-throughs will go, across **all** dates rather than only -the ones on screen, because that is what is deleted. It is permanent, affects only your own history, and never -touches reading attributed to a book that still exists. +the ones on screen, because that is what is deleted. It is permanent, affects +only your own history, and never touches reading attributed to a book that +still exists. + +Before deleting, check whether you have a reading progress export taken while +those books still existed. Importing it after the files are scanned again puts +this reading back on its books instead of discarding it; see +[Carrying Reading Progress Across a Library Split](./backup-migration/reading-progress-transfer.md). The same is available through the API: `GET /api/v1/reading-stats/orphaned` returns the totals, and `DELETE /api/v1/reading-stats/orphaned` removes them. @@ -167,3 +173,11 @@ Books you had already finished before this feature existed are counted: the first upgrade records one completion for every book currently marked read, dated from its completion time where that was stored and from its last-updated time otherwise. You do not need to re-read anything to get a starting count. + +## Moving progress across a library reorganisation + +Splitting a library, moving files, or moving to a different Codex instance +mints new series and book ids, and none of the state above follows +automatically. See +[Carrying Reading Progress Across a Library Split](./backup-migration/reading-progress-transfer.md) +for the export/import workflow that moves it. diff --git a/docs/sidebars.ts b/docs/sidebars.ts index 728e834ad..1c0d5019f 100644 --- a/docs/sidebars.ts +++ b/docs/sidebars.ts @@ -108,6 +108,7 @@ const sidebars: SidebarsConfig = { items: [ "backup-migration/export-import-copy", "backup-migration/migrate-postgres", + "backup-migration/reading-progress-transfer", ], }, { diff --git a/tests/api/mod.rs b/tests/api/mod.rs index e3764a904..7bfa5ad07 100644 --- a/tests/api/mod.rs +++ b/tests/api/mod.rs @@ -50,6 +50,7 @@ mod rate_limit; mod read_history; mod read_progress; mod reading_direction; +mod reading_progress_transfer; mod reading_sessions; mod reading_stats; mod readlists; diff --git a/tests/api/openapi_spec.rs b/tests/api/openapi_spec.rs index 0349e0390..211f6107a 100644 --- a/tests/api/openapi_spec.rs +++ b/tests/api/openapi_spec.rs @@ -397,6 +397,7 @@ const ACCEPTED_UNREFERENCED_COMPONENTS: &[&str] = &[ // because they are also registered in `docs.rs` `schemas()`, which is // unnecessary but harmless. "BooksPaginationQuery", + "ExportReadingProgressQuery", "ListFilterPresetsQuery", "ListSettingsQuery", "OrphanStatsQuery", diff --git a/tests/api/reading_progress_transfer.rs b/tests/api/reading_progress_transfer.rs new file mode 100644 index 000000000..e6639526a --- /dev/null +++ b/tests/api/reading_progress_transfer.rs @@ -0,0 +1,1143 @@ +//! Integration tests for `GET /api/v1/reading-progress/export` and +//! `POST /api/v1/reading-progress/import`. +//! +//! These exercise the properties the feature exists for: a library split +//! (export, delete the old library, rescan under new roots, import) must land +//! state on the new book ids; importing the same file twice must not +//! duplicate anything; a dry run must write nothing; and a book the importing +//! user cannot see must resolve as unmatched rather than leak through a +//! permission error. + +#[path = "../common/mod.rs"] +mod common; + +use chrono::{Duration, Utc}; +use codex::api::routes::v1::dto::{ + BookDisposition, ConflictPolicy, ExportBookDto, ExportCompletionDto, ExportProgressDto, + ExportSeriesDto, ExportSessionDto, FieldOutcome, HashMode, ImportReadingProgressRequest, + ImportReadingProgressResponse, READING_PROGRESS_FORMAT, READING_PROGRESS_VERSION, + ReadingProgressExportDocument, SeriesDisposition, +}; +use codex::db::ScanningStrategy; +use codex::db::entities::reading_sessions::SessionKind; +use codex::db::entities::user_sharing_tags::AccessMode; +use codex::db::entities::{ + books, read_completions, read_progress, reading_sessions, user_series_ratings, +}; +use codex::db::repositories::{ + BookRepository, LibraryRepository, NewSession, ReadCompletionRepository, + ReadProgressRepository, SeriesRepository, SharingTagRepository, UserRepository, + UserSeriesRatingRepository, +}; +use codex::utils::password; +use common::*; +use hyper::StatusCode; +use sea_orm::{ColumnTrait, DatabaseConnection, EntityTrait, QueryFilter}; +use uuid::Uuid; + +async fn admin_and_token( + db: &DatabaseConnection, + state: &codex::api::extractors::AuthState, + username: &str, +) -> (Uuid, String) { + let password_hash = password::hash_password("pw123456").unwrap(); + let user = create_test_user( + username, + &format!("{username}@example.com"), + &password_hash, + true, + ); + let created = UserRepository::create(db, &user).await.unwrap(); + let token = state + .jwt_service + .generate_token(created.id, created.username.clone(), created.get_role()) + .unwrap(); + (created.id, token) +} + +fn book_model( + series_id: Uuid, + library_id: Uuid, + path: &str, + file_name: &str, + hash: &str, +) -> books::Model { + books::Model { + id: Uuid::new_v4(), + series_id, + library_id, + path: path.to_string(), + file_name: file_name.to_string(), + file_size: 100, + file_hash: hash.to_string(), + partial_hash: String::new(), + format: "cbz".to_string(), + page_count: 20, + deleted: false, + analyzed: !hash.is_empty(), + analysis_error: None, + analysis_errors: None, + modified_at: Utc::now(), + created_at: Utc::now(), + updated_at: Utc::now(), + thumbnail_path: None, + thumbnail_generated_at: None, + koreader_hash: None, + epub_positions: None, + epub_spine_items: None, + } +} + +async fn table_counts(db: &DatabaseConnection) -> (usize, usize, usize, usize) { + ( + read_progress::Entity::find().all(db).await.unwrap().len(), + read_completions::Entity::find() + .all(db) + .await + .unwrap() + .len(), + reading_sessions::Entity::find() + .all(db) + .await + .unwrap() + .len(), + user_series_ratings::Entity::find() + .all(db) + .await + .unwrap() + .len(), + ) +} + +fn minimal_book_doc(path: &str, file_name: &str, hash: &str, current_page: i32) -> ExportBookDto { + let now = Utc::now(); + ExportBookDto { + path: path.to_string(), + file_name: file_name.to_string(), + file_hash: hash.to_string(), + partial_hash: String::new(), + progress: Some(ExportProgressDto { + current_page, + progress_percentage: None, + completed: false, + started_at: now - Duration::hours(1), + updated_at: now, + completed_at: None, + r2_progression: None, + }), + completions: vec![], + sessions: None, + } +} + +fn document( + series: Vec, + includes_sessions: bool, +) -> ReadingProgressExportDocument { + ReadingProgressExportDocument { + format: READING_PROGRESS_FORMAT.to_string(), + version: READING_PROGRESS_VERSION, + exported_at: Utc::now(), + includes_sessions, + series, + } +} + +fn import_request( + file: ReadingProgressExportDocument, + dry_run: bool, +) -> ImportReadingProgressRequest { + ImportReadingProgressRequest { + dry_run, + hash_mode: HashMode::Verify, + source_preference: vec![], + conflict_policy: ConflictPolicy::Overwrite, + reattach_sessions: true, + accept_stem_matches: false, + file, + } +} + +#[tokio::test] +async fn export_returns_a_downloadable_document_with_a_content_disposition_header() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(&db, &state, "exporter").await; + + let library = LibraryRepository::create(&db, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(&db, library.id, "Naruto", None) + .await + .unwrap(); + let book = BookRepository::create( + &db, + &book_model( + series.id, + library.id, + "/lib/Naruto/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + ReadProgressRepository::upsert(&db, user_id, book.id, 5, false) + .await + .unwrap(); + + let app = create_test_router(state.clone()).await; + let request = get_request_with_auth("/api/v1/reading-progress/export", &token); + let (status, headers, body) = make_full_request(app, request).await; + + assert_eq!(status, StatusCode::OK); + let disposition = headers + .get("content-disposition") + .expect("content-disposition header"); + assert!( + disposition + .to_str() + .unwrap() + .contains("codex-reading-progress") + ); + + let doc: ReadingProgressExportDocument = serde_json::from_slice(&body).unwrap(); + assert_eq!(doc.format, READING_PROGRESS_FORMAT); + assert_eq!(doc.series.len(), 1); + assert_eq!(doc.series[0].library_relative_path, "Naruto"); + assert_eq!(doc.series[0].books[0].path, "v01.cbz"); +} + +/// The scenario the feature exists for: two series exported from one library, +/// the library deleted, the same files rescanned under two new library roots, +/// and the import landing progress, completions, sessions, and the series +/// rating on the new ids. +#[tokio::test] +async fn library_split_round_trip_moves_state_to_the_new_books() { + let (db, _tmp) = setup_test_db().await; + exercise_library_split_round_trip(&db).await; +} + +async fn exercise_library_split_round_trip(db: &DatabaseConnection) { + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(db, &state, "splitter").await; + + // --- Old layout: one library, two series --- + let old_library = LibraryRepository::create(db, "Old", "/old", ScanningStrategy::Default) + .await + .unwrap(); + let old_series_a = SeriesRepository::create(db, old_library.id, "Naruto", None) + .await + .unwrap(); + let old_series_b = SeriesRepository::create(db, old_library.id, "Bleach", None) + .await + .unwrap(); + let old_book_a = BookRepository::create( + db, + &book_model( + old_series_a.id, + old_library.id, + "/old/Naruto/v01.cbz", + "v01.cbz", + "hash-naruto", + ), + None, + ) + .await + .unwrap(); + let old_book_b = BookRepository::create( + db, + &book_model( + old_series_b.id, + old_library.id, + "/old/Bleach/v01.cbz", + "v01.cbz", + "hash-bleach", + ), + None, + ) + .await + .unwrap(); + + // Progress, a completion, a session, and a rating on series A. + ReadProgressRepository::upsert(db, user_id, old_book_a.id, 12, false) + .await + .unwrap(); + let completion = ReadCompletionRepository::record( + db, + user_id, + old_book_a.id, + Utc::now() - Duration::days(1), + Utc::now(), + ) + .await + .unwrap(); + let session = NewSession::from_client( + Uuid::new_v4(), + user_id, + old_book_a.id, + "device-1", + Some("Test Device".to_string()), + SessionKind::Progress, + Some(600_000), + Some(12), + Utc::now() - Duration::minutes(10), + Utc::now(), + ) + .with_page(12); + ReadProgressRepository::record_session(db, session) + .await + .unwrap(); + UserSeriesRatingRepository::create(db, user_id, old_series_a.id, 90, Some("great".to_string())) + .await + .unwrap(); + + // Progress on series B too, to prove both series move. + ReadProgressRepository::upsert(db, user_id, old_book_b.id, 3, false) + .await + .unwrap(); + + // --- Export before the split --- + let app = create_test_router(state.clone()).await; + let request = get_request_with_auth( + "/api/v1/reading-progress/export?include_sessions=true", + &token, + ); + let (status, exported): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + let exported = exported.expect("export body"); + assert_eq!(exported.series.len(), 2); + + // --- The split: delete the old library, rescan under two new roots --- + LibraryRepository::delete(db, old_library.id).await.unwrap(); + + let new_library_a = LibraryRepository::create(db, "New A", "/new-a", ScanningStrategy::Default) + .await + .unwrap(); + let new_series_a = SeriesRepository::create(db, new_library_a.id, "Naruto", None) + .await + .unwrap(); + let new_book_a = BookRepository::create( + db, + &book_model( + new_series_a.id, + new_library_a.id, + "/new-a/Naruto/v01.cbz", + "v01.cbz", + "hash-naruto", + ), + None, + ) + .await + .unwrap(); + + let new_library_b = LibraryRepository::create(db, "New B", "/new-b", ScanningStrategy::Default) + .await + .unwrap(); + let new_series_b = SeriesRepository::create(db, new_library_b.id, "Bleach", None) + .await + .unwrap(); + let new_book_b = BookRepository::create( + db, + &book_model( + new_series_b.id, + new_library_b.id, + "/new-b/Bleach/v01.cbz", + "v01.cbz", + "hash-bleach", + ), + None, + ) + .await + .unwrap(); + + // --- Import into the new layout --- + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(exported, false), + &token, + ); + let (status, response): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + let response = response.expect("import response"); + assert!(!response.dry_run); + assert_eq!(response.summary.series_matched, 2); + assert_eq!(response.summary.series_committed, 2); + assert_eq!(response.summary.books_matched, 2); + + // Progress landed on the new books. + let progress_a = ReadProgressRepository::get_by_user_and_book(db, user_id, new_book_a.id) + .await + .unwrap() + .expect("progress on new book A"); + assert_eq!(progress_a.current_page, 12); + let progress_b = ReadProgressRepository::get_by_user_and_book(db, user_id, new_book_b.id) + .await + .unwrap() + .expect("progress on new book B"); + assert_eq!(progress_b.current_page, 3); + + // The completion and session, both originally recorded on the deleted + // book, were reattached (same id) rather than re-inserted. + let completion_row = read_completions::Entity::find_by_id(completion.id) + .one(db) + .await + .unwrap() + .expect("completion row still exists"); + assert_eq!(completion_row.book_id, Some(new_book_a.id)); + + // Two sessions land here: the explicit one recorded above, plus the + // legacy-write session `ReadProgressRepository::upsert` records under the + // hood for its own progress update. Both must be reattached exactly once + // each, never duplicated. + let sessions_for_a = reading_sessions::Entity::find() + .filter(reading_sessions::Column::BookId.eq(new_book_a.id)) + .all(db) + .await + .unwrap(); + assert_eq!( + sessions_for_a.len(), + 2, + "every orphaned session must be reattached exactly once" + ); + + // The rating followed series A to its new id. + let rating = UserSeriesRatingRepository::get_by_user_and_series(db, user_id, new_series_a.id) + .await + .unwrap() + .expect("rating on new series A"); + assert_eq!(rating.rating, 90); +} + +#[tokio::test] +async fn importing_the_same_document_twice_does_not_change_row_counts() { + let (db, _tmp) = setup_test_db().await; + exercise_idempotent_import(&db).await; +} + +async fn exercise_idempotent_import(db: &DatabaseConnection) { + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(db, &state, "idempotent").await; + + let library = LibraryRepository::create(db, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(db, library.id, "Series", None) + .await + .unwrap(); + let _book = BookRepository::create( + db, + &book_model( + series.id, + library.id, + "/lib/Series/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + + let mut book_doc = minimal_book_doc("v01.cbz", "v01.cbz", "h1", 7); + book_doc.completions.push(ExportCompletionDto { + id: Uuid::new_v4(), + started_at: Utc::now() - Duration::days(1), + completed_at: Utc::now(), + }); + book_doc.sessions = Some(vec![ExportSessionDto { + id: Uuid::new_v4(), + device_id: "device-1".to_string(), + device_name: None, + pass: 1, + kind: "progress".to_string(), + to_page: Some(7), + to_percentage: None, + active_duration_ms: Some(60_000), + duration_source: "measured".to_string(), + pages_read: Some(7), + client_started_at: Utc::now() - Duration::minutes(10), + client_ended_at: Utc::now(), + server_recorded_at: Utc::now(), + }]); + + let series_doc = ExportSeriesDto { + external_ids: vec![], + library_relative_path: "Series".to_string(), + name: "Series".to_string(), + rating: Some(77), + notes: None, + rating_updated_at: None, + books: vec![book_doc], + }; + let doc = document(vec![series_doc], true); + + // Relative to what is already there: the PostgreSQL run shares one + // database across several scenarios. + let before = table_counts(db).await; + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc.clone(), false), + &token, + ); + let (status, _): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + + let counts_after_first = table_counts(db).await; + assert_eq!( + counts_after_first, + (before.0 + 1, before.1 + 1, before.2 + 1, before.3 + 1), + "the first import writes one row to each table" + ); + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc, false), + &token, + ); + let (status, _): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + + let counts_after_second = table_counts(db).await; + assert_eq!( + counts_after_first, counts_after_second, + "importing the same document twice must not change row counts" + ); +} + +#[tokio::test] +async fn dry_run_reports_matches_but_writes_nothing() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(&db, &state, "dryrunner").await; + + let library = LibraryRepository::create(&db, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(&db, library.id, "Series", None) + .await + .unwrap(); + BookRepository::create( + &db, + &book_model( + series.id, + library.id, + "/lib/Series/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + + let series_doc = ExportSeriesDto { + external_ids: vec![], + library_relative_path: "Series".to_string(), + name: "Series".to_string(), + rating: Some(50), + notes: None, + rating_updated_at: None, + books: vec![minimal_book_doc("v01.cbz", "v01.cbz", "h1", 9)], + }; + let doc = document(vec![series_doc], false); + + let before = table_counts(&db).await; + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc, true), + &token, + ); + let (status, response): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + let response = response.expect("dry run response"); + assert!(response.dry_run); + assert_eq!(response.summary.series_matched, 1); + assert_eq!(response.summary.books_matched, 1); + assert_eq!( + response.summary.series_committed, 0, + "a dry run commits nothing" + ); + + let after = table_counts(&db).await; + assert_eq!(before, after, "a dry run must leave the database unchanged"); +} + +/// The one security-relevant requirement: a book behind a sharing-tag deny +/// resolves as unmatched, not as a permission error, and nothing is written. +#[tokio::test] +async fn a_book_the_importing_user_cannot_see_resolves_as_unmatched_and_nothing_is_written() { + let (db, _tmp) = setup_test_db().await; + exercise_visibility_denies_unmatched(&db).await; + exercise_two_entries_one_book(&db).await; + exercise_same_instance_split(&db).await; +} + +async fn exercise_visibility_denies_unmatched(db: &DatabaseConnection) { + // Relative, because the PostgreSQL run shares one database across scenarios. + let before = table_counts(db).await; + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(db, &state, "restricted-reader").await; + + let library = LibraryRepository::create(db, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(db, library.id, "Hidden", None) + .await + .unwrap(); + BookRepository::create( + db, + &book_model( + series.id, + library.id, + "/lib/Hidden/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + + let tag = SharingTagRepository::create(db, &format!("restricted-{}", Uuid::new_v4()), None) + .await + .unwrap(); + SharingTagRepository::add_tag_to_series(db, series.id, tag.id) + .await + .unwrap(); + SharingTagRepository::set_user_grant(db, user_id, tag.id, AccessMode::Deny) + .await + .unwrap(); + + let series_doc = ExportSeriesDto { + external_ids: vec![], + library_relative_path: "Hidden".to_string(), + name: "Hidden".to_string(), + rating: Some(100), + notes: None, + rating_updated_at: None, + books: vec![minimal_book_doc("v01.cbz", "v01.cbz", "h1", 1)], + }; + let doc = document(vec![series_doc], false); + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc, false), + &token, + ); + let (status, response): (StatusCode, Option) = + make_json_request(app, request).await; + + // Never a permission error: the response is a normal 200 that simply + // could not resolve anything, which does not confirm the series exists. + assert_eq!(status, StatusCode::OK); + let response = response.expect("import response"); + assert_eq!(response.series.len(), 1); + assert_eq!(response.series[0].disposition, SeriesDisposition::Unmatched); + assert_eq!( + response.series[0].books[0].disposition, + BookDisposition::Unmatched + ); + assert!(!response.series[0].books[0].applied); + + let counts = table_counts(db).await; + assert_eq!( + counts, before, + "nothing may be written against an invisible book" + ); +} + +#[tokio::test] +async fn unknown_format_is_rejected_with_400() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(&db, &state, "u1").await; + + let mut doc = document(vec![], false); + doc.format = "something-else".to_string(); + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc, true), + &token, + ); + let (status, body) = make_request(app, request).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + let text = String::from_utf8_lossy(&body); + assert!( + text.to_lowercase().contains("format"), + "error should name the problem: {text}" + ); +} + +#[tokio::test] +async fn a_version_newer_than_supported_is_rejected_with_400() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(&db, &state, "u2").await; + + let mut doc = document(vec![], false); + doc.version = READING_PROGRESS_VERSION + 1; + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(doc, true), + &token, + ); + let (status, body) = make_request(app, request).await; + assert_eq!(status, StatusCode::BAD_REQUEST); + let text = String::from_utf8_lossy(&body); + assert!( + text.to_lowercase().contains("version"), + "error should name the problem: {text}" + ); +} + +/// The narrow reattach path called out explicitly: record history, hard-delete +/// the book with `books::Entity::delete_by_id`, then import and assert the +/// orphaned session and completion are reattached rather than re-inserted. +#[tokio::test] +async fn hard_deleting_a_book_then_reimporting_reattaches_its_orphaned_history() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(&db, &state, "reattacher").await; + + let library = LibraryRepository::create(&db, "Lib", "/lib", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(&db, library.id, "Series", None) + .await + .unwrap(); + let book = BookRepository::create( + &db, + &book_model( + series.id, + library.id, + "/lib/Series/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + + let completion = ReadCompletionRepository::record( + &db, + user_id, + book.id, + Utc::now() - Duration::days(1), + Utc::now(), + ) + .await + .unwrap(); + let session = NewSession::from_client( + Uuid::new_v4(), + user_id, + book.id, + "device-1", + None, + SessionKind::Progress, + Some(60_000), + Some(5), + Utc::now() - Duration::minutes(5), + Utc::now(), + ) + .with_page(5); + ReadProgressRepository::record_session(&db, session) + .await + .unwrap(); + + // Export while the book still exists, so the file has something to + // match back against. + let app = create_test_router(state.clone()).await; + let request = get_request_with_auth( + "/api/v1/reading-progress/export?include_sessions=true", + &token, + ); + let (status, exported): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + let exported = exported.unwrap(); + + // Hard-delete the book: the session and completion survive as orphans. + books::Entity::delete_by_id(book.id) + .exec(&db) + .await + .unwrap(); + + // A rescan recreates the book at the same path with a new id. + let rescanned = BookRepository::create( + &db, + &book_model( + series.id, + library.id, + "/lib/Series/v01.cbz", + "v01.cbz", + "h1", + ), + None, + ) + .await + .unwrap(); + assert_ne!(rescanned.id, book.id); + + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(exported, false), + &token, + ); + let (status, response): (StatusCode, Option) = + make_json_request(app, request).await; + assert_eq!(status, StatusCode::OK); + let response = response.unwrap(); + assert_eq!(response.summary.completions_reattached, 1); + assert_eq!(response.summary.sessions_reattached, 1); + assert_eq!(response.summary.completions_inserted, 0); + assert_eq!(response.summary.sessions_inserted, 0); + + let completion_row = read_completions::Entity::find_by_id(completion.id) + .one(&db) + .await + .unwrap() + .unwrap(); + assert_eq!(completion_row.book_id, Some(rescanned.id)); + + let session_count = reading_sessions::Entity::find() + .all(&db) + .await + .unwrap() + .len(); + assert_eq!( + session_count, 1, + "the session must be reattached, not duplicated" + ); +} + +// ============================================================================ +// Regressions: values the normal write paths reject, colliding entries, the +// same-instance split, and a history larger than axum's default body limit. +// ============================================================================ + +fn series_doc(path: &str, name: &str, books: Vec) -> ExportSeriesDto { + ExportSeriesDto { + external_ids: vec![], + library_relative_path: path.to_string(), + name: name.to_string(), + rating: None, + notes: None, + rating_updated_at: None, + books, + } +} + +/// Ratings feed an average every user sees, so a file must not get around +/// the range the rating endpoint enforces. +#[tokio::test] +async fn a_rating_outside_1_to_100_is_rejected_with_400() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(&db, &state, "rater").await; + + for rating in [0, 101, 2_000_000_000] { + let mut series = series_doc("S", "S", vec![]); + series.rating = Some(rating); + let app = create_test_router(state.clone()).await; + let request = post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(document(vec![series], false), true), + &token, + ); + let (status, _body) = make_request(app, request).await; + assert_eq!(status, StatusCode::BAD_REQUEST, "rating {rating}"); + } +} + +/// A moved file leaves its old row soft-deleted, the export carries both, and +/// both resolve to the one current book. The second entry is decided against +/// the first under the conflict policy instead of colliding with it, so the +/// series commits and the dry run predicted exactly that. +#[tokio::test] +async fn two_entries_landing_on_one_book_are_resolved_by_the_policy() { + let (db, _tmp) = setup_test_db().await; + exercise_two_entries_one_book(&db).await; +} + +async fn exercise_two_entries_one_book(db: &DatabaseConnection) { + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(db, &state, "mover").await; + + let library = LibraryRepository::create(db, "Lib", "/twice", ScanningStrategy::Default) + .await + .unwrap(); + let series = SeriesRepository::create(db, library.id, "Twice Told", None) + .await + .unwrap(); + let book = BookRepository::create( + db, + &book_model( + series.id, + library.id, + "/twice/Twice Told/Vol 01/v01.cbz", + "v01.cbz", + "", + ), + None, + ) + .await + .unwrap(); + + let mut older = minimal_book_doc("v01.cbz", "v01.cbz", "", 5); + older.progress.as_mut().unwrap().updated_at = Utc::now() - Duration::days(2); + let newer = minimal_book_doc("Vol 01/v01.cbz", "v01.cbz", "", 9); + let doc = document( + vec![series_doc("Twice Told", "Twice Told", vec![newer, older])], + false, + ); + + let mut request = import_request(doc, true); + request.conflict_policy = ConflictPolicy::Newest; + let app = create_test_router(state.clone()).await; + let (status, preview): (StatusCode, Option) = make_json_request( + app, + post_json_request_with_auth("/api/v1/reading-progress/import", &request, &token), + ) + .await; + assert_eq!(status, StatusCode::OK); + let preview = preview.expect("dry-run body"); + let outcomes: Vec<_> = preview.series[0].books.iter().map(|b| b.progress).collect(); + assert_eq!( + outcomes, + vec![Some(FieldOutcome::Inserted), Some(FieldOutcome::Skipped)], + "the older entry loses to the one planned before it" + ); + + request.dry_run = false; + let app = create_test_router(state.clone()).await; + let (status, applied): (StatusCode, Option) = make_json_request( + app, + post_json_request_with_auth("/api/v1/reading-progress/import", &request, &token), + ) + .await; + assert_eq!(status, StatusCode::OK); + let applied = applied.expect("import body"); + assert!(applied.series[0].committed, "{:?}", applied.series[0].error); + + let progress = ReadProgressRepository::get_by_user_and_book(db, user_id, book.id) + .await + .unwrap() + .expect("progress landed"); + assert_eq!(progress.current_page, 9, "the newer position wins"); +} + +/// Splitting on the same instance: the files move, a rescan soft-deletes the +/// old books, and the reader imports before deleting the old library. The +/// leftover series must not capture the match, and history still sitting on +/// the soft-deleted books moves to the new ones rather than being skipped. +#[tokio::test] +async fn importing_before_the_old_library_is_deleted_moves_history_to_the_new_books() { + let (db, _tmp) = setup_test_db().await; + exercise_same_instance_split(&db).await; +} + +async fn exercise_same_instance_split(db: &DatabaseConnection) { + use sea_orm::{ActiveModelTrait, Set}; + + let state = create_test_auth_state(db.clone()).await; + let (user_id, token) = admin_and_token(db, &state, "same-instance").await; + + let old_library = LibraryRepository::create(db, "Manga", "/manga", ScanningStrategy::Default) + .await + .unwrap(); + let old_series = SeriesRepository::create(db, old_library.id, "Split Series", None) + .await + .unwrap(); + SeriesRepository::update_path(db, old_series.id, "shonen/Split Series".to_string()) + .await + .unwrap(); + let old_book = BookRepository::create( + db, + &book_model( + old_series.id, + old_library.id, + "/manga/shonen/Split Series/v01.cbz", + "v01.cbz", + "hash-split", + ), + None, + ) + .await + .unwrap(); + + let completion = + ReadCompletionRepository::record(db, user_id, old_book.id, Utc::now(), Utc::now()) + .await + .unwrap(); + let session_id = Uuid::new_v4(); + ReadProgressRepository::record_session( + db, + NewSession::from_client( + session_id, + user_id, + old_book.id, + "device-1", + None, + SessionKind::Progress, + Some(60_000), + Some(4), + Utc::now() - Duration::minutes(5), + Utc::now(), + ) + .with_page(4), + ) + .await + .unwrap(); + + // The files move to a new library; the old library's rescan marks the + // book deleted but nobody has deleted the old library yet. + let app = create_test_router(state.clone()).await; + let (_, exported): (StatusCode, Option) = make_json_request( + app, + get_request_with_auth("/api/v1/reading-progress/export", &token), + ) + .await; + let exported = exported.expect("export body"); + + let mut gone: books::ActiveModel = old_book.clone().into(); + gone.deleted = Set(true); + gone.update(db).await.unwrap(); + + let new_library = LibraryRepository::create(db, "Shonen", "/shonen", ScanningStrategy::Default) + .await + .unwrap(); + let new_series = SeriesRepository::create(db, new_library.id, "Split Series", None) + .await + .unwrap(); + let new_book = BookRepository::create( + db, + &book_model( + new_series.id, + new_library.id, + "/shonen/Split Series/v01.cbz", + "v01.cbz", + "hash-split", + ), + None, + ) + .await + .unwrap(); + + let app = create_test_router(state.clone()).await; + let (status, response): (StatusCode, Option) = + make_json_request( + app, + post_json_request_with_auth( + "/api/v1/reading-progress/import", + &import_request(exported, false), + &token, + ), + ) + .await; + assert_eq!(status, StatusCode::OK); + let response = response.expect("import body"); + assert_eq!( + response.series[0].matched_series_id, + Some(new_series.id), + "the leftover series with nothing on disk must not capture the match" + ); + assert!(response.series[0].committed); + + let completion = read_completions::Entity::find_by_id(completion.id) + .one(db) + .await + .unwrap() + .unwrap(); + assert_eq!(completion.book_id, Some(new_book.id)); + let session = reading_sessions::Entity::find_by_id(session_id) + .one(db) + .await + .unwrap() + .unwrap(); + assert_eq!(session.book_id, Some(new_book.id)); +} + +/// A large history has to fit: axum's 2 MB default would stop around 2,500 +/// books with their sessions. A dry run keeps this fast; the limit is the +/// point. +#[tokio::test] +async fn an_import_larger_than_two_megabytes_is_accepted() { + let (db, _tmp) = setup_test_db().await; + let state = create_test_auth_state(db.clone()).await; + let (_user_id, token) = admin_and_token(&db, &state, "heavy").await; + + let now = Utc::now(); + let sessions: Vec = (0..12_000) + .map(|i| ExportSessionDto { + id: Uuid::new_v4(), + device_id: "device-with-a-reasonably-long-identifier".to_string(), + device_name: Some("A reader with a descriptive name".to_string()), + pass: 1, + kind: "progress".to_string(), + to_page: Some(i), + to_percentage: None, + active_duration_ms: Some(60_000), + duration_source: "measured".to_string(), + pages_read: Some(1), + client_started_at: now - Duration::minutes(1), + client_ended_at: now, + server_recorded_at: now, + }) + .collect(); + let mut book = minimal_book_doc("v01.cbz", "v01.cbz", "", 1); + book.sessions = Some(sessions); + let request = import_request( + document(vec![series_doc("Heavy", "Heavy", vec![book])], true), + true, + ); + let body = serde_json::to_vec(&request).unwrap(); + assert!( + body.len() > 2 * 1024 * 1024, + "fixture must exceed the default" + ); + + let app = create_test_router(state).await; + let (status, _body) = make_request( + app, + post_json_request_with_auth("/api/v1/reading-progress/import", &request, &token), + ) + .await; + assert_eq!(status, StatusCode::OK); +} + +/// The scenarios that matter most, against PostgreSQL, sequenced in one test +/// on purpose: `setup_test_db_postgres` truncates a database shared by the +/// whole run, so two PostgreSQL tests running at once would delete each +/// other's fixtures. +#[tokio::test] +#[ignore] // Requires PostgreSQL test database +async fn reading_progress_transfer_postgres() { + let Some(db) = setup_test_db_postgres().await else { + eprintln!("PostgreSQL test database not available, skipping"); + return; + }; + + exercise_library_split_round_trip(&db).await; + exercise_idempotent_import(&db).await; + exercise_visibility_denies_unmatched(&db).await; +} diff --git a/web/openapi.json b/web/openapi.json index 157908ff7..d23780206 100644 --- a/web/openapi.json +++ b/web/openapi.json @@ -10237,6 +10237,105 @@ ] } }, + "/api/v1/reading-progress/export": { + "get": { + "tags": [ + "Reading Progress Transfer" + ], + "summary": "Export the authenticated user's reading progress", + "description": "Produces the whole `read_progress` / `read_completions` / `reading_sessions`\n/ `user_series_ratings` state for the caller, keyed by external ids and\nseries-relative paths instead of database ids so it can be matched back\nagainst a differently organised library (or a different Codex instance\nentirely) by `POST /api/v1/reading-progress/import`.\n\nTake this export **before** deleting an old library during a split: the\nunderlying rows survive a hard delete as orphans, but nothing except this\nfile can say which book an orphan used to belong to.", + "operationId": "export_reading_progress", + "parameters": [ + { + "name": "include_sessions", + "in": "query", + "description": "Include the reading-session log (default: true). Sessions are the only source of every reading statistic, so this is opt-out rather than opt-in.", + "required": false, + "schema": { + "type": "boolean" + } + } + ], + "responses": { + "200": { + "description": "The export document", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ReadingProgressExportDocument" + } + } + } + }, + "401": { + "description": "Unauthorized" + }, + "403": { + "description": "Forbidden" + } + }, + "security": [ + { + "jwt_bearer": [] + }, + { + "api_key": [] + } + ] + } + }, + "/api/v1/reading-progress/import": { + "post": { + "tags": [ + "Reading Progress Transfer" + ], + "summary": "Import reading progress from an export document", + "description": "Resolves the file's series and books against the current library through\nthe same access-group / sharing-tag visibility that an ordinary read\nrespects: a book the importing user cannot see resolves as `unmatched`,\nnever as a permission error that would confirm it exists, and nothing is\never written against it.\n\nMatching never guesses: any step (external id, path, file name, or, under\n`hash_mode = \"match\"`, hash) that finds more than one candidate reports\n`ambiguous` and writes nothing for that series or book. `dry_run: true`\nreturns the identical response shape without writing anything, which is\nwhat makes it safe to preview before committing.\n\nEach series is applied in its own transaction; the response's per-series\n`committed` field says which ones actually landed. `read_completions` and\n`reading_sessions` reuse their exported ids on insert, so importing the\nsame file twice leaves row counts unchanged rather than duplicating\nhistory.", + "operationId": "import_reading_progress", + "requestBody": { + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ImportReadingProgressRequest" + } + } + }, + "required": true + }, + "responses": { + "200": { + "description": "Import processed (or, for a dry run, previewed)", + "content": { + "application/json": { + "schema": { + "$ref": "#/components/schemas/ImportReadingProgressResponse" + } + } + } + }, + "400": { + "description": "Unknown export format, a version newer than this server supports, or a value the normal write paths reject" + }, + "401": { + "description": "Unauthorized" + }, + "403": { + "description": "Forbidden" + }, + "413": { + "description": "The file is larger than the import limit" + } + }, + "security": [ + { + "jwt_bearer": [] + }, + { + "api_key": [] + } + ] + } + }, "/api/v1/reading-sessions": { "post": { "tags": [ @@ -10474,7 +10573,7 @@ "Reading Statistics" ], "summary": "Delete the caller's reading history for books that no longer exist", - "description": "When a book is deleted from the server its reading sessions and finished\nread-throughs are kept, so the time still counts towards every statistic;\nthe series and format breakdowns show it as one \"removed from library\" row.\nThis discards those rows for the caller, and only for the caller.\n\nIrreversible. Only history already detached from any book is touched;\nattributed reading is never affected.", + "description": "When a book is deleted from the server its reading sessions and finished\nread-throughs are kept, so the time still counts towards every statistic;\nthe series and format breakdowns show it as one \"removed from library\" row.\nThis discards those rows for the caller, and only for the caller.\n\nIrreversible, and it forecloses the other way out: importing a reading\nprogress export taken before the delete puts those sessions back on their\nbooks once the files are scanned again. Only history already detached from\nany book is touched; attributed reading is never affected.", "operationId": "purge_orphaned_reading_history", "responses": { "200": { @@ -24698,6 +24797,17 @@ } } }, + "BookDisposition": { + "type": "string", + "description": "Why a book in the import file could not be resolved to exactly one book\nin its matched series.", + "enum": [ + "matched", + "stem_match", + "ambiguous", + "unmatched", + "hash_mismatch" + ] + }, "BookDto": { "type": "object", "description": "Book data transfer object", @@ -28315,6 +28425,16 @@ } } }, + "ConflictPolicy": { + "type": "string", + "description": "How a conflict between an imported value and an existing row is resolved.", + "enum": [ + "newest", + "furthest", + "skip_existing", + "overwrite" + ] + }, "Contributor": { "type": "object", "description": "Contributor information (author, artist, etc.)", @@ -30538,6 +30658,93 @@ } } }, + "ExportBookDto": { + "type": "object", + "description": "One book inside a series, keyed for matching by path, name, and hash\nrather than by id: the whole point of the file is that ids on the far side\nare expected to be different.", + "required": [ + "path", + "file_name" + ], + "properties": { + "completions": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportCompletionDto" + } + }, + "file_hash": { + "type": "string", + "description": "Empty when the book was never analyzed; never treated as a value to\nmatch on in that case." + }, + "file_name": { + "type": "string", + "example": "v01.cbz" + }, + "partial_hash": { + "type": "string" + }, + "path": { + "type": "string", + "description": "Relative to the series folder, so a series move does not invalidate it.", + "example": "Vol 01/v01.cbz" + }, + "progress": { + "$ref": "#/components/schemas/ExportProgressDto" + }, + "sessions": { + "type": [ + "array", + "null" + ], + "items": { + "$ref": "#/components/schemas/ExportSessionDto" + }, + "description": "Omitted entirely (not an empty array) when the export was taken with\n`include_sessions=false`." + } + } + }, + "ExportCompletionDto": { + "type": "object", + "description": "One finished read-through. Keeps its original id so re-importing the same\nfile is a no-op rather than a duplicate.", + "required": [ + "id", + "started_at", + "completed_at" + ], + "properties": { + "completed_at": { + "type": "string", + "format": "date-time" + }, + "id": { + "type": "string", + "format": "uuid" + }, + "started_at": { + "type": "string", + "format": "date-time" + } + } + }, + "ExportExternalIdDto": { + "type": "object", + "description": "One external identifier attached to a series (a plugin match, a ComicInfo\nvalue, or a manual entry).", + "required": [ + "source", + "id" + ], + "properties": { + "id": { + "type": "string", + "example": "12345" + }, + "source": { + "type": "string", + "description": "`plugin:`, `comicinfo`, `epub`, or `manual`.", + "example": "plugin:mangabaka" + } + } + }, "ExportFieldCatalogResponse": { "type": "object", "description": "Response for the field catalog", @@ -30619,6 +30826,198 @@ } } }, + "ExportProgressDto": { + "type": "object", + "description": "The live resume position for one book. Retains `r2_progression`: it is the\nonly place the EPUB locator survives, since sessions strip it.", + "required": [ + "current_page", + "completed", + "started_at", + "updated_at" + ], + "properties": { + "completed": { + "type": "boolean" + }, + "completed_at": { + "type": [ + "string", + "null" + ], + "format": "date-time" + }, + "current_page": { + "type": "integer", + "format": "int32" + }, + "progress_percentage": { + "type": [ + "number", + "null" + ], + "format": "double" + }, + "r2_progression": { + "type": [ + "string", + "null" + ] + }, + "started_at": { + "type": "string", + "format": "date-time" + }, + "updated_at": { + "type": "string", + "format": "date-time" + } + } + }, + "ExportReadingProgressQuery": { + "type": "object", + "description": "Query parameters for `GET /api/v1/reading-progress/export`.", + "properties": { + "include_sessions": { + "type": "boolean", + "description": "Sessions are opt-out: they are the only source of every reading\nstatistic, so leaving them out is easy to do by accident and hard to\nnotice until the numbers are gone." + } + } + }, + "ExportSeriesDto": { + "type": "object", + "description": "One series and everything the exporting user recorded against its books.", + "required": [ + "library_relative_path", + "name" + ], + "properties": { + "books": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportBookDto" + } + }, + "external_ids": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportExternalIdDto" + } + }, + "library_relative_path": { + "type": "string", + "description": "The series path as stored, relative to the library root.", + "example": "shonen/Naruto" + }, + "name": { + "type": "string", + "example": "Naruto" + }, + "notes": { + "type": [ + "string", + "null" + ] + }, + "rating": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "rating_updated_at": { + "type": [ + "string", + "null" + ], + "format": "date-time", + "description": "When the rating was last changed. The `newest` conflict policy needs it\nto tell a stale rating from a fresh one; a file without it never\noverwrites an existing rating except under `overwrite`." + } + } + }, + "ExportSessionDto": { + "type": "object", + "description": "One row from the reading-session log. `r2_progression` is deliberately\nabsent: nothing reads a session's historical locator, and it is the only\nnon-scalar column on the row.", + "required": [ + "id", + "device_id", + "pass", + "kind", + "duration_source", + "client_started_at", + "client_ended_at", + "server_recorded_at" + ], + "properties": { + "active_duration_ms": { + "type": [ + "integer", + "null" + ], + "format": "int64" + }, + "client_ended_at": { + "type": "string", + "format": "date-time" + }, + "client_started_at": { + "type": "string", + "format": "date-time" + }, + "device_id": { + "type": "string" + }, + "device_name": { + "type": [ + "string", + "null" + ] + }, + "duration_source": { + "type": "string", + "description": "`\"measured\"`, `\"inferred\"`, or `\"unknown\"`.", + "example": "measured" + }, + "id": { + "type": "string", + "format": "uuid" + }, + "kind": { + "type": "string", + "description": "`\"progress\"`, `\"completed\"`, or `\"reset\"`.", + "example": "progress" + }, + "pages_read": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "pass": { + "type": "integer", + "format": "int32" + }, + "server_recorded_at": { + "type": "string", + "format": "date-time" + }, + "to_page": { + "type": [ + "integer", + "null" + ], + "format": "int32" + }, + "to_percentage": { + "type": [ + "number", + "null" + ], + "format": "double" + } + } + }, "ExternalIdContextDto": { "type": "object", "description": "External ID context for template evaluation.\n\nRepresents an external ID from a metadata provider (plugin, comicinfo, etc.)\nin a simplified format suitable for template access.", @@ -31111,6 +31510,15 @@ ], "description": "Operators for string and equality comparisons" }, + "FieldOutcome": { + "type": "string", + "description": "What happened to one scalar field write (progress or rating) under the\nactive conflict policy.", + "enum": [ + "inserted", + "updated", + "skipped" + ] + }, "FileSystemEntry": { "type": "object", "required": [ @@ -32045,6 +32453,15 @@ } } }, + "HashMode": { + "type": "string", + "description": "How aggressively hashes are used to match a book.", + "enum": [ + "off", + "verify", + "match" + ] + }, "ImageLink": { "type": "object", "description": "Image link with optional dimensions\n\nUsed for cover images and thumbnails in publications.", @@ -32079,6 +32496,283 @@ } } }, + "ImportBookReport": { + "type": "object", + "description": "The outcome for one book in the import file.", + "required": [ + "path", + "file_name", + "disposition", + "applied", + "completions", + "sessions" + ], + "properties": { + "applied": { + "type": "boolean", + "description": "Whether any writes were attempted for this book. False for every\ndisposition except `matched`, and except `stem_match` when\n`accept_stem_matches` is off." + }, + "completions": { + "$ref": "#/components/schemas/WriteCounts" + }, + "disposition": { + "$ref": "#/components/schemas/BookDisposition" + }, + "file_name": { + "type": "string" + }, + "matched_book_id": { + "type": [ + "string", + "null" + ], + "format": "uuid" + }, + "path": { + "type": "string" + }, + "progress": { + "$ref": "#/components/schemas/FieldOutcome" + }, + "sessions": { + "$ref": "#/components/schemas/WriteCounts" + } + } + }, + "ImportReadingProgressRequest": { + "type": "object", + "description": "`POST /api/v1/reading-progress/import` request body.", + "required": [ + "file" + ], + "properties": { + "accept_stem_matches": { + "type": "boolean", + "description": "A stem match (`v01.cbr` renamed to `v01.cbz`) is reported either way,\nbut only written when this is set: two files can share a stem, and\napplying it silently risks writing progress onto the wrong one." + }, + "conflict_policy": { + "$ref": "#/components/schemas/ConflictPolicy" + }, + "dry_run": { + "type": "boolean", + "description": "Compute and report the outcome without writing anything." + }, + "file": { + "$ref": "#/components/schemas/ReadingProgressExportDocument" + }, + "hash_mode": { + "$ref": "#/components/schemas/HashMode" + }, + "reattach_sessions": { + "type": "boolean", + "description": "When a session or completion in the file already exists as the\nimporter's own row but is not on a live book (its book was hard-deleted,\nleaving `book_id` null, or the scanner marked it deleted after the file\nmoved), move it onto the matched book instead of skipping it." + }, + "source_preference": { + "type": "array", + "items": { + "type": "string" + }, + "description": "External-id sources to try, in order, before falling back to path and\nthen normalized name. An empty list skips straight to path matching." + } + } + }, + "ImportReadingProgressResponse": { + "type": "object", + "description": "The response for both a real import and a dry run: the shape is identical\neither way, so a client cannot tell from the response alone whether\nanything was written. Only `dry_run` (and the DB) says that.", + "required": [ + "dry_run", + "sessions_in_file", + "summary", + "series" + ], + "properties": { + "dry_run": { + "type": "boolean" + }, + "notices": { + "type": "array", + "items": { + "type": "string" + }, + "description": "Informational notes about the request, e.g. `reattach_sessions` having\nno effect because the file carries no sessions." + }, + "series": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ImportSeriesReport" + } + }, + "sessions_in_file": { + "type": "boolean" + }, + "summary": { + "$ref": "#/components/schemas/ImportSummary" + } + } + }, + "ImportSeriesReport": { + "type": "object", + "description": "The outcome for one series in the import file.", + "required": [ + "library_relative_path", + "name", + "disposition", + "attempted", + "committed", + "books" + ], + "properties": { + "attempted": { + "type": "boolean", + "description": "Whether book/rating processing was attempted at all (only when\n`disposition == matched`)." + }, + "books": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ImportBookReport" + } + }, + "committed": { + "type": "boolean", + "description": "Whether this series' transaction was committed. Always `false` in a\ndry run, and `false` if `attempted` but a write failed." + }, + "disposition": { + "$ref": "#/components/schemas/SeriesDisposition" + }, + "error": { + "type": [ + "string", + "null" + ] + }, + "library_relative_path": { + "type": "string" + }, + "matched_series_id": { + "type": [ + "string", + "null" + ], + "format": "uuid" + }, + "name": { + "type": "string" + }, + "rating": { + "$ref": "#/components/schemas/FieldOutcome" + } + } + }, + "ImportSummary": { + "type": "object", + "description": "Totals across every series in the file, for a one-line summary.", + "required": [ + "series_total", + "series_matched", + "series_ambiguous", + "series_unmatched", + "series_committed", + "books_total", + "books_matched", + "books_stem_matched", + "books_ambiguous", + "books_unmatched", + "books_hash_mismatch", + "progress_written", + "ratings_written", + "completions_inserted", + "completions_reattached", + "sessions_inserted", + "sessions_reattached" + ], + "properties": { + "books_ambiguous": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_hash_mismatch": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_stem_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_total": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "books_unmatched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "completions_inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "completions_reattached": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "progress_written": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "ratings_written": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_ambiguous": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_committed": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_matched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_total": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "series_unmatched": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "sessions_inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "sessions_reattached": { + "type": "integer", + "format": "int32", + "minimum": 0 + } + } + }, "InheritedFrom": { "type": "string", "description": "Which layer supplied a value the user is inheriting.", @@ -39320,6 +40014,40 @@ } } }, + "ReadingProgressExportDocument": { + "type": "object", + "description": "The whole export: one user's reading state, self-describing enough to be\nmatched back against a differently organised library.", + "required": [ + "format", + "version", + "exported_at", + "includes_sessions" + ], + "properties": { + "exported_at": { + "type": "string", + "format": "date-time" + }, + "format": { + "type": "string", + "example": "codex-reading-progress" + }, + "includes_sessions": { + "type": "boolean" + }, + "series": { + "type": "array", + "items": { + "$ref": "#/components/schemas/ExportSeriesDto" + } + }, + "version": { + "type": "integer", + "format": "int32", + "example": 1 + } + } + }, "ReadingSessionDto": { "type": "object", "description": "One reading session measured by a client.", @@ -41948,6 +42676,15 @@ } } }, + "SeriesDisposition": { + "type": "string", + "description": "Why a series in the import file could not be resolved to exactly one\nseries in the current library.", + "enum": [ + "matched", + "ambiguous", + "unmatched" + ] + }, "SeriesDto": { "type": "object", "description": "Series data transfer object", @@ -47381,6 +48118,34 @@ "description": "URL template with config placeholders already substituted. The only\nremaining placeholder is `{externalId}`, filled client-side." } } + }, + "WriteCounts": { + "type": "object", + "description": "Insert/reattach/skip counts for an append-only table (completions or\nsessions) within one book.", + "required": [ + "inserted", + "reattached", + "skipped" + ], + "properties": { + "inserted": { + "type": "integer", + "format": "int32", + "minimum": 0 + }, + "reattached": { + "type": "integer", + "format": "int32", + "description": "Adopted an orphaned row (`book_id IS NULL`) rather than inserting a\nnew one.", + "minimum": 0 + }, + "skipped": { + "type": "integer", + "format": "int32", + "description": "Already present with the same book attached; re-importing is a no-op.", + "minimum": 0 + } + } } }, "securitySchemes": { @@ -47474,6 +48239,10 @@ "name": "Reading Progress", "description": "Reading progress tracking" }, + { + "name": "Reading Progress Transfer", + "description": "Exporting and importing reading state across a library reorganisation or an instance move" + }, { "name": "Task Queue", "description": "Background job queue management" diff --git a/web/src/App.tsx b/web/src/App.tsx index bd4dc93a1..50b9efd1d 100644 --- a/web/src/App.tsx +++ b/web/src/App.tsx @@ -44,6 +44,7 @@ import { PluginStorageSettings, PluginsSettings, ProfileSettings, + ReadingProgressTransferSettings, ReleaseTrackingSettings, SeriesExportsSettings, ServerSettings, @@ -529,6 +530,17 @@ function App() { } /> + + + + + + } + /> + => { + const response = await api.get( + "/reading-progress/export", + { params: { include_sessions: includeSessions } }, + ); + return response.data; + }, + + /** + * Import reading progress from a previously exported document. + * Pass `dryRun: true` on the request to preview without writing. + */ + importProgress: async ( + request: ImportReadingProgressRequest, + ): Promise => { + const response = await api.post( + "/reading-progress/import", + request, + // A large history applies one series at a time; the client-wide 30 s + // timeout would abandon it part-way and lose the report of what landed. + { timeout: IMPORT_TIMEOUT_MS }, + ); + return response.data; + }, +}; diff --git a/web/src/components/layout/Sidebar.tsx b/web/src/components/layout/Sidebar.tsx index 844f34fa1..52746e576 100644 --- a/web/src/components/layout/Sidebar.tsx +++ b/web/src/components/layout/Sidebar.tsx @@ -16,6 +16,7 @@ import { useMediaQuery } from "@mantine/hooks"; import { notifications } from "@mantine/notifications"; import { IconAlertTriangle, + IconArrowsRightLeft, IconBookmark, IconBooks, IconBrush, @@ -820,6 +821,18 @@ export function Sidebar({ active={currentPath.startsWith("/settings/exports")} onClick={onNavigate} /> + + } + active={currentPath.startsWith( + "/settings/reading-progress", + )} + onClick={onNavigate} + /> {/* Account Section */} )} + {settled && !nothingLeft && ( + + If you have a reading progress export taken before these books + were removed, importing it puts this reading back on its books + instead. + + )} {nothingLeft && ( There is no removed history left to delete. diff --git a/web/src/hooks/useReadingProgressTransfer.ts b/web/src/hooks/useReadingProgressTransfer.ts new file mode 100644 index 000000000..909ac5508 --- /dev/null +++ b/web/src/hooks/useReadingProgressTransfer.ts @@ -0,0 +1,68 @@ +import { notifications } from "@mantine/notifications"; +import { useMutation } from "@tanstack/react-query"; +import { + type ImportReadingProgressRequest, + readingProgressTransferApi, +} from "@/api/readingProgressTransfer"; + +type ApiErrorLike = Error & { + response?: { data?: { message?: string; error?: string } }; +}; + +function errorMessage(error: ApiErrorLike, fallback: string): string { + return ( + error.response?.data?.message || + error.response?.data?.error || + error.message || + fallback + ); +} + +/** + * Export the current user's reading progress and trigger a browser download. + * The document is fetched as JSON (not a blob endpoint), so the download is + * built client-side from the response body. + */ +export function useExportReadingProgress() { + return useMutation({ + mutationFn: async (includeSessions: boolean) => { + const exported = + await readingProgressTransferApi.exportProgress(includeSessions); + + const timestamp = new Date().toISOString().slice(0, 10); + const filename = `codex-reading-progress-${timestamp}.json`; + const blob = new Blob([JSON.stringify(exported, null, 2)], { + type: "application/json", + }); + const url = URL.createObjectURL(blob); + const anchor = document.createElement("a"); + anchor.href = url; + anchor.download = filename; + document.body.appendChild(anchor); + anchor.click(); + document.body.removeChild(anchor); + URL.revokeObjectURL(url); + + return exported; + }, + onError: (error: ApiErrorLike) => { + notifications.show({ + title: "Export failed", + message: errorMessage(error, "Could not export reading progress."), + color: "red", + }); + }, + }); +} + +/** + * Run an import (dry run or real, per `request.dryRun`). Errors are surfaced + * to the caller rather than shown as a notification here, because the + * calling page renders a full report either way. + */ +export function useImportReadingProgress() { + return useMutation({ + mutationFn: (request: ImportReadingProgressRequest) => + readingProgressTransferApi.importProgress(request), + }); +} diff --git a/web/src/pages/settings/ReadingProgressTransferSettings.test.tsx b/web/src/pages/settings/ReadingProgressTransferSettings.test.tsx new file mode 100644 index 000000000..8dd89ff26 --- /dev/null +++ b/web/src/pages/settings/ReadingProgressTransferSettings.test.tsx @@ -0,0 +1,295 @@ +import { beforeEach, describe, expect, it, vi } from "vitest"; +import type { + ImportReadingProgressResponse, + ReadingProgressExportDocument, +} from "@/api/readingProgressTransfer"; +import { renderWithProviders, screen, userEvent, waitFor } from "@/test/utils"; +import { ReadingProgressTransferSettings } from "./ReadingProgressTransferSettings"; + +const exportProgress = vi.fn(); +const importProgress = vi.fn(); + +vi.mock("@/api/readingProgressTransfer", async () => { + const actual = await vi.importActual< + typeof import("@/api/readingProgressTransfer") + >("@/api/readingProgressTransfer"); + return { + ...actual, + readingProgressTransferApi: { + exportProgress: (...args: unknown[]) => exportProgress(...args), + importProgress: (...args: unknown[]) => importProgress(...args), + }, + }; +}); + +const exportDocument: ReadingProgressExportDocument = { + format: "codex-reading-progress", + version: 1, + exported_at: "2026-09-27T00:00:00Z", + includes_sessions: true, + series: [ + { + external_ids: [], + library_relative_path: "Naruto", + name: "Naruto", + books: [ + { + path: "v01.cbz", + file_name: "v01.cbz", + file_hash: "", + partial_hash: "", + completions: [], + }, + ], + }, + ], +}; + +function dryRunResponse(): ImportReadingProgressResponse { + return { + dry_run: true, + sessions_in_file: true, + notices: [], + summary: { + series_total: 1, + series_matched: 1, + series_ambiguous: 0, + series_unmatched: 0, + series_committed: 0, + books_total: 1, + books_matched: 1, + books_stem_matched: 0, + books_ambiguous: 0, + books_unmatched: 0, + books_hash_mismatch: 0, + progress_written: 1, + ratings_written: 0, + completions_inserted: 0, + completions_reattached: 0, + sessions_inserted: 0, + sessions_reattached: 0, + }, + series: [ + { + library_relative_path: "Naruto", + name: "Naruto", + disposition: "matched", + matched_series_id: "11111111-1111-1111-1111-111111111111", + attempted: true, + committed: false, + books: [ + { + path: "v01.cbz", + file_name: "v01.cbz", + disposition: "matched", + matched_book_id: "22222222-2222-2222-2222-222222222222", + applied: true, + progress: "inserted", + completions: { inserted: 0, reattached: 0, skipped: 0 }, + sessions: { inserted: 0, reattached: 0, skipped: 0 }, + }, + ], + }, + ], + }; +} + +async function uploadDocument(user: ReturnType) { + const file = new File([JSON.stringify(exportDocument)], "export.json", { + type: "application/json", + }); + const input = fileInput(); + await user.upload(input, file); + // The button label updates as soon as the file is selected, but parsing + // (`FileReader`) is async; wait for the preview button to actually unlock + // rather than the label, or a click can race ahead of the parsed document. + await waitFor(() => { + expect( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ).toBeEnabled(); + }); +} + +// The FileButton render prop wraps a hidden native file input with no +// accessible label, so it is queried directly rather than by role/label. +function fileInput(): HTMLInputElement { + const input = window.document.querySelector('input[type="file"]'); + if (!input) throw new Error("file input not found"); + return input as HTMLInputElement; +} + +describe("ReadingProgressTransferSettings", () => { + beforeEach(() => { + vi.clearAllMocks(); + }); + + it("downloads an export when the button is clicked", async () => { + exportProgress.mockResolvedValue(exportDocument); + const user = userEvent.setup(); + renderWithProviders(); + + await user.click(screen.getByRole("button", { name: /download export/i })); + + await waitFor(() => { + expect(exportProgress).toHaveBeenCalledWith(true); + }); + }); + + it("disables Apply until a dry run has been previewed for the selected file", async () => { + importProgress.mockResolvedValue(dryRunResponse()); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + + const applyButton = await screen.findByRole("button", { + name: /apply import/i, + }); + expect(applyButton).toBeDisabled(); + + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + + await waitFor(() => { + expect(applyButton).toBeEnabled(); + }); + expect(importProgress).toHaveBeenCalledWith( + expect.objectContaining({ dry_run: true }), + ); + }); + + it("re-locks Apply when an option changes after the preview", async () => { + importProgress.mockResolvedValue(dryRunResponse()); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + + const applyButton = await screen.findByRole("button", { + name: /apply import/i, + }); + await waitFor(() => expect(applyButton).toBeEnabled()); + + await user.click( + screen.getByRole("checkbox", { name: /accept filename-stem matches/i }), + ); + + expect(applyButton).toBeDisabled(); + }); + + it("renders the report table with per-series and per-book disposition", async () => { + importProgress.mockResolvedValue(dryRunResponse()); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + + expect((await screen.findAllByText("Naruto")).length).toBeGreaterThan(0); + expect(screen.getByText("v01.cbz")).toBeInTheDocument(); + expect(screen.getAllByText("matched").length).toBeGreaterThan(0); + }); + + it("calls import with dry_run: false when Apply is pressed", async () => { + importProgress.mockResolvedValue(dryRunResponse()); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + + const applyButton = await screen.findByRole("button", { + name: /apply import/i, + }); + await waitFor(() => expect(applyButton).toBeEnabled()); + + importProgress.mockResolvedValue({ + ...dryRunResponse(), + dry_run: false, + series: [ + { + ...dryRunResponse().series[0], + committed: true, + }, + ], + }); + await user.click(applyButton); + + await waitFor(() => { + expect(importProgress).toHaveBeenLastCalledWith( + expect.objectContaining({ dry_run: false }), + ); + }); + }); + + /// Options cannot change under a dry run in flight, or its result would + /// unlock Apply for options it never previewed. + it("locks the options while a dry run is in flight", async () => { + let finish: (value: ImportReadingProgressResponse) => void = () => {}; + importProgress.mockReturnValue( + new Promise((resolve) => { + finish = resolve; + }), + ); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + + await waitFor(() => + expect( + screen.getByRole("checkbox", { name: /accept filename-stem matches/i }), + ).toBeDisabled(), + ); + + finish(dryRunResponse()); + await waitFor(() => + expect( + screen.getByRole("checkbox", { name: /accept filename-stem matches/i }), + ).toBeEnabled(), + ); + expect(screen.getByRole("button", { name: /apply import/i })).toBeEnabled(); + }); + + /// A failed apply may have committed some series, so the old preview no + /// longer describes what a second apply would do. + it("relocks Apply after an apply fails", async () => { + importProgress.mockResolvedValue(dryRunResponse()); + const user = userEvent.setup(); + renderWithProviders(); + + await uploadDocument(user); + await user.click( + screen.getByRole("button", { name: /preview \(dry run\)/i }), + ); + const applyButton = await screen.findByRole("button", { + name: /apply import/i, + }); + await waitFor(() => expect(applyButton).toBeEnabled()); + + importProgress.mockRejectedValue( + Object.assign(new Error("Request failed"), { + response: { status: 413 }, + }), + ); + await user.click(applyButton); + + expect( + await screen.findByText(/larger than the server accepts/i), + ).toBeInTheDocument(); + expect( + screen.getByRole("button", { name: /apply import/i }), + ).toBeDisabled(); + }); +}); diff --git a/web/src/pages/settings/ReadingProgressTransferSettings.tsx b/web/src/pages/settings/ReadingProgressTransferSettings.tsx new file mode 100644 index 000000000..e04d836a4 --- /dev/null +++ b/web/src/pages/settings/ReadingProgressTransferSettings.tsx @@ -0,0 +1,482 @@ +import { + Alert, + Badge, + Button, + Card, + Checkbox, + Divider, + FileButton, + Group, + Select, + Stack, + Table, + Text, + Title, +} from "@mantine/core"; +import { + IconAlertTriangle, + IconCheck, + IconDownload, + IconUpload, +} from "@tabler/icons-react"; +import { useRef, useState } from "react"; +import type { + BookDisposition, + ConflictPolicy, + HashMode, + ImportReadingProgressResponse, + ReadingProgressExportDocument, + SeriesDisposition, +} from "@/api/readingProgressTransfer"; +import { + useExportReadingProgress, + useImportReadingProgress, +} from "@/hooks/useReadingProgressTransfer"; + +type ApiErrorLike = Error & { + response?: { + status?: number; + data?: { message?: string; error?: string }; + }; +}; + +function errorMessage(error: unknown, fallback: string): string { + const err = error as ApiErrorLike; + if (err?.response?.status === 413) { + return "This file is larger than the server accepts for an import (64 MB)."; + } + return ( + err?.response?.data?.message || + err?.response?.data?.error || + err?.message || + fallback + ); +} + +/** + * Read a `File` as text via `FileReader` rather than `Blob.text()`: broader + * runtime support (older WebViews, some test environments) for what is + * otherwise a one-line read. + */ +function readFileAsText(file: File): Promise { + return new Promise((resolve, reject) => { + const reader = new FileReader(); + reader.onload = () => resolve(String(reader.result ?? "")); + reader.onerror = () => + reject(reader.error ?? new Error("Failed to read file")); + reader.readAsText(file); + }); +} + +function seriesDispositionColor(disposition: SeriesDisposition): string { + switch (disposition) { + case "matched": + return "green"; + case "ambiguous": + return "yellow"; + default: + return "gray"; + } +} + +function bookDispositionColor(disposition: BookDisposition): string { + switch (disposition) { + case "matched": + return "green"; + case "stem_match": + return "blue"; + case "ambiguous": + return "yellow"; + case "hash_mismatch": + return "red"; + default: + return "gray"; + } +} + +function SummaryLine({ report }: { report: ImportReadingProgressResponse }) { + const { summary } = report; + return ( + + + Series: {summary.series_matched} matched,{" "} + {summary.series_ambiguous} ambiguous,{" "} + {summary.series_unmatched} unmatched + {!report.dry_run && ( + <> + , {summary.series_committed} committed + + )} + + + Books: {summary.books_matched} matched,{" "} + {summary.books_stem_matched} stem match,{" "} + {summary.books_ambiguous} ambiguous,{" "} + {summary.books_unmatched} unmatched,{" "} + {summary.books_hash_mismatch} hash mismatch + + + Progress written: {summary.progress_written} · + Completions: {summary.completions_inserted} inserted /{" "} + {summary.completions_reattached} reattached · Sessions:{" "} + {summary.sessions_inserted} inserted /{" "} + {summary.sessions_reattached} reattached · Ratings:{" "} + {summary.ratings_written} + + + ); +} + +function ReportTable({ report }: { report: ImportReadingProgressResponse }) { + return ( + + + + + Series + Disposition + Committed + Books + + + + {report.series.map((series, index) => ( + // A split exports same-named series from several libraries, so + // path and name alone are not unique. + + + + {series.name} + + + {series.library_relative_path} + + {series.error && ( + + {series.error} + + )} + + + + {series.disposition} + + + + {series.attempted ? ( + series.committed ? ( + + ) : ( + + no + + ) + ) : ( + + – + + )} + + + + {series.books.map((book) => ( + + {book.file_name} + + ))} + + + + ))} + +
+
+ ); +} + +export function ReadingProgressTransferSettings() { + const [includeSessions, setIncludeSessions] = useState(true); + + const [file, setFile] = useState(null); + const [parsedDocument, setParsedDocument] = + useState(null); + const [parseError, setParseError] = useState(null); + + const [conflictPolicy, setConflictPolicy] = + useState("newest"); + const [hashMode, setHashMode] = useState("verify"); + const [reattachSessions, setReattachSessions] = useState(true); + const [acceptStemMatches, setAcceptStemMatches] = useState(false); + + const [report, setReport] = useState( + null, + ); + // Every change to the file or an option starts a new generation. A preview + // unlocks Apply only for the generation it was run against, so a dry run + // still in flight when an option changes cannot unlock Apply for options + // it never previewed. + const generation = useRef(0); + const [previewedGeneration, setPreviewedGeneration] = useState( + null, + ); + const [importError, setImportError] = useState(null); + + const exportMutation = useExportReadingProgress(); + const importMutation = useImportReadingProgress(); + + const clearPreview = () => { + generation.current += 1; + setReport(null); + setPreviewedGeneration(null); + setImportError(null); + }; + + const handleFile = async (selected: File | null) => { + setFile(selected); + setParsedDocument(null); + setParseError(null); + clearPreview(); + if (!selected) return; + try { + const text = await readFileAsText(selected); + setParsedDocument(JSON.parse(text) as ReadingProgressExportDocument); + } catch { + setParseError("This file is not valid JSON."); + } + }; + + const runImport = (dryRun: boolean) => { + if (!parsedDocument) return; + const requestedFor = generation.current; + setImportError(null); + importMutation.mutate( + { + dry_run: dryRun, + hash_mode: hashMode, + source_preference: [], + conflict_policy: conflictPolicy, + reattach_sessions: reattachSessions, + accept_stem_matches: acceptStemMatches, + file: parsedDocument, + }, + { + onSuccess: (response) => { + if (requestedFor !== generation.current) return; + setReport(response); + setPreviewedGeneration(dryRun ? requestedFor : null); + }, + onError: (error) => { + if (requestedFor !== generation.current) return; + setImportError(errorMessage(error, "Import failed.")); + setReport(null); + // A failed apply may have committed some series, so the old + // preview no longer describes what applying would do. + setPreviewedGeneration(null); + }, + }, + ); + }; + + const previewed = + previewedGeneration !== null && previewedGeneration === generation.current; + const canApply = + Boolean(parsedDocument) && previewed && !importMutation.isPending; + const busy = importMutation.isPending; + + return ( + +
+ Reading Progress + + Export your reading progress, completions, sessions, and ratings to a + file, and import it back after reorganising or moving your library. + +
+ + } + title="Export before you delete" + > + Take an export before deleting an old library during a split or + migration. History survives a hard delete as orphaned rows, but nothing + except an export taken beforehand can say which book an orphan used to + belong to. + + + + + Export + + setIncludeSessions(event.currentTarget.checked) + } + /> + + + + + + + + + Import + + + + {(props) => ( + + )} + + + + {parseError && ( + }> + {parseError} + + )} + + + + + { + if (value) setHashMode(value as HashMode); + clearPreview(); + }} + /> + + + { + setReattachSessions(event.currentTarget.checked); + clearPreview(); + }} + /> + { + setAcceptStemMatches(event.currentTarget.checked); + clearPreview(); + }} + /> + + {importError && ( + } + title="Import failed" + > + {importError} + + )} + + + + + + + {report && ( + + + {!report.sessions_in_file && ( + + This file does not include sessions. + + )} + {(report.notices ?? []).map((notice) => ( + + {notice} + + ))} + + + + )} + + +
+ ); +} diff --git a/web/src/pages/settings/index.ts b/web/src/pages/settings/index.ts index 6d443aeae..b716909c4 100644 --- a/web/src/pages/settings/index.ts +++ b/web/src/pages/settings/index.ts @@ -10,6 +10,7 @@ export { PdfCacheSettings } from "./PdfCacheSettings"; export { PluginStorageSettings } from "./PluginStorageSettings"; export { PluginsSettings } from "./PluginsSettings"; export { ProfileSettings } from "./ProfileSettings"; +export { ReadingProgressTransferSettings } from "./ReadingProgressTransferSettings"; export { ReleaseTrackingSettings } from "./ReleaseTrackingSettings"; export { SeriesExportsSettings } from "./SeriesExportsSettings"; export { ServerSettings } from "./ServerSettings"; diff --git a/web/src/types/api.generated.ts b/web/src/types/api.generated.ts index 00df781f0..6f6d83db9 100644 --- a/web/src/types/api.generated.ts +++ b/web/src/types/api.generated.ts @@ -3276,6 +3276,70 @@ export interface paths { patch?: never; trace?: never; }; + "/api/v1/reading-progress/export": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** + * Export the authenticated user's reading progress + * @description Produces the whole `read_progress` / `read_completions` / `reading_sessions` + * / `user_series_ratings` state for the caller, keyed by external ids and + * series-relative paths instead of database ids so it can be matched back + * against a differently organised library (or a different Codex instance + * entirely) by `POST /api/v1/reading-progress/import`. + * + * Take this export **before** deleting an old library during a split: the + * underlying rows survive a hard delete as orphans, but nothing except this + * file can say which book an orphan used to belong to. + */ + get: operations["export_reading_progress"]; + put?: never; + post?: never; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/api/v1/reading-progress/import": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** + * Import reading progress from an export document + * @description Resolves the file's series and books against the current library through + * the same access-group / sharing-tag visibility that an ordinary read + * respects: a book the importing user cannot see resolves as `unmatched`, + * never as a permission error that would confirm it exists, and nothing is + * ever written against it. + * + * Matching never guesses: any step (external id, path, file name, or, under + * `hash_mode = "match"`, hash) that finds more than one candidate reports + * `ambiguous` and writes nothing for that series or book. `dry_run: true` + * returns the identical response shape without writing anything, which is + * what makes it safe to preview before committing. + * + * Each series is applied in its own transaction; the response's per-series + * `committed` field says which ones actually landed. `read_completions` and + * `reading_sessions` reuse their exported ids on insert, so importing the + * same file twice leaves row counts unchanged rather than duplicating + * history. + */ + post: operations["import_reading_progress"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/api/v1/reading-sessions": { parameters: { query?: never; @@ -3382,8 +3446,10 @@ export interface paths { * the series and format breakdowns show it as one "removed from library" row. * This discards those rows for the caller, and only for the caller. * - * Irreversible. Only history already detached from any book is touched; - * attributed reading is never affected. + * Irreversible, and it forecloses the other way out: importing a reading + * progress export taken before the delete puts those sessions back on their + * books once the files are scanned again. Only history already detached from + * any book is touched; attributed reading is never affected. */ delete: operations["purge_orphaned_reading_history"]; options?: never; @@ -8935,6 +9001,12 @@ export interface components { /** @description Optional metadata from ComicInfo.xml or similar */ metadata?: components["schemas"]["BookMetadataDto"]; }; + /** + * @description Why a book in the import file could not be resolved to exactly one book + * in its matched series. + * @enum {string} + */ + BookDisposition: "matched" | "stem_match" | "ambiguous" | "unmatched" | "hash_mismatch"; /** @description Book data transfer object */ BookDto: { /** @@ -11103,6 +11175,11 @@ export interface components { /** @description Number of settings configured */ settingsConfigured: number; }; + /** + * @description How a conflict between an imported value and an existing row is resolved. + * @enum {string} + */ + ConflictPolicy: "newest" | "furthest" | "skip_existing" | "overwrite"; /** @description Contributor information (author, artist, etc.) */ Contributor: { /** @description Name of the contributor */ @@ -12207,6 +12284,58 @@ export interface components { /** @description Whether the execution succeeded */ success: boolean; }; + /** + * @description One book inside a series, keyed for matching by path, name, and hash + * rather than by id: the whole point of the file is that ids on the far side + * are expected to be different. + */ + ExportBookDto: { + completions?: components["schemas"]["ExportCompletionDto"][]; + /** + * @description Empty when the book was never analyzed; never treated as a value to + * match on in that case. + */ + file_hash?: string; + /** @example v01.cbz */ + file_name: string; + partial_hash?: string; + /** + * @description Relative to the series folder, so a series move does not invalidate it. + * @example Vol 01/v01.cbz + */ + path: string; + progress?: components["schemas"]["ExportProgressDto"]; + /** + * @description Omitted entirely (not an empty array) when the export was taken with + * `include_sessions=false`. + */ + sessions?: components["schemas"]["ExportSessionDto"][] | null; + }; + /** + * @description One finished read-through. Keeps its original id so re-importing the same + * file is a no-op rather than a duplicate. + */ + ExportCompletionDto: { + /** Format: date-time */ + completed_at: string; + /** Format: uuid */ + id: string; + /** Format: date-time */ + started_at: string; + }; + /** + * @description One external identifier attached to a series (a plugin match, a ComicInfo + * value, or a manual entry). + */ + ExportExternalIdDto: { + /** @example 12345 */ + id: string; + /** + * @description `plugin:`, `comicinfo`, `epub`, or `manual`. + * @example plugin:mangabaka + */ + source: string; + }; /** @description Response for the field catalog */ ExportFieldCatalogResponse: { /** @description Book export fields */ @@ -12231,6 +12360,92 @@ export interface components { /** @description LLM-friendly book field preset */ llmSelectBooks: string[]; }; + /** + * @description The live resume position for one book. Retains `r2_progression`: it is the + * only place the EPUB locator survives, since sessions strip it. + */ + ExportProgressDto: { + completed: boolean; + /** Format: date-time */ + completed_at?: string | null; + /** Format: int32 */ + current_page: number; + /** Format: double */ + progress_percentage?: number | null; + r2_progression?: string | null; + /** Format: date-time */ + started_at: string; + /** Format: date-time */ + updated_at: string; + }; + /** @description Query parameters for `GET /api/v1/reading-progress/export`. */ + ExportReadingProgressQuery: { + /** + * @description Sessions are opt-out: they are the only source of every reading + * statistic, so leaving them out is easy to do by accident and hard to + * notice until the numbers are gone. + */ + include_sessions?: boolean; + }; + /** @description One series and everything the exporting user recorded against its books. */ + ExportSeriesDto: { + books?: components["schemas"]["ExportBookDto"][]; + external_ids?: components["schemas"]["ExportExternalIdDto"][]; + /** + * @description The series path as stored, relative to the library root. + * @example shonen/Naruto + */ + library_relative_path: string; + /** @example Naruto */ + name: string; + notes?: string | null; + /** Format: int32 */ + rating?: number | null; + /** + * Format: date-time + * @description When the rating was last changed. The `newest` conflict policy needs it + * to tell a stale rating from a fresh one; a file without it never + * overwrites an existing rating except under `overwrite`. + */ + rating_updated_at?: string | null; + }; + /** + * @description One row from the reading-session log. `r2_progression` is deliberately + * absent: nothing reads a session's historical locator, and it is the only + * non-scalar column on the row. + */ + ExportSessionDto: { + /** Format: int64 */ + active_duration_ms?: number | null; + /** Format: date-time */ + client_ended_at: string; + /** Format: date-time */ + client_started_at: string; + device_id: string; + device_name?: string | null; + /** + * @description `"measured"`, `"inferred"`, or `"unknown"`. + * @example measured + */ + duration_source: string; + /** Format: uuid */ + id: string; + /** + * @description `"progress"`, `"completed"`, or `"reset"`. + * @example progress + */ + kind: string; + /** Format: int32 */ + pages_read?: number | null; + /** Format: int32 */ + pass: number; + /** Format: date-time */ + server_recorded_at: string; + /** Format: int32 */ + to_page?: number | null; + /** Format: double */ + to_percentage?: number | null; + }; /** * @description External ID context for template evaluation. * @@ -12480,6 +12695,12 @@ export interface components { operator: "endsWith"; value: string; }; + /** + * @description What happened to one scalar field write (progress or rating) under the + * active conflict policy. + * @enum {string} + */ + FieldOutcome: "inserted" | "updated" | "skipped"; /** * @example { * "isDirectory": true, @@ -13023,6 +13244,11 @@ export interface components { /** @description Publications within this group */ publications?: components["schemas"]["Publication"][] | null; }; + /** + * @description How aggressively hashes are used to match a book. + * @enum {string} + */ + HashMode: "off" | "verify" | "match"; /** * @description Image link with optional dimensions * @@ -13044,6 +13270,123 @@ export interface components { */ width?: number | null; }; + /** @description The outcome for one book in the import file. */ + ImportBookReport: { + /** + * @description Whether any writes were attempted for this book. False for every + * disposition except `matched`, and except `stem_match` when + * `accept_stem_matches` is off. + */ + applied: boolean; + completions: components["schemas"]["WriteCounts"]; + disposition: components["schemas"]["BookDisposition"]; + file_name: string; + /** Format: uuid */ + matched_book_id?: string | null; + path: string; + progress?: components["schemas"]["FieldOutcome"]; + sessions: components["schemas"]["WriteCounts"]; + }; + /** @description `POST /api/v1/reading-progress/import` request body. */ + ImportReadingProgressRequest: { + /** + * @description A stem match (`v01.cbr` renamed to `v01.cbz`) is reported either way, + * but only written when this is set: two files can share a stem, and + * applying it silently risks writing progress onto the wrong one. + */ + accept_stem_matches?: boolean; + conflict_policy?: components["schemas"]["ConflictPolicy"]; + /** @description Compute and report the outcome without writing anything. */ + dry_run?: boolean; + file: components["schemas"]["ReadingProgressExportDocument"]; + hash_mode?: components["schemas"]["HashMode"]; + /** + * @description When a session or completion in the file already exists as the + * importer's own row but is not on a live book (its book was hard-deleted, + * leaving `book_id` null, or the scanner marked it deleted after the file + * moved), move it onto the matched book instead of skipping it. + */ + reattach_sessions?: boolean; + /** + * @description External-id sources to try, in order, before falling back to path and + * then normalized name. An empty list skips straight to path matching. + */ + source_preference?: string[]; + }; + /** + * @description The response for both a real import and a dry run: the shape is identical + * either way, so a client cannot tell from the response alone whether + * anything was written. Only `dry_run` (and the DB) says that. + */ + ImportReadingProgressResponse: { + dry_run: boolean; + /** + * @description Informational notes about the request, e.g. `reattach_sessions` having + * no effect because the file carries no sessions. + */ + notices?: string[]; + series: components["schemas"]["ImportSeriesReport"][]; + sessions_in_file: boolean; + summary: components["schemas"]["ImportSummary"]; + }; + /** @description The outcome for one series in the import file. */ + ImportSeriesReport: { + /** + * @description Whether book/rating processing was attempted at all (only when + * `disposition == matched`). + */ + attempted: boolean; + books: components["schemas"]["ImportBookReport"][]; + /** + * @description Whether this series' transaction was committed. Always `false` in a + * dry run, and `false` if `attempted` but a write failed. + */ + committed: boolean; + disposition: components["schemas"]["SeriesDisposition"]; + error?: string | null; + library_relative_path: string; + /** Format: uuid */ + matched_series_id?: string | null; + name: string; + rating?: components["schemas"]["FieldOutcome"]; + }; + /** @description Totals across every series in the file, for a one-line summary. */ + ImportSummary: { + /** Format: int32 */ + books_ambiguous: number; + /** Format: int32 */ + books_hash_mismatch: number; + /** Format: int32 */ + books_matched: number; + /** Format: int32 */ + books_stem_matched: number; + /** Format: int32 */ + books_total: number; + /** Format: int32 */ + books_unmatched: number; + /** Format: int32 */ + completions_inserted: number; + /** Format: int32 */ + completions_reattached: number; + /** Format: int32 */ + progress_written: number; + /** Format: int32 */ + ratings_written: number; + /** Format: int32 */ + series_ambiguous: number; + /** Format: int32 */ + series_committed: number; + /** Format: int32 */ + series_matched: number; + /** Format: int32 */ + series_total: number; + /** Format: int32 */ + series_unmatched: number; + /** Format: int32 */ + sessions_inserted: number; + /** Format: int32 */ + sessions_reattached: number; + }; /** * @description Which layer supplied a value the user is inheriting. * @enum {string} @@ -17091,6 +17434,23 @@ export interface components { */ totalPages: number; }; + /** + * @description The whole export: one user's reading state, self-describing enough to be + * matched back against a differently organised library. + */ + ReadingProgressExportDocument: { + /** Format: date-time */ + exported_at: string; + /** @example codex-reading-progress */ + format: string; + includes_sessions: boolean; + series?: components["schemas"]["ExportSeriesDto"][]; + /** + * Format: int32 + * @example 1 + */ + version: number; + }; /** @description One reading session measured by a client. */ ReadingSessionDto: { /** @@ -18591,6 +18951,12 @@ export interface components { /** @description List of covers */ covers: components["schemas"]["SeriesCoverDto"][]; }; + /** + * @description Why a series in the import file could not be resolved to exactly one + * series in the current library. + * @enum {string} + */ + SeriesDisposition: "matched" | "ambiguous" | "unmatched"; /** @description Series data transfer object */ SeriesDto: { /** @@ -21598,6 +21964,25 @@ export interface components { */ urlTemplate: string; }; + /** + * @description Insert/reattach/skip counts for an append-only table (completions or + * sessions) within one book. + */ + WriteCounts: { + /** Format: int32 */ + inserted: number; + /** + * Format: int32 + * @description Adopted an orphaned row (`book_id IS NULL`) rather than inserting a + * new one. + */ + reattached: number; + /** + * Format: int32 + * @description Already present with the same book attached; re-importing is a no-op. + */ + skipped: number; + }; }; responses: never; parameters: never; @@ -29387,6 +29772,95 @@ export interface operations { }; }; }; + export_reading_progress: { + parameters: { + query?: { + /** @description Include the reading-session log (default: true). Sessions are the only source of every reading statistic, so this is opt-out rather than opt-in. */ + include_sessions?: boolean; + }; + header?: never; + path?: never; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description The export document */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["ReadingProgressExportDocument"]; + }; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + }; + }; + import_reading_progress: { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["ImportReadingProgressRequest"]; + }; + }; + responses: { + /** @description Import processed (or, for a dry run, previewed) */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["ImportReadingProgressResponse"]; + }; + }; + /** @description Unknown export format, a version newer than this server supports, or a value the normal write paths reject */ + 400: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Unauthorized */ + 401: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description Forbidden */ + 403: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + /** @description The file is larger than the import limit */ + 413: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + }; + }; record_reading_sessions: { parameters: { query?: never;