diff --git a/deno.lock b/deno.lock index 9eac1f2..f607a6a 100644 --- a/deno.lock +++ b/deno.lock @@ -11,6 +11,7 @@ "npm:@sveltejs/kit@^2.63.0": "2.69.2_@sveltejs+vite-plugin-svelte@7.2.0__svelte@5.56.4__vite@8.1.4___@types+node@26.1.1__@types+node@26.1.1_svelte@5.56.4_typescript@6.0.3_vite@8.1.4__@types+node@26.1.1_@types+node@26.1.1", "npm:@sveltejs/vite-plugin-svelte@^7.1.2": "7.2.0_svelte@5.56.4_vite@8.1.4__@types+node@26.1.1_@types+node@26.1.1", "npm:@types/node@^26.1.1": "26.1.1", + "npm:csv-parse@^5.6.0": "5.6.0", "npm:poline@~0.13.1": "0.13.1", "npm:svelte-check@*": "4.7.2_svelte@5.56.4_typescript@6.0.3", "npm:svelte-check@^4.6.0": "4.7.2_svelte@5.56.4_typescript@6.0.3", @@ -19,6 +20,11 @@ "npm:vite@*": "8.1.4_@types+node@26.1.1", "npm:vite@^8.0.16": "8.1.4_@types+node@26.1.1" }, + "jsr": { + "@std/streams@1.0.17": { + "integrity": "7859f3d9deed83cf4b41f19223d4a67661b3d3819e9fc117698f493bf5992140" + } + }, "npm": { "@atproto-labs/did-resolver@0.3.5": { "integrity": "sha512-0dMM+hj40VQiD/EJhlC1UMQgPXRwKeqM7NgJte7fVuYMv5b3P0W6+Lu3iDumHULcSjMmMXxJZzoi3i493Y0gCA==", @@ -645,6 +651,9 @@ "integrity": "sha512-es1U2+YTtzpwkxVLwAFdSpaIMyQaq0PBgm3YD1W3Qpsn1NAmO3KSgZfu+oGSWVu6NvLHoHCV/aYcsE5wiB7ALg==", "scripts": true }, + "csv-parse@5.6.0": { + "integrity": "sha512-l3nz3euub2QMg5ouu5U09Ew9Wf6/wQ8I++ch1loQ0ljmzhmfZYrH9fflS22i/PQEvsPvxCwxgz5q7UB8K1JO4Q==" + }, "deepmerge@4.3.1": { "integrity": "sha512-3sUqbMEc77XqpdNO7FRyRog+eW3ph+GYCbj+rK+uYyRMuwsVy0rMiVtPn+QJlKFvWP/1PYpapqYn0Me2knFn+A==" }, @@ -1052,6 +1061,7 @@ "npm:@sveltejs/kit@^2.63.0", "npm:@sveltejs/vite-plugin-svelte@^7.1.2", "npm:@types/node@^26.1.1", + "npm:csv-parse@^5.6.0", "npm:poline@~0.13.1", "npm:svelte-check@^4.6.0", "npm:svelte@^5.56.1", diff --git a/migrations/005_csv_import.sql b/migrations/005_csv_import.sql new file mode 100644 index 0000000..b70aedb --- /dev/null +++ b/migrations/005_csv_import.sql @@ -0,0 +1,28 @@ +-- CSV backfill import (csv-backfill-import). A connected account whose SimpleFIN +-- feed reaches back only a week has a permanent hole in its past that no sync can +-- fill; the bank publishes that history as a CSV. Imports are additive only: they +-- attach to an account sync already discovered, and never reconcile, sweep, or +-- overwrite anything sync owns. + +-- One row per uploaded file. `payload` is verbatim and never mutated, mirroring +-- raw_syncs. `mapping` and `decisions` live here too because the outcome of an +-- import is a function of the bytes AND the human's choices — bytes alone would +-- not replay deterministically. +CREATE TABLE imports ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + account_id TEXT NOT NULL REFERENCES accounts (id), + filename TEXT, + uploaded_at TEXT NOT NULL, + payload TEXT NOT NULL, -- verbatim uploaded bytes; never mutated + mapping TEXT, -- JSON: field -> column, chosen by the user + decisions TEXT, -- JSON: synthetic id -> 'keep' | 'skip' + status TEXT NOT NULL DEFAULT 'draft' CHECK (status IN ('draft', 'committed', 'undone')), + committed_at TEXT, + undone_at TEXT +); + +-- NULL means synced, so every pre-existing row is already correct and no backfill +-- is needed. Carries origin for the ledger source filter and scopes undo. +ALTER TABLE transactions ADD COLUMN import_id INTEGER REFERENCES imports (id); + +CREATE INDEX idx_transactions_import ON transactions (import_id); diff --git a/openspec/changes/csv-backfill-import/design.md b/openspec/changes/csv-backfill-import/design.md index 38dc4dd..0ac7cf6 100644 --- a/openspec/changes/csv-backfill-import/design.md +++ b/openspec/changes/csv-backfill-import/design.md @@ -181,6 +181,16 @@ shape sync emits for posted rows, and every existing query behaves identically. The residual imprecision — the file's single date column may be the transaction date rather than the post date — is absorbed by the ±1 day duplicate window. +**Date order is part of the mapping.** `03/04/2026` is March 4th at one bank and +April 3rd at another, and the file never says which; guessing wrong silently shifts +an entire import by months. The format (`iso` / `mdy` / `dmy`) is therefore a +stored, user-confirmable field of the mapping, archived like every other mapping +choice so replay stays deterministic. It is inferred from the data where the data +settles it — any component above 12 can only be a day — and falls back to +month-first when every row is ambiguous. ISO dates are unambiguous and always +parse as themselves regardless of the declared format. The preview's stated date +range is the human backstop against a wrong inference. + ### 7. Amount parsing: a CSV pre-pass, not a change to `parseAmountToCents` **Decision:** Normalize CSV negative conventions (`(12.34)`, `12.34-`, currency @@ -272,10 +282,31 @@ forward-only and safe to re-run against a populated database. Rollback is droppi the table and column; imported rows would need removal first, which the undo flow already performs. +## Resolved During Implementation + +- **Hard-refuse implausible date ranges?** No. The preview states the parsed range + and shows the first rows as parsed, which makes a wrong date mapping obvious at a + glance, and undo makes a mistake cheap. A hard refusal would need a notion of + "the account's opening date" that does not exist, and would block legitimate + imports at the edges. Stating the range is enough. +- **List individual skipped rows in the import record?** No — counts only. The + skipped rows are, by construction, transactions the ledger already contains; the + interesting artifact is the archived file, which is retained. +- **Date-only values are anchored at noon UTC, not midnight.** Found by running a + real file: `formatDay` renders in local time, so UTC-midnight rows displayed a day + early across the Americas, and a month's first day would have displayed in the + previous month while still reporting under the correct one. Noon holds the + calendar day for UTC-12..+11 — every timezone these two users will see. + (`formatMonth` already anchors to the 15th for the same reason.) Not universal: + UTC+12 and beyond would still shift. Documented in the tests; the honest fix, if + it ever matters, is rendering date-only rows in UTC. +- **The duplicate window counts whole UTC calendar days, not ±86400 seconds.** A + consequence of the above: a CSV row sits at noon while a synced row sits at + whatever hour the bank posted it, so a seconds-based window would mean "within 24 + hours" and would miss rows one calendar day apart (noon vs. midnight is 36 hours). + The spec says "within one day" and now the code means it. + ## Open Questions -- Should the preview step hard-refuse a file whose parsed range extends into the - future or predates the account's opening, as a cheap guard against a - catastrophically wrong date mapping — or is stating the range enough? -- Does the import record in Settings need to list the individual skipped rows, or - is a count sufficient? +- Abandoned drafts are visible in the import record as "not finished". If they + accumulate enough to be noise, they want either a reaper or a hide. diff --git a/openspec/changes/csv-backfill-import/specs/csv-import/spec.md b/openspec/changes/csv-backfill-import/specs/csv-import/spec.md index 484d04d..4f16bbb 100644 --- a/openspec/changes/csv-backfill-import/specs/csv-import/spec.md +++ b/openspec/changes/csv-backfill-import/specs/csv-import/spec.md @@ -50,10 +50,13 @@ outcome of an import is a deterministic function of the archived record alone. The system SHALL detect a CSV's date, amount, and description columns from its header row where possible, and SHALL require the user to confirm or correct the mapping before commit. The system SHALL support amounts expressed as a single -signed column and as separate debit and credit columns. The system SHALL accept -the negative conventions common to bank exports, including parenthesized values -and trailing minus signs, and SHALL convert amounts to integer cents without -floating-point arithmetic. The system SHALL map the file's date to the +signed column and as separate debit and credit columns. The system SHALL treat the +interpretation of ambiguous numeric dates (whether `03/04/2026` is March 4th or +April 3rd) as a user-confirmable part of the mapping, inferring it from the data +where the data settles it and defaulting to month-first otherwise. The system +SHALL accept the negative conventions common to bank exports, including +parenthesized values and trailing minus signs, and SHALL convert amounts to +integer cents without floating-point arithmetic. The system SHALL map the file's date to the transaction's posted timestamp and record imported transactions as not pending; when the file exposes a distinct transaction date, the system SHALL map it to the transaction date. @@ -69,6 +72,18 @@ transaction date. - **WHEN** the auto-detected mapping is wrong - **THEN** the user can reassign any field to any column before committing +#### Scenario: Date order inferred from the data + +- **WHEN** a file's date column contains `13/04/2026`, which can only be + day-first +- **THEN** the whole file is interpreted day-first + +#### Scenario: Ambiguous date order is confirmable + +- **WHEN** every date in a file is ambiguous, such as `03/04/2026` +- **THEN** the system defaults to month-first, states the resulting date range + before commit, and allows the user to select day-first instead + #### Scenario: Parenthesized negative - **WHEN** an amount column contains `(12.34)` diff --git a/openspec/changes/csv-backfill-import/tasks.md b/openspec/changes/csv-backfill-import/tasks.md index fff8033..b3e69d0 100644 --- a/openspec/changes/csv-backfill-import/tasks.md +++ b/openspec/changes/csv-backfill-import/tasks.md @@ -1,77 +1,80 @@ ## 1. Schema -- [ ] 1.1 Add `migrations/005_csv_import.sql` creating the `imports` table: `id`, +- [x] 1.1 Add `migrations/005_csv_import.sql` creating the `imports` table: `id`, `account_id` (NOT NULL, FK to accounts), `filename`, `uploaded_at`, `payload` (verbatim bytes, never mutated), `mapping` (JSON, nullable), `decisions` (JSON, nullable), `status` NOT NULL DEFAULT `'draft'` CHECK in (`draft`, `committed`, `undone`), `committed_at`, `undone_at`. -- [ ] 1.2 In the same migration, `ALTER TABLE transactions ADD COLUMN import_id +- [x] 1.2 In the same migration, `ALTER TABLE transactions ADD COLUMN import_id INTEGER REFERENCES imports (id)` (nullable; NULL means synced) and add an index on `transactions (import_id)`. -- [ ] 1.3 Extend `src/lib/server/db.test.ts` to assert the migration applies to a +- [x] 1.3 Extend `src/lib/server/db.test.ts` to assert the migration applies to a populated database and that existing transactions read back with `import_id IS NULL`. ## 2. CSV parsing and mapping (pure) -- [ ] 2.1 Add `@std/csv` (JSR) as a dependency. Do not hand-roll parsing — - quoted delimiters, embedded newlines, and BOMs are the failure modes. -- [ ] 2.2 Create `src/lib/server/services/csv-import.ts` with a pure +- [x] 2.1 Add `csv-parse` (npm) as a dependency. Do not hand-roll parsing — + quoted delimiters, embedded newlines, and BOMs are the failure modes. An npm + package rather than JSR's `@std/csv`: Vite-under-Deno resolves a JSR specifier + fine, but `svelte-check` resolves Node-style and cannot, which would have left + `deno task check` permanently red. +- [x] 2.2 Create `src/lib/server/services/csv-import.ts` with a pure `parseCsv(payloadText)` returning header row plus data rows, tolerating a BOM and CRLF line endings. -- [ ] 2.3 Implement pure `detectMapping(headers)` returning best-guess column +- [x] 2.3 Implement pure `detectMapping(headers)` returning best-guess column assignments for date, amount (single signed column or debit/credit pair), and description, plus optional payee/memo and a distinct transaction-date column. -- [ ] 2.4 Implement pure `normalizeCsvAmount(raw)` handling parenthesized +- [x] 2.4 Implement pure `normalizeCsvAmount(raw)` handling parenthesized negatives `(12.34)`, trailing minus `12.34-`, currency symbols, and thousands separators, emitting a signed decimal string for the existing `parseAmountToCents`. Do not loosen `parseAmountToCents` itself — its strict regex is the sync path's contract. -- [ ] 2.5 Implement pure `normalizeCsv(payloadText, mapping, accountId)` producing +- [x] 2.5 Implement pure `normalizeCsv(payloadText, mapping, accountId)` producing candidate rows with `posted` set from the file's date, `pending = 0`, `transacted_at` set only when the mapping names a distinct transaction-date column, and amounts as integer cents. -- [ ] 2.6 Implement the synthetic identity: `csv:` + hash over (accountId, date, +- [x] 2.6 Implement the synthetic identity: `csv:` + hash over (accountId, date, amountCents, description) plus an occurrence index among identical rows within the file. -- [ ] 2.7 Unit-test 2.2–2.6 in `csv-import.test.ts`: BOM/CRLF, both amount modes, +- [x] 2.7 Unit-test 2.2–2.6 in `csv-import.test.ts`: BOM/CRLF, both amount modes, each negative convention, date parsing, and that two identical rows receive distinct ids while a reordered or superset file re-derives the same id set. ## 3. Duplicate detection -- [ ] 3.1 Implement `findPotentialDuplicates(db, accountId, candidates)` matching +- [x] 3.1 Implement `findPotentialDuplicates(db, accountId, candidates)` matching each candidate against non-removed transactions in the account by identical `amount_cents` and effective date within ±1 day (compare against `COALESCE(posted, transacted_at)`). Never match on description. -- [ ] 3.2 Return, per candidate, one of: `new`, `already-present` (its synthetic +- [x] 3.2 Return, per candidate, one of: `new`, `already-present` (its synthetic id already exists — never surfaced to the user), or `flagged` with the existing transaction's date, amount, and description for side-by-side display. -- [ ] 3.3 Test: a cross-source duplicate with differently worded descriptions is +- [x] 3.3 Test: a cross-source duplicate with differently worded descriptions is flagged; a row one day off is flagged; a row two days off is not; an already-ingested row reports `already-present` rather than `flagged`. ## 4. Import lifecycle and additive ingestion -- [ ] 4.1 Implement `createDraftImport(db, accountId, filename, payload)` writing +- [x] 4.1 Implement `createDraftImport(db, accountId, filename, payload)` writing the `imports` row with verbatim payload and `status = 'draft'` before any parsing. -- [ ] 4.2 Implement `saveMapping` and `saveDecisions` persisting to the draft row, +- [x] 4.2 Implement `saveMapping` and `saveDecisions` persisting to the draft row, so the archived record alone determines the outcome. -- [ ] 4.3 Implement `commitImport(db, importId)` — an **insert-only** writer. +- [x] 4.3 Implement `commitImport(db, importId)` — an **insert-only** writer. It MUST NOT reconcile, MUST NOT sweep stale pending rows, MUST NOT update existing transactions, MUST NOT write balance snapshots, and MUST NOT touch account state. Do not call or extend `ingestTransactions`; its stale-pending sweep (`sync.ts:291-301`) would soft-delete every pending row absent from the CSV. -- [ ] 4.4 In `commitImport`, skip candidates marked skipped and those whose +- [x] 4.4 In `commitImport`, skip candidates marked skipped and those whose synthetic id already exists; revive a soft-removed row matching the unique key by clearing `removed_at` and reassigning `import_id`; stamp `import_id` on every inserted row; set `status = 'committed'` and `committed_at`. Run the whole commit in one transaction. -- [ ] 4.5 Call `applyRulesToUncategorized(db)` after the commit transaction +- [x] 4.5 Call `applyRulesToUncategorized(db)` after the commit transaction succeeds. No new categorization event source. -- [ ] 4.6 Test: importing into an account holding pending rows leaves them +- [x] 4.6 Test: importing into an account holding pending rows leaves them untouched; no snapshots are written; account state and `last_successful_data_at` are unchanged; re-importing the same file inserts nothing; an overlapping file inserts only the new rows; committed rows carry @@ -79,63 +82,63 @@ ## 5. Undo -- [ ] 5.1 Implement `undoImport(db, importId)` soft-removing (`removed_at = now`) +- [x] 5.1 Implement `undoImport(db, importId)` soft-removing (`removed_at = now`) every transaction with that `import_id`, setting `status = 'undone'` and `undone_at`, in one transaction. Never hard-delete — categorization events reference `transaction_id` and the event log is append-only. -- [ ] 5.2 Test: undo removes exactly that import's rows from ledger and report +- [x] 5.2 Test: undo removes exactly that import's rows from ledger and report scope, leaves synced rows untouched, preserves categorization events for the removed rows, and a corrected re-import afterwards does not resurrect the old rows. ## 6. Ledger surfacing -- [ ] 6.1 Add `source?: 'synced' | 'imported'` to `LedgerFilters` in +- [x] 6.1 Add `source?: 'synced' | 'imported'` to `LedgerFilters` in `src/lib/server/services/ledger.ts`, translating to `t.import_id IS NULL` / `IS NOT NULL`, composing with every existing filter. -- [ ] 6.2 Add origin to `LedgerRow` (import id, file name, import date; null when +- [x] 6.2 Add origin to `LedgerRow` (import id, file name, import date; null when synced) via a join through `import_id`. -- [ ] 6.3 Test in `ledger.test.ts`: the source filter composes with account, +- [x] 6.3 Test in `ledger.test.ts`: the source filter composes with account, month, category, pending, and text search. -- [ ] 6.4 Add the source filter control to the ledger page alongside the existing +- [x] 6.4 Add the source filter control to the ledger page alongside the existing filters. Add no per-row badge — within a backfilled range every row is imported, so a per-row mark carries no information where it appears. -- [ ] 6.5 Add an origin line to the top of the existing history panel: "Imported +- [x] 6.5 Add an origin line to the top of the existing history panel: "Imported from `` · ``" or "Synced from SimpleFIN". ## 7. Import wizard (Settings) -- [ ] 7.1 Add a Settings route for the import wizard. Step 1: file upload plus +- [x] 7.1 Add a Settings route for the import wizard. Step 1: file upload plus target-account picker (non-hidden accounts only), archiving on submit and redirecting to the draft's id. -- [ ] 7.2 Step 2: mapping confirmation, pre-filled from `detectMapping`, every +- [x] 7.2 Step 2: mapping confirmation, pre-filled from `detectMapping`, every field reassignable, with a few parsed sample rows rendered so a wrong guess is visible. -- [ ] 7.3 Step 3: preview stating the parsed date range, rows to import, rows +- [x] 7.3 Step 3: preview stating the parsed date range, rows to import, rows already present, and rows flagged — the guard against a catastrophically wrong date mapping. -- [ ] 7.4 Step 4: duplicate review listing each flagged row against its existing +- [x] 7.4 Step 4: duplicate review listing each flagged row against its existing match, both descriptions shown, defaulting to skip, each flippable to keep. -- [ ] 7.5 Commit action, then a result summary linking to the ledger filtered to +- [x] 7.5 Commit action, then a result summary linking to the ledger filtered to that account and source. -- [ ] 7.6 Style per DESIGN.md: tabular numerals on money and dates, calm empty and +- [x] 7.6 Style per DESIGN.md: tabular numerals on money and dates, calm empty and error states (one sentence plus one action), no new red unless data is at risk. ## 8. Import record (Settings) -- [ ] 8.1 Add a Settings section listing imports: file name, account, parsed date +- [x] 8.1 Add a Settings section listing imports: file name, account, parsed date range, rows imported, rows skipped, commit time, status. Not in the main nav. -- [ ] 8.2 Add the undo action with a confirmation stating exactly how many +- [x] 8.2 Add the undo action with a confirmation stating exactly how many transactions will be removed. -- [ ] 8.3 Render an empty state for the no-imports-yet case. +- [x] 8.3 Render an empty state for the no-imports-yet case. ## 9. Verification -- [ ] 9.1 Run `deno task test` and `deno task check`. -- [ ] 9.2 Drive the wizard end to end against a real bank CSV export in dev: +- [x] 9.1 Run `deno task test` and `deno task check`. +- [x] 9.2 Drive the wizard end to end against a real bank CSV export in dev: confirm the gap closes in the ledger, the month reports for backfilled months change as expected, and the net worth chart is unchanged (no snapshots written). -- [ ] 9.3 Verify a sync runs cleanly after an import: pending rows still +- [x] 9.3 Verify a sync runs cleanly after an import: pending rows still reconcile, and no imported row is disturbed. -- [ ] 9.4 Exercise undo on a real import and confirm the ledger and reports return +- [x] 9.4 Exercise undo on a real import and confirm the ledger and reports return to their pre-import state. diff --git a/package.json b/package.json index 45bd235..a5db315 100644 --- a/package.json +++ b/package.json @@ -16,7 +16,8 @@ "@atproto/jwk-jose": "^0.2.4", "@atproto/oauth-client-node": "^0.4.8", "@atproto/oauth-types": "^0.7.5", - "@fontsource-variable/inter": "^5.2.8" + "@fontsource-variable/inter": "^5.2.8", + "csv-parse": "^5.6.0" }, "devDependencies": { "@sveltejs/adapter-node": "^5.5.0", diff --git a/src/lib/server/db.test.ts b/src/lib/server/db.test.ts index 2e6376d..964c910 100644 --- a/src/lib/server/db.test.ts +++ b/src/lib/server/db.test.ts @@ -47,6 +47,73 @@ Deno.test('foreign keys are enforced', () => { db.close(); }); +Deno.test('csv import migration applies to a populated database', () => { + // Stage only the migrations before 005, populate, then upgrade — the path an + // existing install actually takes. + const stageDir = Deno.makeTempDirSync(); + for (const entry of Deno.readDirSync(MIGRATIONS_DIR)) { + const version = Number(entry.name.match(/^(\d+)_/)?.[1]); + if (Number.isFinite(version) && version < 5) { + Deno.copyFileSync(join(MIGRATIONS_DIR, entry.name), join(stageDir, entry.name)); + } + } + + const dbPath = join(Deno.makeTempDirSync(), 'test.db'); + const db = openDatabase(dbPath, stageDir); + + const now = new Date().toISOString(); + db.prepare('INSERT INTO connections (access_url, claimed_at) VALUES (?, ?)').run( + 'https://example.test', + now + ); + db.prepare( + `INSERT INTO accounts (id, connection_id, name, currency, state, created_at) + VALUES ('acct-1', 1, 'Checking', 'USD', 'ACTIVE', ?)` + ).run(now); + db.prepare( + `INSERT INTO transactions (account_id, sfin_id, posted, amount_cents, description, created_at) + VALUES ('acct-1', 'sfin-1', 1700000000, -1234, 'COFFEE', ?)` + ).run(now); + + const applied = runMigrations(db, MIGRATIONS_DIR); + if (applied !== 1) throw new Error(`expected only 005 to apply, got ${applied}`); + + const row = db.prepare("SELECT import_id FROM transactions WHERE sfin_id = 'sfin-1'").get() as { + import_id: number | null; + }; + if (row.import_id !== null) { + throw new Error(`pre-existing rows must read back as synced, got ${row.import_id}`); + } + + db.close(); +}); + +Deno.test('imports status is a closed enum', () => { + const dbPath = join(Deno.makeTempDirSync(), 'test.db'); + const db = openDatabase(dbPath, MIGRATIONS_DIR); + const now = new Date().toISOString(); + db.prepare('INSERT INTO connections (access_url, claimed_at) VALUES (?, ?)').run( + 'https://example.test', + now + ); + db.prepare( + `INSERT INTO accounts (id, connection_id, name, currency, state, created_at) + VALUES ('acct-1', 1, 'Checking', 'USD', 'ACTIVE', ?)` + ).run(now); + + let threw = false; + try { + db.prepare( + `INSERT INTO imports (account_id, uploaded_at, payload, status) + VALUES ('acct-1', ?, 'raw', 'bogus')` + ).run(now); + } catch { + threw = true; + } + if (!threw) throw new Error('expected status CHECK violation'); + db.close(); +}); + function join(...parts: string[]) { return parts.join('/'); } diff --git a/src/lib/server/services/csv-import.test.ts b/src/lib/server/services/csv-import.test.ts new file mode 100644 index 0000000..871d57d --- /dev/null +++ b/src/lib/server/services/csv-import.test.ts @@ -0,0 +1,312 @@ +/// +import { + detectDateFormat, + detectMapping, + normalizeCsv, + normalizeCsvAmount, + parseCsv, + parseCsvDate, + type CsvMapping +} from './csv-import.ts'; +import { parseAmountToCents } from './normalize.ts'; + +const SIGNED_CSV = `Date,Description,Amount +2026-07-02,Trader Joe's,-88.14 +2026-07-03,Paycheck,2500.00 +`; + +function signedMapping(overrides: Partial = {}): CsvMapping { + return { + date: 'Date', + dateFormat: 'iso', + amountMode: 'signed', + amount: 'Amount', + description: 'Description', + ...overrides + }; +} + +Deno.test('parseCsv strips a BOM and tolerates CRLF', () => { + const parsed = parseCsv('Date,Description,Amount\r\n2026-07-02,Coffee,-4.75\r\n'); + if (parsed.headers[0] !== 'Date') throw new Error(`BOM leaked: ${JSON.stringify(parsed.headers[0])}`); + if (parsed.rows.length !== 1) throw new Error(`expected 1 row, got ${parsed.rows.length}`); +}); + +Deno.test('parseCsv handles quoted delimiters and embedded newlines', () => { + const parsed = parseCsv('Date,Description,Amount\n2026-07-02,"Shop, Inc.\nStore #4",-10.00\n'); + if (parsed.rows[0][1] !== 'Shop, Inc.\nStore #4') { + throw new Error(`quoting mishandled: ${JSON.stringify(parsed.rows[0][1])}`); + } + if (parsed.rows[0][2] !== '-10.00') throw new Error('column alignment broken by quoting'); +}); + +Deno.test('parseCsv rejects an empty file', () => { + let threw = false; + try { + parseCsv('\n\n'); + } catch { + threw = true; + } + if (!threw) throw new Error('expected an empty file to throw'); +}); + +Deno.test('normalizeCsvAmount handles bank negative conventions', () => { + // Thousands separators pass through by design — parseAmountToCents strips them. + const cases: [string, string][] = [ + ['(12.34)', '-12.34'], + ['12.34-', '-12.34'], + ['-12.34', '-12.34'], + ['$1,234.56', '1,234.56'], + ['($1,234.56)', '-1,234.56'], + ['12.34', '12.34'], + ['', ''], + [' ', ''] + ]; + for (const [input, expected] of cases) { + const actual = normalizeCsvAmount(input); + if (actual !== expected) { + throw new Error(`${JSON.stringify(input)} -> ${JSON.stringify(actual)}, want ${expected}`); + } + } +}); + +Deno.test('normalizeCsvAmount output is accepted by parseAmountToCents', () => { + // The handoff is the point: this pre-pass exists so parseAmountToCents can stay + // strict for the sync path. + const cases: [string, number][] = [ + ['(12.34)', -1234], + ['12.34-', -1234], + ['$1,234.56', 123456], + ['($1,234.56)', -123456], + ['12.34', 1234] + ]; + for (const [input, expected] of cases) { + const actual = parseAmountToCents(normalizeCsvAmount(input)); + if (actual !== expected) { + throw new Error(`${JSON.stringify(input)} -> ${actual} cents, want ${expected}`); + } + } +}); + +Deno.test('parenthesized negative reaches integer cents', () => { + const { candidates, errors } = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,(12.34)\n', + signedMapping(), + 'acct-1' + ); + if (errors.length) throw new Error(errors.join('; ')); + if (candidates[0].amountCents !== -1234) { + throw new Error(`got ${candidates[0].amountCents}, want -1234`); + } +}); + +Deno.test('detectDateFormat reads the order off the data', () => { + if (detectDateFormat(['2026-07-02']) !== 'iso') throw new Error('iso not detected'); + // 13 can only be a day, so day comes first. + if (detectDateFormat(['03/04/2026', '13/04/2026']) !== 'dmy') throw new Error('dmy not detected'); + if (detectDateFormat(['03/04/2026', '04/13/2026']) !== 'mdy') throw new Error('mdy not detected'); + // Wholly ambiguous input falls back to month-first. + if (detectDateFormat(['03/04/2026']) !== 'mdy') throw new Error('ambiguous should default to mdy'); +}); + +Deno.test('parseCsvDate respects the mapped order', () => { + const mdy = parseCsvDate('03/04/2026', 'mdy'); + const dmy = parseCsvDate('03/04/2026', 'dmy'); + if (mdy !== Date.UTC(2026, 2, 4, 12) / 1000) throw new Error('mdy parsed wrong'); + if (dmy !== Date.UTC(2026, 3, 3, 12) / 1000) throw new Error('dmy parsed wrong'); + // ISO is unambiguous and ignores the declared order. + if (parseCsvDate('2026-07-02', 'dmy') !== Date.UTC(2026, 6, 2, 12) / 1000) { + throw new Error('iso should win regardless of format'); + } +}); + +Deno.test('a date-only value survives local rendering from UTC-12 to UTC+11', () => { + // The reason for anchoring at noon: formatDay renders in local time, so a + // UTC-midnight stamp would read as the previous day across the Americas, and a + // month's first day would display in the month before the one it reports under. + // Noon buys 12 hours of slack each way, which covers every timezone the two + // users of this app will ever be in. It is not universal: at UTC+12 and beyond + // (New Zealand, Kiribati) noon crosses into the next day. Documented, not fixed + // — the honest fix is rendering date-only rows in UTC, which is worth doing only + // if that ever matters. + const posted = parseCsvDate('2026-07-01', 'iso'); + for (let offset = -12; offset <= 11; offset++) { + const local = new Date((posted + offset * 3600) * 1000); + if (local.getUTCDate() !== 1 || local.getUTCMonth() !== 6) { + throw new Error(`UTC${offset >= 0 ? '+' : ''}${offset} shifts the calendar day`); + } + } +}); + +Deno.test('parseCsvDate rejects nonsense', () => { + for (const bad of ['', 'not a date', '99/99/2026']) { + let threw = false; + try { + parseCsvDate(bad, 'mdy'); + } catch { + threw = true; + } + if (!threw) throw new Error(`expected ${JSON.stringify(bad)} to throw`); + } +}); + +Deno.test('detectMapping finds signed-column headers', () => { + const mapping = detectMapping(parseCsv(SIGNED_CSV)); + if (!mapping) throw new Error('expected a mapping'); + if (mapping.amountMode !== 'signed' || mapping.amount !== 'Amount') { + throw new Error(`wrong amount mapping: ${JSON.stringify(mapping)}`); + } + if (mapping.date !== 'Date' || mapping.description !== 'Description') { + throw new Error(`wrong column mapping: ${JSON.stringify(mapping)}`); + } + if (mapping.dateFormat !== 'iso') throw new Error('expected iso'); +}); + +Deno.test('detectMapping finds a debit/credit pair', () => { + const mapping = detectMapping( + parseCsv('Date,Description,Debit,Credit\n07/02/2026,Coffee,4.75,\n') + ); + if (!mapping) throw new Error('expected a mapping'); + if (mapping.amountMode !== 'debit-credit') throw new Error('expected debit-credit mode'); + if (mapping.debit !== 'Debit' || mapping.credit !== 'Credit') { + throw new Error(`wrong pair: ${JSON.stringify(mapping)}`); + } +}); + +Deno.test('detectMapping gives up rather than guessing', () => { + if (detectMapping(parseCsv('Foo,Bar,Baz\n1,2,3\n')) !== null) { + throw new Error('expected null for unrecognizable headers'); + } +}); + +Deno.test('debit and credit columns resolve to one signed amount', () => { + const mapping: CsvMapping = { + date: 'Date', + dateFormat: 'mdy', + amountMode: 'debit-credit', + debit: 'Debit', + credit: 'Credit', + description: 'Description' + }; + const { candidates, errors } = normalizeCsv( + 'Date,Description,Debit,Credit\n07/02/2026,Coffee,4.75,\n07/03/2026,Refund,,20.00\n', + mapping, + 'acct-1' + ); + if (errors.length) throw new Error(errors.join('; ')); + if (candidates[0].amountCents !== -475) throw new Error(`debit -> ${candidates[0].amountCents}`); + if (candidates[1].amountCents !== 2000) throw new Error(`credit -> ${candidates[1].amountCents}`); +}); + +Deno.test('a bad row is reported without sinking the file', () => { + const { candidates, errors } = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\nnot-a-date,Broken,-1.00\n', + signedMapping(), + 'acct-1' + ); + if (candidates.length !== 1) throw new Error(`expected 1 good row, got ${candidates.length}`); + if (errors.length !== 1 || !errors[0].includes('Line 3')) { + throw new Error(`expected a line-3 error, got ${JSON.stringify(errors)}`); + } +}); + +Deno.test('imported rows carry posted and never pend', () => { + const { candidates } = normalizeCsv(SIGNED_CSV, signedMapping(), 'acct-1'); + if (candidates[0].posted !== Date.UTC(2026, 6, 2, 12) / 1000) throw new Error('posted not set'); + if (candidates[0].transactedAt !== null) throw new Error('transactedAt should be null'); +}); + +Deno.test('a distinct transaction-date column maps separately', () => { + const { candidates, errors } = normalizeCsv( + 'Post Date,Transaction Date,Description,Amount\n2026-07-04,2026-07-02,Coffee,-4.75\n', + signedMapping({ date: 'Post Date', transactedDate: 'Transaction Date' }), + 'acct-1' + ); + if (errors.length) throw new Error(errors.join('; ')); + if (candidates[0].posted !== Date.UTC(2026, 6, 4, 12) / 1000) throw new Error('wrong posted'); + if (candidates[0].transactedAt !== Date.UTC(2026, 6, 2, 12) / 1000) { + throw new Error('wrong transactedAt'); + } +}); + +Deno.test('identity is stable across runs', () => { + const a = normalizeCsv(SIGNED_CSV, signedMapping(), 'acct-1'); + const b = normalizeCsv(SIGNED_CSV, signedMapping(), 'acct-1'); + const idsA = a.candidates.map((c) => c.syntheticId).join(','); + const idsB = b.candidates.map((c) => c.syntheticId).join(','); + if (idsA !== idsB) throw new Error('ids must be deterministic'); +}); + +Deno.test('identity is namespaced and account-scoped', () => { + const { candidates } = normalizeCsv(SIGNED_CSV, signedMapping(), 'acct-1'); + if (!candidates[0].syntheticId.startsWith('csv:')) throw new Error('missing csv: namespace'); + + const other = normalizeCsv(SIGNED_CSV, signedMapping(), 'acct-2'); + if (candidates[0].syntheticId === other.candidates[0].syntheticId) { + throw new Error('the same row in a different account must not share an id'); + } +}); + +Deno.test('two genuinely identical rows stay distinct', () => { + const { candidates } = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\n2026-07-02,Coffee,-4.75\n', + signedMapping(), + 'acct-1' + ); + if (candidates.length !== 2) throw new Error(`expected 2 rows, got ${candidates.length}`); + if (candidates[0].syntheticId === candidates[1].syntheticId) { + throw new Error('identical rows must get distinct ids via the occurrence index'); + } +}); + +Deno.test('cosmetic description re-rendering does not mint a new identity', () => { + const a = normalizeCsv('Date,Description,Amount\n2026-07-02,Trader Joe\'s,-88.14\n', signedMapping(), 'acct-1'); + const b = normalizeCsv( + 'Date,Description,Amount\n2026-07-02, TRADER JOE\'S ,-88.14\n', + signedMapping(), + 'acct-1' + ); + if (a.candidates[0].syntheticId !== b.candidates[0].syntheticId) { + throw new Error('folded description should yield the same id'); + } +}); + +Deno.test('a reordered group re-derives the same id set', () => { + // Rows within a content group are interchangeable, so export order must not + // change the set of ids — this is what makes an overlapping re-import a no-op. + const first = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\n2026-07-02,Bagel,-3.00\n2026-07-02,Coffee,-4.75\n', + signedMapping(), + 'acct-1' + ); + const reordered = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\n2026-07-02,Coffee,-4.75\n2026-07-02,Bagel,-3.00\n', + signedMapping(), + 'acct-1' + ); + const setA = new Set(first.candidates.map((c) => c.syntheticId)); + const setB = new Set(reordered.candidates.map((c) => c.syntheticId)); + if (setA.size !== setB.size || [...setA].some((id) => !setB.has(id))) { + throw new Error('id set must be independent of row order within a group'); + } +}); + +Deno.test('a superset re-export only adds genuinely new ids', () => { + const first = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\n2026-07-02,Coffee,-4.75\n', + signedMapping(), + 'acct-1' + ); + const superset = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Coffee,-4.75\n2026-07-02,Coffee,-4.75\n2026-07-02,Coffee,-4.75\n', + signedMapping(), + 'acct-1' + ); + const before = new Set(first.candidates.map((c) => c.syntheticId)); + const after = superset.candidates.map((c) => c.syntheticId); + const added = after.filter((id) => !before.has(id)); + if (added.length !== 1) throw new Error(`expected exactly 1 new id, got ${added.length}`); + for (const id of before) { + if (!after.includes(id)) throw new Error('previously ingested ids must survive a superset'); + } +}); diff --git a/src/lib/server/services/csv-import.ts b/src/lib/server/services/csv-import.ts new file mode 100644 index 0000000..419adeb --- /dev/null +++ b/src/lib/server/services/csv-import.ts @@ -0,0 +1,309 @@ +// Pure parsing and normalization of a bank CSV export. No I/O, no DB: a function +// from archived payload text + mapping to typed candidate rows, mirroring +// normalize.ts so an import can be replayed from its archived record. +// +// CSV import is an ADDITIVE writer. Nothing here reconciles, sweeps, or mutates. + +import { parse } from 'csv-parse/sync'; +import { createHash } from 'node:crypto'; +import { parseAmountToCents } from './normalize.ts'; + +/** + * `03/04/2026` is March 4th at one bank and April 3rd at another, and the file + * never says which. The format is part of the mapping so the user's answer is + * archived and the import replays deterministically. + */ +export type CsvDateFormat = 'iso' | 'mdy' | 'dmy'; + +export type CsvAmountMode = 'signed' | 'debit-credit'; + +export interface CsvMapping { + date: string; + /** Distinct transaction date, when the file separates it from the post date. */ + transactedDate?: string | null; + dateFormat: CsvDateFormat; + amountMode: CsvAmountMode; + /** `signed` mode: one column carrying the sign. */ + amount?: string | null; + /** `debit-credit` mode: two columns of unsigned magnitudes. */ + debit?: string | null; + credit?: string | null; + description: string; + payee?: string | null; + memo?: string | null; +} + +export interface CsvCandidate { + /** Deterministic content-derived id; namespaced to never collide with SimpleFIN's. */ + syntheticId: string; + posted: number; // Unix seconds, UTC midnight of the file's date + transactedAt: number | null; + amountCents: number; + description: string; + payee: string | null; + memo: string | null; +} + +export interface ParsedCsv { + headers: string[]; + rows: string[][]; +} + +export interface NormalizedCsv { + candidates: CsvCandidate[]; + /** Row-level failures, kept as messages so one bad row doesn't sink the file. */ + errors: string[]; +} + +/** + * Parse CSV text into a header row and data rows. Quoting, embedded newlines, + * CRLF, and the UTF-8 BOM that Excel-exported bank files carry are all delegated + * to csv-parse — hand-rolling those is where CSV handling quietly eats data. + * Ragged rows are tolerated: bank exports pad or truncate trailing columns, and a + * stray comma must not sink an otherwise good file. + */ +export function parseCsv(payloadText: string): ParsedCsv { + const rows = parse(payloadText, { + bom: true, + skip_empty_lines: true, + relax_column_count: true, + relax_quotes: true + }) as string[][]; + const nonEmpty = rows.filter((row) => row.some((cell) => cell.trim() !== '')); + if (nonEmpty.length === 0) throw new Error('The file has no rows.'); + const [headers, ...data] = nonEmpty; + return { headers: headers.map((h) => h.trim()), rows: data }; +} + +const DATE_RE = /^(date|posted|post date|posting date|transaction date|trans date)$/i; +const TXN_DATE_RE = /^(transaction date|trans date)$/i; +const AMOUNT_RE = /^(amount|value|transaction amount)$/i; +const DEBIT_RE = /^(debit|withdrawal|withdrawals|money out|paid out)$/i; +const CREDIT_RE = /^(credit|deposit|deposits|money in|paid in)$/i; +const DESC_RE = /^(description|details|name|merchant|payee|narrative|transaction)$/i; +const PAYEE_RE = /^(payee|merchant|name)$/i; +const MEMO_RE = /^(memo|note|notes|reference)$/i; + +function findHeader(headers: string[], re: RegExp): string | null { + return headers.find((h) => re.test(h.trim())) ?? null; +} + +/** + * Infer the date format from the data itself: a component above 12 can only be a + * day, which settles the order. When every row is ambiguous (all components ≤ 12) + * there is no signal, so fall back to month-first and let the user correct it — + * the preview's stated date range is the backstop. + */ +export function detectDateFormat(values: string[]): CsvDateFormat { + let sawIso = false; + for (const value of values) { + const text = value.trim(); + if (/^\d{4}-\d{1,2}-\d{1,2}/.test(text)) { + sawIso = true; + continue; + } + const match = text.match(/^(\d{1,2})[/.-](\d{1,2})[/.-](\d{2,4})$/); + if (!match) continue; + if (Number(match[1]) > 12) return 'dmy'; + if (Number(match[2]) > 12) return 'mdy'; + } + return sawIso ? 'iso' : 'mdy'; +} + +/** Best-guess column assignment from a header row. Always user-confirmable. */ +export function detectMapping(parsed: ParsedCsv): CsvMapping | null { + const { headers, rows } = parsed; + const date = findHeader(headers, DATE_RE); + const description = findHeader(headers, DESC_RE); + if (!date || !description) return null; + + const amount = findHeader(headers, AMOUNT_RE); + const debit = findHeader(headers, DEBIT_RE); + const credit = findHeader(headers, CREDIT_RE); + + const dateIndex = headers.indexOf(date); + const dateFormat = detectDateFormat(rows.map((r) => r[dateIndex] ?? '')); + + // A distinct transaction-date column only counts when it isn't the one already + // serving as the post date. + const txnDate = findHeader(headers, TXN_DATE_RE); + const transactedDate = txnDate && txnDate !== date ? txnDate : null; + + // Prefer an explicit signed column; fall back to a debit/credit pair. + const payee = findHeader(headers, PAYEE_RE); + const base = { + date, + transactedDate, + dateFormat, + description, + payee: payee && payee !== description ? payee : null, + memo: findHeader(headers, MEMO_RE) + }; + if (amount) return { ...base, amountMode: 'signed', amount }; + if (debit || credit) return { ...base, amountMode: 'debit-credit', debit, credit }; + return null; +} + +/** + * Normalize a bank CSV's negative conventions into a signed decimal string that + * parseAmountToCents accepts. That function's strict regex is the sync path's + * contract — widening it to swallow accountant's parentheses would weaken + * validation on SimpleFIN payloads to serve this one, so the tolerance lives here. + * Returns '' for a blank cell. + */ +export function normalizeCsvAmount(raw: string): string { + let text = String(raw ?? '').trim(); + if (!text) return ''; + + let negative = false; + const parenthesized = text.match(/^\((.*)\)$/); + if (parenthesized) { + negative = true; + text = parenthesized[1].trim(); + } + if (text.endsWith('-')) { + negative = !negative; + text = text.slice(0, -1).trim(); + } + if (text.startsWith('-')) { + negative = !negative; + text = text.slice(1).trim(); + } + + // Drop currency symbols and stray spaces; parseAmountToCents handles commas. + text = text.replace(/[^\d.,]/g, ''); + if (!text) return ''; + return negative ? `-${text}` : text; +} + +/** + * Parse a CSV date cell to Unix seconds at **noon UTC**. ISO always wins. + * + * Noon, not midnight: a CSV gives a calendar day with no time, but `formatDay` + * renders in local time, so a UTC-midnight stamp displays as the *previous* day + * anywhere west of Greenwich — and a July 1 row would read "Jun 30" while still + * counting in July's report, which buckets by UTC. Noon is the same calendar day + * in local time for every offset from -11 to +12, and still lands inside the + * right UTC month. `formatMonth` already anchors to the 15th for the same reason. + */ +export function parseCsvDate(raw: string, format: CsvDateFormat): number { + const text = String(raw ?? '').trim(); + let year: number, month: number, day: number; + + const iso = text.match(/^(\d{4})-(\d{1,2})-(\d{1,2})/); + if (iso) { + [year, month, day] = [Number(iso[1]), Number(iso[2]), Number(iso[3])]; + } else { + const parts = text.match(/^(\d{1,2})[/.-](\d{1,2})[/.-](\d{2,4})$/); + if (!parts) throw new Error(`Unparseable date: ${JSON.stringify(raw)}`); + const first = Number(parts[1]); + const second = Number(parts[2]); + year = Number(parts[3]); + if (year < 100) year += 2000; + if (format === 'dmy') { + day = first; + month = second; + } else { + month = first; + day = second; + } + } + + if (month < 1 || month > 12 || day < 1 || day > 31) { + throw new Error(`Unparseable date: ${JSON.stringify(raw)}`); + } + return Date.UTC(year, month - 1, day, 12) / 1000; +} + +/** + * The grouping key IS the row's content, so rows within a group are + * interchangeable: a re-export listing them in a different order, or a superset + * of them, re-derives the same id set for rows already ingested and only genuinely + * new rows fall outside it. Description is folded (trimmed, whitespace-collapsed, + * lowercased) so cosmetic re-rendering doesn't mint a new identity. + */ +function contentKey(accountId: string, posted: number, amountCents: number, description: string) { + const folded = description.trim().replace(/\s+/g, ' ').toLowerCase(); + return [accountId, String(posted), String(amountCents), folded].join(''); +} + +function syntheticId(key: string, occurrence: number): string { + const digest = createHash('sha256').update(`${key}${occurrence}`).digest('hex'); + return `csv:${digest.slice(0, 32)}`; +} + +function cell(row: string[], headers: string[], name: string | null | undefined): string { + if (!name) return ''; + const index = headers.indexOf(name); + return index === -1 ? '' : (row[index] ?? ''); +} + +function resolveAmountCents(row: string[], headers: string[], mapping: CsvMapping): number { + if (mapping.amountMode === 'signed') { + const normalized = normalizeCsvAmount(cell(row, headers, mapping.amount)); + if (!normalized) throw new Error('amount is blank'); + return parseAmountToCents(normalized); + } + + const debitText = normalizeCsvAmount(cell(row, headers, mapping.debit)); + const creditText = normalizeCsvAmount(cell(row, headers, mapping.credit)); + const debit = debitText ? parseAmountToCents(debitText) : 0; + const credit = creditText ? parseAmountToCents(creditText) : 0; + + if (debit !== 0 && credit !== 0) throw new Error('both debit and credit are populated'); + if (debit !== 0) return -Math.abs(debit); + if (credit !== 0) return Math.abs(credit); + throw new Error('neither debit nor credit is populated'); +} + +/** + * Pure: archived bytes + mapping + account -> candidate rows. Imported rows are + * settled history, so they carry `posted` and are never pending — matching the + * shape sync emits for posted rows, so every downstream query behaves identically. + */ +export function normalizeCsv( + payloadText: string, + mapping: CsvMapping, + accountId: string +): NormalizedCsv { + const { headers, rows } = parseCsv(payloadText); + const candidates: CsvCandidate[] = []; + const errors: string[] = []; + const seen = new Map(); + + rows.forEach((row, index) => { + const line = index + 2; // 1-based, and the header occupies line 1 + try { + const posted = parseCsvDate(cell(row, headers, mapping.date), mapping.dateFormat); + const amountCents = resolveAmountCents(row, headers, mapping); + const description = cell(row, headers, mapping.description).trim(); + if (!description) throw new Error('description is blank'); + + const transactedRaw = cell(row, headers, mapping.transactedDate).trim(); + const transactedAt = transactedRaw + ? parseCsvDate(transactedRaw, mapping.dateFormat) + : null; + + const key = contentKey(accountId, posted, amountCents, description); + const occurrence = seen.get(key) ?? 0; + seen.set(key, occurrence + 1); + + const payee = cell(row, headers, mapping.payee).trim(); + const memo = cell(row, headers, mapping.memo).trim(); + + candidates.push({ + syntheticId: syntheticId(key, occurrence), + posted, + transactedAt, + amountCents, + description, + payee: payee || null, + memo: memo || null + }); + } catch (err) { + errors.push(`Line ${line}: ${err instanceof Error ? err.message : String(err)}`); + } + }); + + return { candidates, errors }; +} diff --git a/src/lib/server/services/imports.test.ts b/src/lib/server/services/imports.test.ts new file mode 100644 index 0000000..020440f --- /dev/null +++ b/src/lib/server/services/imports.test.ts @@ -0,0 +1,491 @@ +/// +import { openDatabase } from '../db.ts'; +import { + commitImport, + createDraftImport, + findPotentialDuplicates, + previewImport, + saveDecisions, + saveMapping, + undoImport +} from './imports.ts'; +import { normalizeCsv, type CsvMapping } from './csv-import.ts'; +import { createRule } from './rules.ts'; +import { categorizeManually } from './categorization.ts'; +import { monthlyReport, netWorthSeries } from './reports.ts'; +import type { DatabaseSync } from 'node:sqlite'; + +const MIGRATIONS_DIR = new URL('../../../../migrations', import.meta.url).pathname.replace( + /^\/([A-Za-z]:)/, + '$1' +); + +/** The synced row sits just after UTC midnight on purpose: a CSV row is anchored at + * noon, so a naive ±86400s window would be hour-sensitive. The duplicate check + * must compare calendar days. */ +const JULY_02_EARLY = Math.floor(Date.UTC(2026, 6, 2, 1, 30) / 1000); +/** Where a CSV row for that date lands. */ +const JULY_02_NOON = Math.floor(Date.UTC(2026, 6, 2, 12) / 1000); +const JULY_03_NOON = Math.floor(Date.UTC(2026, 6, 3, 12) / 1000); + +const MAPPING: CsvMapping = { + date: 'Date', + dateFormat: 'iso', + amountMode: 'signed', + amount: 'Amount', + description: 'Description' +}; + +/** The account holds a synced posted row, a synced pending row, and a category. */ +function testDb(): DatabaseSync { + const db = openDatabase(`${Deno.makeTempDirSync()}/t.db`, MIGRATIONS_DIR); + const now = new Date().toISOString(); + db.prepare("INSERT INTO users (did, handle, created_at) VALUES ('did:plc:test', 'tester', ?)").run( + now + ); + db.prepare("INSERT INTO connections (access_url, claimed_at) VALUES ('https://x', ?)").run(now); + for (const [id, state] of [ + ['chk', 'ACTIVE'], + ['ghost', 'HIDDEN'] + ]) { + db.prepare( + `INSERT INTO accounts (id, connection_id, name, currency, state, last_successful_data_at, created_at) + VALUES (?, 1, ?, 'USD', ?, ?, ?)` + ).run(id, id, state, now, now); + } + db.prepare("INSERT INTO categories (name, kind, created_at) VALUES ('Dining','expense',?)").run( + now + ); + const insert = db.prepare( + `INSERT INTO transactions + (account_id, sfin_id, posted, amount_cents, description, pending, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?)` + ); + // The bank's own rendering of the same July 2 purchase the CSV also contains. + insert.run('chk', 'sfin-amazon', JULY_02_EARLY, -5231, 'AMZN Mktp US*2K4LM9QR3', 0, now); + insert.run('chk', 'sfin-pending', null, -500, 'PENDING COFFEE', 1, now); + return db; +} + +function draft(db: DatabaseSync, csv: string, accountId = 'chk'): number { + const id = createDraftImport(db, accountId, 'bank.csv', csv); + saveMapping(db, id, MAPPING); + return id; +} + +Deno.test('createDraftImport archives bytes verbatim before parsing', () => { + const db = testDb(); + // Garbage that could never parse still has to survive the upload. + const raw = 'not,really\nvalid ""csv'; + const id = createDraftImport(db, 'chk', 'junk.csv', raw); + const row = db.prepare('SELECT payload, status FROM imports WHERE id = ?').get(id) as { + payload: string; + status: string; + }; + if (row.payload !== raw) throw new Error('payload was not archived verbatim'); + if (row.status !== 'draft') throw new Error(`expected draft, got ${row.status}`); + db.close(); +}); + +Deno.test('a hidden account cannot be an import target', () => { + const db = testDb(); + let threw = false; + try { + createDraftImport(db, 'ghost', 'x.csv', 'Date,Description,Amount\n'); + } catch { + threw = true; + } + if (!threw) throw new Error('expected hidden account to be rejected'); + db.close(); +}); + +Deno.test('cross-source duplicate is flagged despite a different description', () => { + const db = testDb(); + const csv = 'Date,Description,Amount\n2026-07-02,Amazon,-52.31\n'; + const { candidates } = normalizeCsv(csv, MAPPING, 'chk'); + const [result] = findPotentialDuplicates(db, 'chk', candidates); + if (result.status !== 'flagged') throw new Error(`expected flagged, got ${result.status}`); + if (result.match?.description !== 'AMZN Mktp US*2K4LM9QR3') throw new Error('wrong match'); + if (result.match?.source !== 'synced') throw new Error('match should be synced'); + db.close(); +}); + +Deno.test('duplicate window spans one calendar day, not 24 hours', () => { + const db = testDb(); + // The synced row is at 01:30 UTC on Jul 2; this candidate lands at noon Jul 3. + // That is 34.5 hours apart but one calendar day, and must still flag. + const near = normalizeCsv( + 'Date,Description,Amount\n2026-07-03,Amazon,-52.31\n', + MAPPING, + 'chk' + ).candidates; + if (findPotentialDuplicates(db, 'chk', near)[0].status !== 'flagged') { + throw new Error('one calendar day off should be flagged regardless of the hours'); + } + const far = normalizeCsv( + 'Date,Description,Amount\n2026-07-04,Amazon,-52.31\n', + MAPPING, + 'chk' + ).candidates; + if (findPotentialDuplicates(db, 'chk', far)[0].status !== 'new') { + throw new Error('two calendar days off should not be flagged'); + } + // And the day before, symmetrically. + const before = normalizeCsv( + 'Date,Description,Amount\n2026-07-01,Amazon,-52.31\n', + MAPPING, + 'chk' + ).candidates; + if (findPotentialDuplicates(db, 'chk', before)[0].status !== 'flagged') { + throw new Error('the window must be symmetric'); + } + db.close(); +}); + +Deno.test('a different amount on the same day is not a duplicate', () => { + const db = testDb(); + const { candidates } = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Amazon,-52.30\n', + MAPPING, + 'chk' + ); + if (findPotentialDuplicates(db, 'chk', candidates)[0].status !== 'new') { + throw new Error('a one-cent difference is a different transaction'); + } + db.close(); +}); + +Deno.test('matches are consumed, so a real second charge stays importable', () => { + const db = testDb(); + // Two identical charges in the file, one already synced: one dupe, one new. + const { candidates } = normalizeCsv( + 'Date,Description,Amount\n2026-07-02,Amazon,-52.31\n2026-07-02,Amazon,-52.31\n', + MAPPING, + 'chk' + ); + const results = findPotentialDuplicates(db, 'chk', candidates); + const statuses = results.map((r) => r.status).sort(); + if (statuses.join(',') !== 'flagged,new') { + throw new Error(`expected one flagged and one new, got ${statuses.join(',')}`); + } + db.close(); +}); + +Deno.test('commit defaults flagged rows to skip and leaves the synced row alone', () => { + const db = testDb(); + const id = draft(db, 'Date,Description,Amount\n2026-07-02,Amazon,-52.31\n2026-06-01,Shell,-40.00\n'); + const result = commitImport(db, id); + + if (result.skipped !== 1) throw new Error(`expected 1 skipped, got ${result.skipped}`); + if (result.inserted !== 1) throw new Error(`expected 1 inserted, got ${result.inserted}`); + + const amazon = db + .prepare("SELECT COUNT(*) AS n FROM transactions WHERE account_id='chk' AND amount_cents=-5231 AND removed_at IS NULL") + .get() as { n: number }; + if (amazon.n !== 1) throw new Error(`double-counted: ${amazon.n} rows at -5231`); + + const synced = db + .prepare("SELECT description, import_id FROM transactions WHERE sfin_id='sfin-amazon'") + .get() as { description: string; import_id: number | null }; + if (synced.description !== 'AMZN Mktp US*2K4LM9QR3' || synced.import_id !== null) { + throw new Error('the synced row must be untouched'); + } + db.close(); +}); + +Deno.test('a kept flagged row is imported alongside the existing one', () => { + const db = testDb(); + const csv = 'Date,Description,Amount\n2026-07-02,Amazon,-52.31\n'; + const id = draft(db, csv); + const { candidates } = normalizeCsv(csv, MAPPING, 'chk'); + saveDecisions(db, id, { [candidates[0].syntheticId]: 'keep' }); + + const result = commitImport(db, id); + if (result.inserted !== 1) throw new Error(`expected 1 inserted, got ${result.inserted}`); + const rows = db + .prepare("SELECT COUNT(*) AS n FROM transactions WHERE account_id='chk' AND amount_cents=-5231 AND removed_at IS NULL") + .get() as { n: number }; + if (rows.n !== 2) throw new Error(`expected both rows, got ${rows.n}`); + db.close(); +}); + +Deno.test('import never disturbs pending rows, snapshots, or account state', () => { + const db = testDb(); + const before = db.prepare("SELECT state, last_successful_data_at FROM accounts WHERE id='chk'").get() as { + state: string; + last_successful_data_at: string; + }; + + const id = draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n'); + commitImport(db, id); + + // The stale-pending sweep in ingestTransactions would have removed this; the + // additive writer must not. + const pending = db + .prepare("SELECT removed_at FROM transactions WHERE sfin_id='sfin-pending'") + .get() as { removed_at: string | null }; + if (pending.removed_at !== null) throw new Error('an import must never remove a pending row'); + + const snaps = db.prepare('SELECT COUNT(*) AS n FROM balance_snapshots').get() as { n: number }; + if (snaps.n !== 0) throw new Error('an import must not write balance snapshots'); + + const after = db.prepare("SELECT state, last_successful_data_at FROM accounts WHERE id='chk'").get() as { + state: string; + last_successful_data_at: string; + }; + if (after.state !== before.state || after.last_successful_data_at !== before.last_successful_data_at) { + throw new Error('an import must not touch account state'); + } + db.close(); +}); + +Deno.test('re-importing the same file inserts nothing', () => { + const db = testDb(); + const csv = 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n2026-06-02,Coffee,-4.75\n'; + commitImport(db, draft(db, csv)); + const second = commitImport(db, draft(db, csv)); + + if (second.inserted !== 0) throw new Error(`expected 0 inserted, got ${second.inserted}`); + if (second.alreadyPresent !== 2) throw new Error(`expected 2 already-present, got ${second.alreadyPresent}`); + const total = db + .prepare("SELECT COUNT(*) AS n FROM transactions WHERE import_id IS NOT NULL AND removed_at IS NULL") + .get() as { n: number }; + if (total.n !== 2) throw new Error(`expected 2 imported rows total, got ${total.n}`); + db.close(); +}); + +Deno.test('an overlapping file adds only the new rows', () => { + const db = testDb(); + commitImport(db, draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n2026-06-02,Coffee,-4.75\n')); + const second = commitImport( + db, + draft(db, 'Date,Description,Amount\n2026-06-02,Coffee,-4.75\n2026-06-03,Bagel,-3.00\n') + ); + if (second.inserted !== 1) throw new Error(`expected 1 inserted, got ${second.inserted}`); + if (second.alreadyPresent !== 1) throw new Error(`expected 1 already-present, got ${second.alreadyPresent}`); + db.close(); +}); + +Deno.test('committed rows carry import_id and are categorized by existing rules', () => { + const db = testDb(); + createRule(db, { + pattern: 'shell', + matchType: 'contains', + categoryId: 2, + createdByDid: 'did:plc:test' + }); + + const id = draft(db, 'Date,Description,Amount\n2026-06-01,SHELL OIL 4417,-40.00\n'); + const result = commitImport(db, id); + if (result.ruleCategorized < 1) throw new Error('rules should have caught the imported row'); + + const row = db + .prepare("SELECT import_id, category_id FROM transactions WHERE description='SHELL OIL 4417'") + .get() as { import_id: number; category_id: number | null }; + if (row.import_id !== id) throw new Error(`import_id not stamped: ${row.import_id}`); + if (row.category_id !== 2) throw new Error('imported row should be rule-categorized'); + + const event = db + .prepare( + `SELECT source FROM categorization_events WHERE transaction_id = + (SELECT id FROM transactions WHERE description='SHELL OIL 4417')` + ) + .get() as { source: string }; + if (event.source !== 'rule') throw new Error(`expected a rule event, got ${event.source}`); + db.close(); +}); + +Deno.test('backfilled rows reach the month report — the point of the feature', () => { + // The whole path: a CSV fills a gap, existing rules categorize it, and the + // month that had no data now reports. Report totals only count categorized + // rows, so this exercises import -> rules -> report end to end. + const db = testDb(); + db.prepare("INSERT INTO categories (name, kind, created_at) VALUES ('Salary','income',?)").run( + new Date().toISOString() + ); + createRule(db, { + pattern: 'shell', + matchType: 'contains', + categoryId: 2, + createdByDid: 'did:plc:test' + }); + createRule(db, { + pattern: 'paycheck', + matchType: 'contains', + categoryId: 3, + createdByDid: 'did:plc:test' + }); + + if (monthlyReport(db, '2026-06').expenseTotalCents !== 0) throw new Error('June should start empty'); + + const id = draft(db, 'Date,Description,Amount\n2026-06-15,SHELL OIL,-40.00\n2026-06-20,ACME PAYCHECK,1000.00\n'); + commitImport(db, id); + + const june = monthlyReport(db, '2026-06'); + if (june.expenseTotalCents !== -4000) { + throw new Error(`imported expense missing from the report: ${june.expenseTotalCents}`); + } + if (june.incomeTotalCents !== 100000) { + throw new Error(`imported income missing from the report: ${june.incomeTotalCents}`); + } + + // And they leave again with an undo. + undoImport(db, id); + const after = monthlyReport(db, '2026-06'); + if (after.expenseTotalCents !== 0 || after.incomeTotalCents !== 0) { + throw new Error('undone rows must leave the report'); + } + db.close(); +}); + +Deno.test('an import writes no snapshots, so net worth is untouched', () => { + const db = testDb(); + const before = netWorthSeries(db).length; + commitImport(db, draft(db, 'Date,Description,Amount\n2026-06-15,Shell,-40.00\n')); + if (netWorthSeries(db).length !== before) { + throw new Error('an import must not add points to the net worth series'); + } + db.close(); +}); + +Deno.test('preview states the range and the counts', () => { + const db = testDb(); + const id = draft( + db, + 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n2026-07-02,Amazon,-52.31\nbad,Row,-1.00\n' + ); + const preview = previewImport(db, id); + if (preview.rangeStart !== Math.floor(Date.UTC(2026, 5, 1, 12) / 1000)) { + throw new Error('wrong start'); + } + if (preview.rangeEnd !== JULY_02_NOON) throw new Error('wrong end'); + if (preview.newCount !== 1) throw new Error(`newCount ${preview.newCount}`); + if (preview.flaggedCount !== 1) throw new Error(`flaggedCount ${preview.flaggedCount}`); + if (preview.errors.length !== 1) throw new Error('the bad row should be reported'); + db.close(); +}); + +Deno.test('a draft cannot be committed twice', () => { + const db = testDb(); + const id = draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n'); + commitImport(db, id); + let threw = false; + try { + commitImport(db, id); + } catch { + threw = true; + } + if (!threw) throw new Error('expected a committed import to refuse a second commit'); + db.close(); +}); + +Deno.test('undo removes exactly that import and spares synced rows', () => { + const db = testDb(); + const id = draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n2026-06-02,Coffee,-4.75\n'); + commitImport(db, id); + + const removed = undoImport(db, id); + if (removed !== 2) throw new Error(`expected 2 removed, got ${removed}`); + + const live = db + .prepare('SELECT COUNT(*) AS n FROM transactions WHERE import_id = ? AND removed_at IS NULL') + .get(id) as { n: number }; + if (live.n !== 0) throw new Error('undo should remove every row of the import'); + + const synced = db + .prepare("SELECT removed_at FROM transactions WHERE sfin_id='sfin-amazon'") + .get() as { removed_at: string | null }; + if (synced.removed_at !== null) throw new Error('undo must not touch synced rows'); + + const status = db.prepare('SELECT status FROM imports WHERE id = ?').get(id) as { status: string }; + if (status.status !== 'undone') throw new Error(`expected undone, got ${status.status}`); + db.close(); +}); + +Deno.test('event history survives undo', () => { + const db = testDb(); + const id = draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n'); + commitImport(db, id); + + const txn = db.prepare("SELECT id FROM transactions WHERE description='Shell'").get() as { + id: number; + }; + categorizeManually(db, txn.id, 2, 'did:plc:test'); + undoImport(db, id); + + const events = db + .prepare('SELECT COUNT(*) AS n FROM categorization_events WHERE transaction_id = ?') + .get(txn.id) as { n: number }; + if (events.n === 0) throw new Error('the append-only event log must outlive the rows'); + db.close(); +}); + +Deno.test('re-import after a corrected mapping does not resurrect the old rows', () => { + const db = testDb(); + // A day-first file misread as month-first: June 7 instead of July 6. + const csv = 'Date,Description,Amount\n06/07/2026,Shell,-40.00\n'; + const wrong = createDraftImport(db, 'chk', 'bank.csv', csv); + saveMapping(db, wrong, { ...MAPPING, dateFormat: 'mdy' }); + commitImport(db, wrong); + undoImport(db, wrong); + + const right = createDraftImport(db, 'chk', 'bank.csv', csv); + saveMapping(db, right, { ...MAPPING, dateFormat: 'dmy' }); + const result = commitImport(db, right); + if (result.inserted !== 1) throw new Error(`expected a fresh insert, got ${result.inserted}`); + + const live = db + .prepare("SELECT posted FROM transactions WHERE removed_at IS NULL AND import_id IS NOT NULL") + .all() as { posted: number }[]; + if (live.length !== 1) throw new Error(`expected exactly 1 live imported row, got ${live.length}`); + if (live[0].posted !== Math.floor(Date.UTC(2026, 6, 6, 12) / 1000)) { + throw new Error('the corrected date should be July 6'); + } + db.close(); +}); + +Deno.test('undo then re-import with the same mapping revives rather than duplicating', () => { + const db = testDb(); + const csv = 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n'; + const first = draft(db, csv); + commitImport(db, first); + undoImport(db, first); + + const second = draft(db, csv); + const result = commitImport(db, second); + if (result.revived !== 1) throw new Error(`expected 1 revived, got ${result.revived}`); + + const rows = db + .prepare("SELECT import_id FROM transactions WHERE description='Shell' AND removed_at IS NULL") + .all() as { import_id: number }[]; + if (rows.length !== 1) throw new Error(`expected 1 live row, got ${rows.length}`); + if (rows[0].import_id !== second) throw new Error('the revived row should belong to the new import'); + db.close(); +}); + +Deno.test('only a committed import can be undone', () => { + const db = testDb(); + const id = draft(db, 'Date,Description,Amount\n2026-06-01,Shell,-40.00\n'); + let threw = false; + try { + undoImport(db, id); + } catch { + threw = true; + } + if (!threw) throw new Error('expected undoing a draft to be refused'); + db.close(); +}); + +Deno.test('JULY_03 nearby row exercises the window boundary', () => { + // Guards the constant itself: a row exactly one day out must still flag. + const db = testDb(); + const { candidates } = normalizeCsv( + `Date,Description,Amount\n2026-07-03,Amazon,-52.31\n`, + MAPPING, + 'chk' + ); + const [result] = findPotentialDuplicates(db, 'chk', candidates); + if (result.candidate.posted !== JULY_03_NOON) throw new Error('fixture drift'); + if (result.status !== 'flagged') throw new Error('boundary row should flag'); + db.close(); +}); diff --git a/src/lib/server/services/imports.ts b/src/lib/server/services/imports.ts new file mode 100644 index 0000000..d5f9c54 --- /dev/null +++ b/src/lib/server/services/imports.ts @@ -0,0 +1,392 @@ +// Lifecycle of a CSV backfill import: archive -> map -> review -> commit -> undo. +// The DB-touching half of CSV import; csv-import.ts holds the pure parsing, the +// same way sync.ts and normalize.ts are split. +// +// An import is an ADDITIVE writer. It only ever inserts (or revives a row it +// previously removed). It never reconciles pending transactions, never sweeps, +// never mutates a synced row, never writes balance snapshots, and never touches +// account state — all of that is exclusively sync's authority. + +import type { DatabaseSync } from 'node:sqlite'; +import { normalizeCsv, type CsvCandidate, type CsvMapping } from './csv-import.ts'; +import { applyRulesToUncategorized } from './rules.ts'; + +/** + * A CSV's date is often the transaction date while SimpleFIN's posted date trails + * it by a day, so a strict same-day check would miss the very rows most at risk of + * being double-counted. + * + * Measured in whole UTC calendar days, not in seconds: a CSV row carries only a + * date (anchored at noon) while a synced row carries a real timestamp at whatever + * hour the bank posted it. Comparing raw seconds would make the window mean + * "within 24 hours", so whether two rows a calendar day apart matched would depend + * on the hours involved — noon-vs-midnight is 36 hours and would silently miss. + */ +const DUPLICATE_WINDOW_DAYS = 1; + +/** Whole days since the epoch, UTC. Bank data is post-1970, so truncation is floor. */ +const UTC_DAY = `CAST(COALESCE(posted, transacted_at) / 86400 AS INTEGER)`; + +export type ImportStatus = 'draft' | 'committed' | 'undone'; + +export interface ImportRecord { + id: number; + accountId: string; + filename: string | null; + uploadedAt: string; + payload: string; + mapping: CsvMapping | null; + decisions: Record; + status: ImportStatus; + committedAt: string | null; + undoneAt: string | null; +} + +export interface ExistingMatch { + id: number; + effectiveAt: number | null; + amountCents: number; + description: string; + payee: string | null; + source: 'synced' | 'imported'; +} + +/** `already-present` rows are never surfaced — identity already made them a no-op. */ +export type CandidateStatus = 'new' | 'already-present' | 'flagged'; + +export interface ClassifiedCandidate { + candidate: CsvCandidate; + status: CandidateStatus; + match: ExistingMatch | null; +} + +interface ImportRow { + id: number; + account_id: string; + filename: string | null; + uploaded_at: string; + payload: string; + mapping: string | null; + decisions: string | null; + status: ImportStatus; + committed_at: string | null; + undone_at: string | null; +} + +function hydrate(row: ImportRow): ImportRecord { + return { + id: row.id, + accountId: row.account_id, + filename: row.filename, + uploadedAt: row.uploaded_at, + payload: row.payload, + mapping: row.mapping ? (JSON.parse(row.mapping) as CsvMapping) : null, + decisions: row.decisions ? (JSON.parse(row.decisions) as Record) : {}, + status: row.status, + committedAt: row.committed_at, + undoneAt: row.undone_at + }; +} + +/** Archive the bytes before anything is parsed, so nothing is lost to a bad mapping. */ +export function createDraftImport( + db: DatabaseSync, + accountId: string, + filename: string | null, + payload: string +): number { + const account = db + .prepare("SELECT id, state FROM accounts WHERE id = ? AND state != 'HIDDEN'") + .get(accountId) as { id: string } | undefined; + if (!account) throw new Error(`Unknown or hidden account: ${accountId}`); + + const result = db + .prepare( + `INSERT INTO imports (account_id, filename, uploaded_at, payload, status) + VALUES (?, ?, ?, ?, 'draft')` + ) + .run(accountId, filename, new Date().toISOString(), payload); + return Number(result.lastInsertRowid); +} + +export function getImport(db: DatabaseSync, importId: number): ImportRecord | null { + const row = db.prepare('SELECT * FROM imports WHERE id = ?').get(importId) as + | ImportRow + | undefined; + return row ? hydrate(row) : null; +} + +export function listImports(db: DatabaseSync): (ImportRecord & { rowCount: number })[] { + const rows = db + .prepare( + `SELECT i.*, ( + SELECT COUNT(*) FROM transactions t + WHERE t.import_id = i.id AND t.removed_at IS NULL + ) AS row_count + FROM imports i ORDER BY i.id DESC` + ) + .all() as unknown as (ImportRow & { row_count: number })[]; + return rows.map((r) => ({ ...hydrate(r), rowCount: r.row_count })); +} + +function requireDraft(db: DatabaseSync, importId: number): ImportRecord { + const record = getImport(db, importId); + if (!record) throw new Error(`Unknown import: ${importId}`); + if (record.status !== 'draft') throw new Error(`Import ${importId} is already ${record.status}.`); + return record; +} + +export function saveMapping(db: DatabaseSync, importId: number, mapping: CsvMapping): void { + requireDraft(db, importId); + db.prepare('UPDATE imports SET mapping = ? WHERE id = ?').run(JSON.stringify(mapping), importId); +} + +export function saveDecisions( + db: DatabaseSync, + importId: number, + decisions: Record +): void { + requireDraft(db, importId); + db.prepare('UPDATE imports SET decisions = ? WHERE id = ?').run( + JSON.stringify(decisions), + importId + ); +} + +/** + * Classify each candidate against what the account already holds. + * + * Matching is on amount and date only. The two sources render the same merchant + * differently — SimpleFIN gives the bank's raw string, the CSV its own — so + * matching on description would catch almost nothing while appearing to work. + * Description is evidence for the human, not a filter. + * + * Matches are consumed: if the file has two identical coffees and the account + * already holds one, exactly one candidate is flagged and the other is new. + */ +export function findPotentialDuplicates( + db: DatabaseSync, + accountId: string, + candidates: CsvCandidate[] +): ClassifiedCandidate[] { + const existing = db.prepare( + 'SELECT id FROM transactions WHERE account_id = ? AND sfin_id = ? AND removed_at IS NULL' + ); + const nearby = db.prepare( + `SELECT id, COALESCE(posted, transacted_at) AS effective_at, amount_cents, + description, payee, import_id + FROM transactions + WHERE account_id = ? AND removed_at IS NULL AND amount_cents = ? + AND ${UTC_DAY} BETWEEN ? AND ? + ORDER BY id` + ); + + const consumed = new Set(); + return candidates.map((candidate) => { + if (existing.get(accountId, candidate.syntheticId)) { + return { candidate, status: 'already-present' as const, match: null }; + } + + const day = Math.floor(candidate.posted / 86400); + const rows = nearby.all( + accountId, + candidate.amountCents, + day - DUPLICATE_WINDOW_DAYS, + day + DUPLICATE_WINDOW_DAYS + ) as { + id: number; + effective_at: number | null; + amount_cents: number; + description: string; + payee: string | null; + import_id: number | null; + }[]; + + const hit = rows.find((r) => !consumed.has(r.id)); + if (!hit) return { candidate, status: 'new' as const, match: null }; + + consumed.add(hit.id); + return { + candidate, + status: 'flagged' as const, + match: { + id: hit.id, + effectiveAt: hit.effective_at, + amountCents: hit.amount_cents, + description: hit.description, + payee: hit.payee, + source: hit.import_id === null ? ('synced' as const) : ('imported' as const) + } + }; + }); +} + +export interface ImportPreview { + classified: ClassifiedCandidate[]; + errors: string[]; + /** Unix seconds; null when nothing parsed. */ + rangeStart: number | null; + rangeEnd: number | null; + newCount: number; + alreadyPresentCount: number; + flaggedCount: number; +} + +/** + * Re-parse the archived bytes with the stored mapping. Pure with respect to the + * archive: every wizard step can call this instead of holding state elsewhere. + */ +export function previewImport(db: DatabaseSync, importId: number): ImportPreview { + const record = getImport(db, importId); + if (!record) throw new Error(`Unknown import: ${importId}`); + if (!record.mapping) throw new Error(`Import ${importId} has no mapping yet.`); + + const { candidates, errors } = normalizeCsv(record.payload, record.mapping, record.accountId); + const classified = findPotentialDuplicates(db, record.accountId, candidates); + const dates = candidates.map((c) => c.posted); + + return { + classified, + errors, + rangeStart: dates.length ? Math.min(...dates) : null, + rangeEnd: dates.length ? Math.max(...dates) : null, + newCount: classified.filter((c) => c.status === 'new').length, + alreadyPresentCount: classified.filter((c) => c.status === 'already-present').length, + flaggedCount: classified.filter((c) => c.status === 'flagged').length + }; +} + +export interface CommitResult { + inserted: number; + revived: number; + skipped: number; + alreadyPresent: number; + ruleCategorized: number; +} + +/** + * Insert-only. Deliberately shares nothing with ingestTransactions: that function + * treats its feed as authoritative for the account, and its stale-pending sweep + * would soft-delete every pending row absent from a backfill CSV. + */ +export function commitImport(db: DatabaseSync, importId: number): CommitResult { + const record = requireDraft(db, importId); + if (!record.mapping) throw new Error(`Import ${importId} has no mapping yet.`); + + const { candidates } = normalizeCsv(record.payload, record.mapping, record.accountId); + const classified = findPotentialDuplicates(db, record.accountId, candidates); + const now = new Date().toISOString(); + + let inserted = 0; + let revived = 0; + let skipped = 0; + let alreadyPresent = 0; + + db.exec('BEGIN'); + try { + const findAny = db.prepare( + 'SELECT id FROM transactions WHERE account_id = ? AND sfin_id = ?' + ); + const revive = db.prepare( + `UPDATE transactions SET removed_at = NULL, import_id = ?, posted = ?, transacted_at = ?, + amount_cents = ?, description = ?, payee = ?, memo = ?, pending = 0 + WHERE id = ?` + ); + const insert = db.prepare( + `INSERT INTO transactions + (account_id, sfin_id, posted, transacted_at, amount_cents, description, payee, memo, + pending, import_id, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, 0, ?, ?)` + ); + + for (const { candidate, status } of classified) { + if (status === 'already-present') { + alreadyPresent++; + continue; + } + // Flagged rows default to skip: where both sources claim a transaction the + // synced row is strictly better, so declining the CSV's copy loses nothing. + if (status === 'flagged' && record.decisions[candidate.syntheticId] !== 'keep') { + skipped++; + continue; + } + + const prior = findAny.get(record.accountId, candidate.syntheticId) as + | { id: number } + | undefined; + if (prior) { + // Only reachable for a row removed by an earlier undo — a live row would + // have classified as already-present. + revive.run( + importId, + candidate.posted, + candidate.transactedAt, + candidate.amountCents, + candidate.description, + candidate.payee, + candidate.memo, + prior.id + ); + revived++; + } else { + insert.run( + record.accountId, + candidate.syntheticId, + candidate.posted, + candidate.transactedAt, + candidate.amountCents, + candidate.description, + candidate.payee, + candidate.memo, + importId, + now + ); + inserted++; + } + } + + db.prepare("UPDATE imports SET status = 'committed', committed_at = ? WHERE id = ?").run( + now, + importId + ); + db.exec('COMMIT'); + } catch (err) { + db.exec('ROLLBACK'); + throw err; + } + + // Outside the transaction: appendCategorizationEvent manages its own. Backfilled + // history categorizes itself against rules that already exist, with the ordinary + // `rule` event source — an import introduces no new kind of provenance. + const ruleCategorized = applyRulesToUncategorized(db); + return { inserted, revived, skipped, alreadyPresent, ruleCategorized }; +} + +/** + * Reverse an import as a unit. Soft-removes: categorization events reference + * transaction ids and the event log is append-only, so history outlives the rows. + */ +export function undoImport(db: DatabaseSync, importId: number): number { + const record = getImport(db, importId); + if (!record) throw new Error(`Unknown import: ${importId}`); + if (record.status !== 'committed') { + throw new Error(`Only a committed import can be undone; ${importId} is ${record.status}.`); + } + + const now = new Date().toISOString(); + db.exec('BEGIN'); + try { + const result = db + .prepare('UPDATE transactions SET removed_at = ? WHERE import_id = ? AND removed_at IS NULL') + .run(now, importId); + db.prepare("UPDATE imports SET status = 'undone', undone_at = ? WHERE id = ?").run( + now, + importId + ); + db.exec('COMMIT'); + return Number(result.changes); + } catch (err) { + db.exec('ROLLBACK'); + throw err; + } +} diff --git a/src/lib/server/services/ledger.test.ts b/src/lib/server/services/ledger.test.ts index f75e74b..3249728 100644 --- a/src/lib/server/services/ledger.test.ts +++ b/src/lib/server/services/ledger.test.ts @@ -68,6 +68,92 @@ Deno.test('ledger excludes hidden accounts and soft-removed rows by default', () db.close(); }); +/** Add a committed import plus one backfilled row. Kept out of testDb() so the + * existing count assertions stay meaningful. */ +function withImport(db: DatabaseSync): number { + const now = new Date().toISOString(); + const importId = Number( + db + .prepare( + `INSERT INTO imports (account_id, filename, uploaded_at, payload, status, committed_at) + VALUES ('chk', 'chase-2026.csv', ?, 'Date,Description,Amount', 'committed', ?)` + ) + .run(now, now).lastInsertRowid + ); + db.prepare( + `INSERT INTO transactions + (account_id, sfin_id, posted, amount_cents, description, pending, import_id, created_at) + VALUES ('chk', 'csv:abc123', ?, -1500, 'BACKFILLED LUNCH', 0, ?, ?)` + ).run(JUNE_10, importId, now); + return importId; +} + +Deno.test('source filter separates imported from synced', () => { + const db = testDb(); + withImport(db); + + const imported = listLedger(db, { source: 'imported' }); + if (imported.length !== 1 || imported[0].description !== 'BACKFILLED LUNCH') { + throw new Error(`imported filter wrong: ${imported.map((r) => r.description).join(',')}`); + } + + const synced = listLedger(db, { source: 'synced' }); + if (synced.some((r) => r.description === 'BACKFILLED LUNCH')) { + throw new Error('imported row leaked into the synced filter'); + } + if (synced.length !== 3) throw new Error(`expected the 3 synced rows, got ${synced.length}`); + + // Unfiltered shows both — origin is an on-demand question, not a default lens. + if (listLedger(db).length !== 4) throw new Error('unfiltered should show every row'); + db.close(); +}); + +Deno.test('source filter composes with other filters', () => { + const db = testDb(); + withImport(db); + + if (listLedger(db, { source: 'imported', month: '2026-06' }).length !== 1) { + throw new Error('source + month wrong'); + } + if (listLedger(db, { source: 'imported', month: '2026-07' }).length !== 0) { + throw new Error('source + month should exclude other months'); + } + if (listLedger(db, { source: 'imported', accountId: 'chk' }).length !== 1) { + throw new Error('source + account wrong'); + } + if (listLedger(db, { source: 'synced', month: '2026-06' }).length !== 1) { + throw new Error('source + month should still find the synced June row'); + } + if (listLedger(db, { source: 'imported', q: 'lunch' }).length !== 1) { + throw new Error('source + search wrong'); + } + if (listLedger(db, { source: 'synced', q: 'lunch' }).length !== 0) { + throw new Error('source + search should exclude the imported match'); + } + if (listLedger(db, { source: 'imported', category: 'uncategorized' }).length !== 1) { + throw new Error('source + category wrong'); + } + db.close(); +}); + +Deno.test('ledger row carries its origin', () => { + const db = testDb(); + const importId = withImport(db); + + const [row] = listLedger(db, { source: 'imported' }); + if (row.source !== 'imported') throw new Error('source not set'); + if (row.importId !== importId) throw new Error('importId not joined'); + if (row.importFilename !== 'chase-2026.csv') throw new Error('filename not joined'); + if (!row.importedAt) throw new Error('importedAt not joined'); + + const synced = listLedger(db, { source: 'synced' })[0]; + if (synced.source !== 'synced') throw new Error('synced row mislabeled'); + if (synced.importId !== null || synced.importFilename !== null) { + throw new Error('synced row should carry no import origin'); + } + db.close(); +}); + Deno.test('ledger filters: month, category, uncategorized, pending, account', () => { const db = testDb(); const june = listLedger(db, { month: '2026-06' }); diff --git a/src/lib/server/services/ledger.ts b/src/lib/server/services/ledger.ts index eeb20a9..9f2c17a 100644 --- a/src/lib/server/services/ledger.ts +++ b/src/lib/server/services/ledger.ts @@ -9,6 +9,11 @@ export interface LedgerFilters { month?: string; /** true = only pending, false = only posted, undefined = both */ pending?: boolean; + /** + * Where the row came from: synced from a connection, or backfilled from a CSV. + * A separate axis from `provenance`, which is about who chose the category. + */ + source?: 'synced' | 'imported'; /** * Case-insensitive substring search over description, payee, memo, and the * effective rule-applied display name — search matches what the ledger shows. @@ -40,6 +45,15 @@ export interface LedgerRow { provenanceRulePattern: string | null; provenanceActorHandle: string | null; provenanceActorDid: string | null; + /** + * Origin of the row itself. Shown in the history panel rather than on the row: + * within a backfilled date range every row is imported, so a per-row mark would + * carry no information exactly where it is densest. + */ + source: 'synced' | 'imported'; + importId: number | null; + importFilename: string | null; + importedAt: string | null; } export function monthRange(month: string): { start: number; end: number } { @@ -77,6 +91,9 @@ export function listLedger(db: DatabaseSync, filters: LedgerFilters = {}): Ledge where.push('t.pending = ?'); params.push(filters.pending ? 1 : 0); } + if (filters.source) { + where.push(filters.source === 'imported' ? 't.import_id IS NOT NULL' : 't.import_id IS NULL'); + } if (filters.q?.trim()) { // r is the categorizing rule of the latest event (joined below), so the // overlay display name is searchable — search finds what the user sees. @@ -96,7 +113,8 @@ export function listLedger(db: DatabaseSync, filters: LedgerFilters = {}): Ledge COALESCE(r.display_name, t.payee, t.description) AS display_label, t.category_id, c.name AS category_name, e.source AS prov_source, r.pattern AS prov_pattern, u.handle AS prov_handle, - e.actor_did AS prov_actor_did + e.actor_did AS prov_actor_did, + t.import_id, i.filename AS import_filename, i.committed_at AS imported_at FROM transactions t JOIN accounts a ON a.id = t.account_id LEFT JOIN categories c ON c.id = t.category_id @@ -105,6 +123,7 @@ export function listLedger(db: DatabaseSync, filters: LedgerFilters = {}): Ledge ) LEFT JOIN rules r ON r.id = e.rule_id LEFT JOIN users u ON u.did = e.actor_did + LEFT JOIN imports i ON i.id = t.import_id WHERE ${where.join(' AND ')} ORDER BY t.pending DESC, effective_at DESC, t.id DESC LIMIT ?` @@ -125,6 +144,10 @@ export function listLedger(db: DatabaseSync, filters: LedgerFilters = {}): Ledge pending: r.pending === 1, categoryId: r.category_id as number | null, categoryName: r.category_name as string | null, + source: r.import_id == null ? ('synced' as const) : ('imported' as const), + importId: r.import_id as number | null, + importFilename: r.import_filename as string | null, + importedAt: r.imported_at as string | null, provenance: (r.prov_source as EventSource | null) ?? null, provenanceRulePattern: r.prov_pattern as string | null, provenanceActorHandle: r.prov_handle as string | null, diff --git a/src/lib/server/services/sync.test.ts b/src/lib/server/services/sync.test.ts index 3804887..d770823 100644 --- a/src/lib/server/services/sync.test.ts +++ b/src/lib/server/services/sync.test.ts @@ -2,6 +2,7 @@ import { openDatabase } from '../db.ts'; import { runSync } from './sync.ts'; import { categorizeManually } from './categorization.ts'; +import { commitImport, createDraftImport, saveMapping } from './imports.ts'; import type { DatabaseSync } from 'node:sqlite'; const MIGRATIONS_DIR = new URL('../../../../migrations', import.meta.url).pathname.replace( @@ -49,6 +50,47 @@ function count(db: DatabaseSync, sql: string): number { return (db.prepare(sql).get() as { n: number }).n; } +Deno.test('a sync after a backfill leaves the imported rows alone', async () => { + // The hazard this guards: ingestTransactions treats its feed as authoritative + // for the account and soft-removes anything pending that the feed omits. A + // backfill CSV appears in no feed, ever. Imported rows are posted (pending = 0), + // which is what keeps them outside every branch of sync's authority — a + // structural property worth pinning down rather than rediscovering. + const db = testDb(); + const pendingTxn = { id: 'p1', posted: null, amount: '-5.00', description: 'PENDING COFFEE' }; + await runSync(db, fakeFetch(payloadWith([pendingTxn]))); + + // Backfill straight into the discovered account, well before the feed's reach. + const importId = createDraftImport(db, 'act-1', 'old.csv', 'Date,Description,Amount\n2023-02-01,ANCIENT LUNCH,-9.99\n'); + saveMapping(db, importId, { + date: 'Date', + dateFormat: 'iso', + amountMode: 'signed', + amount: 'Amount', + description: 'Description' + }); + commitImport(db, importId); + + // A later sync of the same connection: the feed still knows nothing of 2023. + await runSync(db, fakeFetch(payloadWith([pendingTxn, { id: 't9', posted: NOW_S, amount: '-1.00', description: 'NEW THING' }]))); + + const imported = db + .prepare("SELECT removed_at, pending FROM transactions WHERE description = 'ANCIENT LUNCH'") + .get() as { removed_at: string | null; pending: number }; + if (imported.removed_at !== null) throw new Error('a sync must not remove backfilled rows'); + if (imported.pending !== 0) throw new Error('imported rows must stay posted'); + + if (count(db, "SELECT COUNT(*) n FROM transactions WHERE description = 'ANCIENT LUNCH'") !== 1) { + throw new Error('a sync must not duplicate a backfilled row'); + } + // And the account is still healthy: the backfill did not make it look absent. + const account = db.prepare("SELECT state FROM accounts WHERE id = 'act-1'").get() as { + state: string; + }; + if (account.state === 'INACTIVE') throw new Error('the account should not have gone inactive'); + db.close(); +}); + Deno.test('sync archives raw payload, discovers account as NEW, snapshots balance', async () => { const db = testDb(); const payload = payloadWith([ diff --git a/src/routes/(app)/ledger/+page.server.ts b/src/routes/(app)/ledger/+page.server.ts index 19d8224..d7906e2 100644 --- a/src/routes/(app)/ledger/+page.server.ts +++ b/src/routes/(app)/ledger/+page.server.ts @@ -22,11 +22,15 @@ export const load: PageServerLoad = ({ url }) => { else if (pending === '0') filters.pending = false; const q = url.searchParams.get('q')?.trim(); if (q) filters.q = q; + const source = url.searchParams.get('source'); + if (source === 'synced' || source === 'imported') filters.source = source; const historyId = url.searchParams.get('history'); let history: { id: number; events: ReturnType; + /** Where the row itself came from — a different axis from who categorized it. */ + origin: { source: 'synced' | 'imported'; filename: string | null; importedAt: string | null }; bankRecord: { description: string; payee: string | null; @@ -38,7 +42,13 @@ export const load: PageServerLoad = ({ url }) => { if (historyId) { const id = Number(historyId); const txn = db - .prepare('SELECT description, payee, memo, sfin_id, extra FROM transactions WHERE id = ?') + .prepare( + `SELECT t.description, t.payee, t.memo, t.sfin_id, t.extra, + t.import_id, i.filename AS import_filename, i.committed_at AS imported_at + FROM transactions t + LEFT JOIN imports i ON i.id = t.import_id + WHERE t.id = ?` + ) .get(id) as | { description: string; @@ -46,11 +56,19 @@ export const load: PageServerLoad = ({ url }) => { memo: string | null; sfin_id: string; extra: string | null; + import_id: number | null; + import_filename: string | null; + imported_at: string | null; } | undefined; history = { id, events: listEvents(db, id), + origin: { + source: txn?.import_id == null ? 'synced' : 'imported', + filename: txn?.import_filename ?? null, + importedAt: txn?.imported_at ?? null + }, bankRecord: txn ? { description: txn.description, @@ -76,6 +94,7 @@ export const load: PageServerLoad = ({ url }) => { category: category ?? '', month: month ?? '', pending: pending ?? '', + source: source ?? '', q: q ?? '' } }; diff --git a/src/routes/(app)/ledger/+page.svelte b/src/routes/(app)/ledger/+page.svelte index b991795..0c4d50f 100644 --- a/src/routes/(app)/ledger/+page.svelte +++ b/src/routes/(app)/ledger/+page.svelte @@ -56,6 +56,7 @@ } const timeFmt = new Intl.DateTimeFormat(undefined, { dateStyle: 'medium', timeStyle: 'short' }); + const dayFmt = new Intl.DateTimeFormat(undefined, { dateStyle: 'medium' });

Ledger

@@ -112,6 +113,15 @@ + + {#if data.filters.q} clear search + + {/if} + + {#if form?.message}{/if} + + +
+

Past imports

+ + {#if data.imports.length === 0} +

No imports yet.

+ {:else} +
    + {#each data.imports as record (record.id)} +
  • +
    + {record.filename ?? 'unnamed.csv'} + + {accountLabel(record.accountId)} · + {#if record.status === 'committed'} + {record.rowCount} transaction{record.rowCount === 1 ? '' : 's'} · + imported {record.committedAt ? dateFmt.format(new Date(record.committedAt)) : ''} + {:else if record.status === 'undone'} + undone {record.undoneAt ? dateFmt.format(new Date(record.undoneAt)) : ''} + {:else} + not finished · started {dateFmt.format(new Date(record.uploadedAt))} + {/if} + +
    + +
    + {#if record.status === 'draft'} + Resume + {:else if record.status === 'committed'} +
    { + const ok = confirm( + `Remove ${record.rowCount} transaction${record.rowCount === 1 ? '' : 's'} imported from ${record.filename ?? 'this file'}? Synced transactions are not affected.` + ); + if (!ok) e.preventDefault(); + }} + > + + +
    + {/if} +
    +
  • + {/each} +
+ {/if} +
+ + diff --git a/src/routes/(app)/settings/import/[id]/+page.server.ts b/src/routes/(app)/settings/import/[id]/+page.server.ts new file mode 100644 index 0000000..708fde8 --- /dev/null +++ b/src/routes/(app)/settings/import/[id]/+page.server.ts @@ -0,0 +1,158 @@ +import { error, fail } from '@sveltejs/kit'; +import { getDb } from '$lib/server/db'; +import { listAccounts } from '$lib/server/services/accounts'; +import { + commitImport, + getImport, + previewImport, + saveDecisions, + saveMapping +} from '$lib/server/services/imports'; +import { parseCsv, type CsvDateFormat, type CsvMapping } from '$lib/server/services/csv-import'; +import type { Actions, PageServerLoad } from './$types'; + +/** Enough to see a wrong column at a glance without rendering the whole file. */ +const SAMPLE_ROWS = 5; + +function readMapping(form: FormData): CsvMapping { + const value = (name: string) => { + const raw = String(form.get(name) ?? '').trim(); + return raw === '' ? null : raw; + }; + const date = value('date'); + const description = value('description'); + if (!date) throw new Error('Choose the column holding the date.'); + if (!description) throw new Error('Choose the column holding the description.'); + + const amountMode = String(form.get('amountMode') ?? 'signed') as CsvMapping['amountMode']; + if (amountMode === 'signed' && !value('amount')) { + throw new Error('Choose the column holding the amount.'); + } + if (amountMode === 'debit-credit' && !value('debit') && !value('credit')) { + throw new Error('Choose a debit column, a credit column, or both.'); + } + + return { + date, + transactedDate: value('transactedDate'), + dateFormat: (String(form.get('dateFormat') ?? 'mdy') as CsvDateFormat) ?? 'mdy', + amountMode, + amount: value('amount'), + debit: value('debit'), + credit: value('credit'), + description, + payee: value('payee'), + memo: value('memo') + }; +} + +export const load: PageServerLoad = ({ params }) => { + const db = getDb(); + const record = getImport(db, Number(params.id)); + if (!record) error(404, 'No such import.'); + + const account = listAccounts(db).find((a) => a.id === record.accountId); + const accountLabel = account ? (account.displayName ?? account.name) : record.accountId; + + const base = { + id: record.id, + filename: record.filename, + status: record.status, + accountId: record.accountId, + accountLabel, + committedAt: record.committedAt + }; + + // A committed or undone import is a record, not a wizard. + if (record.status !== 'draft') { + const counts = db + .prepare( + 'SELECT COUNT(*) AS n FROM transactions WHERE import_id = ? AND removed_at IS NULL' + ) + .get(record.id) as { n: number }; + return { ...base, headers: [], mapping: null, sample: [], preview: null, parseError: null, rowCount: counts.n }; + } + + let headers: string[] = []; + let parseError: string | null = null; + try { + headers = parseCsv(record.payload).headers; + } catch (err) { + parseError = err instanceof Error ? err.message : 'This file could not be read as CSV.'; + } + + let preview = null; + if (record.mapping && !parseError) { + try { + const full = previewImport(db, record.id); + preview = { + errors: full.errors, + rangeStart: full.rangeStart, + rangeEnd: full.rangeEnd, + newCount: full.newCount, + alreadyPresentCount: full.alreadyPresentCount, + flaggedCount: full.flaggedCount, + sample: full.classified.slice(0, SAMPLE_ROWS).map((c) => ({ + posted: c.candidate.posted, + amountCents: c.candidate.amountCents, + description: c.candidate.description + })), + flagged: full.classified + .filter((c) => c.status === 'flagged') + .map((c) => ({ + syntheticId: c.candidate.syntheticId, + posted: c.candidate.posted, + amountCents: c.candidate.amountCents, + description: c.candidate.description, + match: c.match + })) + }; + } catch (err) { + parseError = err instanceof Error ? err.message : 'This file could not be read as CSV.'; + } + } + + return { + ...base, + headers, + mapping: record.mapping, + preview, + parseError, + rowCount: 0 + }; +}; + +export const actions: Actions = { + mapping: async ({ request, params }) => { + const form = await request.formData(); + try { + saveMapping(getDb(), Number(params.id), readMapping(form)); + } catch (err) { + return fail(400, { message: err instanceof Error ? err.message : 'Mapping failed.' }); + } + return { message: 'Mapping updated.' }; + }, + + commit: async ({ request, params }) => { + const form = await request.formData(); + const db = getDb(); + const importId = Number(params.id); + + // Only checked rows are kept. An unchecked (or absent) flagged row is skipped: + // where both sources claim a transaction the synced row is the better record. + const decisions: Record = {}; + for (const id of form.getAll('keep')) decisions[String(id)] = 'keep'; + + try { + saveDecisions(db, importId, decisions); + const result = commitImport(db, importId); + const parts = [`Imported ${result.inserted + result.revived}`]; + if (result.skipped) parts.push(`skipped ${result.skipped} possible duplicate${result.skipped === 1 ? '' : 's'}`); + if (result.alreadyPresent) parts.push(`${result.alreadyPresent} already present`); + if (result.ruleCategorized) parts.push(`${result.ruleCategorized} categorized by rules`); + return { message: `${parts.join(', ')}.` }; + } catch (err) { + return fail(400, { message: err instanceof Error ? err.message : 'Import failed.' }); + } + } +}; diff --git a/src/routes/(app)/settings/import/[id]/+page.svelte b/src/routes/(app)/settings/import/[id]/+page.svelte new file mode 100644 index 0000000..9338e1f --- /dev/null +++ b/src/routes/(app)/settings/import/[id]/+page.svelte @@ -0,0 +1,346 @@ + + +

Import CSV

+ +

← All imports

+ +{#if data.status !== 'draft'} +
+

{data.filename ?? 'unnamed.csv'}

+ {#if data.status === 'committed'} +

+ Imported into {data.accountLabel}. {data.rowCount} transaction{data.rowCount === 1 + ? '' + : 's'} are in the ledger from this file. +

+ {#if form?.message}{/if} +

+ + See these transactions in the ledger + +

+ {:else} +

+ This import was undone. Its transactions are no longer in the ledger. +

+ {/if} +
+{:else if data.parseError} +
+

{data.filename ?? 'unnamed.csv'}

+

{data.parseError}

+

+ The file is saved as uploaded. Export it again as CSV and start a new import. +

+
+{:else} +
+

Columns

+

+ {data.filename ?? 'unnamed.csv'} → {data.accountLabel} +

+ +
+ + + + + + + + + + {#if amountMode === 'signed'} + + + {:else} + + + + + + {/if} + + + + + + + + + + + + + + + +
+ + {#if form?.message}{/if} +
+ + {#if !data.mapping} +
+

+ Quantum could not guess this file's columns. Choose them above to continue. +

+
+ {:else if data.preview} +
+

Preview

+ + {#if data.preview.rangeStart === null} +

No rows parsed. Check the columns above.

+ {:else} +

+ {dateFmt.format(new Date(data.preview.rangeStart * 1000))} to + {dateFmt.format(new Date(data.preview.rangeEnd! * 1000))} · + {data.preview.newCount} to import{#if data.preview.alreadyPresentCount}, {data.preview + .alreadyPresentCount} already here{/if}{#if data.preview.flaggedCount}, {data.preview + .flaggedCount} possibly already here{/if}. +

+ + + + {#each data.preview.sample as row, i (i)} + + + + + + {/each} + +
{formatDay(row.posted)}{row.description}{formatCents(row.amountCents)}
+

First rows as Quantum reads them. If the dates or amounts look wrong, fix the columns above.

+ {/if} + + {#if data.preview.errors.length > 0} +
+ + {data.preview.errors.length} row{data.preview.errors.length === 1 ? '' : 's'} could not be + read and will be left out + +
    + {#each data.preview.errors.slice(0, 20) as message (message)} +
  • {message}
  • + {/each} + {#if data.preview.errors.length > 20} +
  • …and {data.preview.errors.length - 20} more.
  • + {/if} +
+
+ {/if} +
+ +
+ {#if data.preview.flaggedCount > 0} +
+

Possible duplicates

+

+ These already look like transactions in {data.accountLabel}. They are left out unless you + say otherwise — the synced record is the better copy. Tick one only if it is genuinely a + separate transaction. +

+ +
    + {#each data.preview.flagged as row (row.syntheticId)} +
  • + +
  • + {/each} +
+
+ {/if} + +
+ +
+
+ {/if} +{/if} + +