From d04eb59c92c8563a32cf757c62aa31b69bdec76c Mon Sep 17 00:00:00 2001 From: Furox-Art <177975472+Furox-Art@users.noreply.github.com> Date: Sat, 19 Sep 2026 21:53:25 +0300 Subject: [PATCH 1/5] feat(export): chunked, resumable dumps with breathing intervals and R2 overflow Large databases crashed the legacy dump endpoint: it accumulated the entire dump in one in-memory string inside a single 30s request. This replaces it with a bounded, resumable dump engine: - ChunkedDumpEngine: rowid-batched table scans, time-budgeted cycles with breathing intervals between them, progress persisted in DO storage (crash/resume safe, no duplicate emission on resume) - Chunks spill to R2 when a bucket is bound, with DO-storage fallback and streaming reassembly - /export/dump stays synchronous for small DBs (backwards compatible); requests that hit the 30s window are upgraded to a job driven by DO alarms, with /export/dump/status polling - New /export/dump/status route and DO RPC (startDumpJob/dumpJobStatus) - 12 unit tests covering resume, chunking, R2, redaction-safe serialization and identifier injection guards --- src/do.ts | 83 +++++++ src/export/chunkedDump.test.ts | 305 ++++++++++++++++++++++++++ src/export/chunkedDump.ts | 389 +++++++++++++++++++++++++++++++++ src/export/dump.ts | 204 ++++++++++++++++- src/handler.ts | 36 ++- 5 files changed, 1004 insertions(+), 13 deletions(-) create mode 100644 src/export/chunkedDump.test.ts create mode 100644 src/export/chunkedDump.ts diff --git a/src/do.ts b/src/do.ts index b6bb2b6..4d0bfb6 100644 --- a/src/do.ts +++ b/src/do.ts @@ -1,4 +1,7 @@ import { DurableObject } from 'cloudflare:workers' +import { DUMP_STATE_KEY, type DumpState } from './export/chunkedDump' +import type { DataSource } from './types' +import type { StarbaseDBConfiguration } from './handler' export class StarbaseDBDurableObject extends DurableObject { // Durable storage for the SQL database @@ -72,6 +75,8 @@ export class StarbaseDBDurableObject extends DurableObject { deleteAlarm: this.deleteAlarm.bind(this), getStatistics: this.getStatistics.bind(this), executeQuery: this.executeQuery.bind(this), + startDumpJob: this.startDumpJob.bind(this), + dumpJobStatus: this.dumpJobStatus.bind(this), } } @@ -106,6 +111,28 @@ export class StarbaseDBDurableObject extends DurableObject { async alarm() { try { + // A chunked dump job in flight takes priority: resume its next + // bounded cycle (breathing interval already elapsed). The dump job + // always targets the internal source, which is this DO itself. + const dumpState = await this.storage.get(DUMP_STATE_KEY) + if (dumpState && !dumpState.completedAt) { + const { runDumpJob } = await import('./export/dump') + await runDumpJob( + { + storage: this.storage, + env: { + R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }) + .R2_DUMP_BUCKET, + }, + dataSource: this.dumpJobDataSource(), + config: { role: 'admin' } as StarbaseDBConfiguration, + setAlarm: (time, options) => this.setAlarm(time, options), + }, + new URLSearchParams() + ) + return + } + // Fetch all the tasks that are marked to emit an event for this cycle. const task = (await this.executeQuery({ sql: 'SELECT * FROM tmp_cron_tasks WHERE is_active = 1;', @@ -284,6 +311,62 @@ export class StarbaseDBDurableObject extends DurableObject { } } + /** + * Internal data source for dump jobs. Built in-process (no RPC hop): the + * engine's queries execute directly against this DO's SQLite storage. + */ + private dumpJobDataSource(): DataSource { + return { + source: 'internal', + rpc: this.init(), + } as unknown as DataSource + } + + /** + * Chunked dump job entry point, executed inside the Durable Object so it + * can drive bounded work cycles, persist progress, mirror chunks to R2 + * (when bound) and resume itself through the DO alarm. Takes only plain + * config/params so the RPC surface avoids circular DataSource typings. + */ + public async startDumpJob( + config: StarbaseDBConfiguration, + searchParams: Record + ): Promise { + const { runDumpJob } = await import('./export/dump') + return runDumpJob( + { + storage: this.storage, + env: { + // Optional binding; deployers add it to wrangler.toml when + // they want R2-backed dumps. Absent = storage-only mode. + R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }) + .R2_DUMP_BUCKET, + }, + dataSource: this.dumpJobDataSource(), + config, + setAlarm: (time, options) => this.setAlarm(time, options), + }, + new URLSearchParams(searchParams) + ) + } + + /** Status/fetch endpoint for a chunked dump job. */ + public async dumpJobStatus( + config: StarbaseDBConfiguration + ): Promise { + const { dumpJobStatus } = await import('./export/dump') + return dumpJobStatus({ + storage: this.storage, + env: { + R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }) + .R2_DUMP_BUCKET, + }, + dataSource: this.dumpJobDataSource(), + config, + setAlarm: (time, options) => this.setAlarm(time, options), + }) + } + public async executeQuery(opts: { sql: string params?: unknown[] diff --git a/src/export/chunkedDump.test.ts b/src/export/chunkedDump.test.ts new file mode 100644 index 0000000..bd9f6ca --- /dev/null +++ b/src/export/chunkedDump.test.ts @@ -0,0 +1,305 @@ +import { describe, it, expect, vi, beforeEach } from 'vitest' +import { + ChunkedDumpEngine, + DEFAULT_DUMP_OPTIONS, + isSafeIdentifier, + makeDumpFileName, + serializeRows, + DUMP_STATE_KEY, + DUMP_CHUNK_KEY, + type ChunkRecord, + type DumpOptions, + type DumpState, +} from './chunkedDump' +import { executeOperation } from './index' +import type { DataSource } from '../types' +import type { StarbaseDBConfiguration } from '../handler' + +vi.mock('./index', () => ({ + executeOperation: vi.fn(), +})) + +type StorageMap = Map + +const makeStorage = () => { + const map: StorageMap = new Map() + return { + get: vi.fn(async (key: string) => map.get(key) as T), + put: vi.fn(async (key: string, value: unknown) => { + map.set(key, value) + }), + delete: vi.fn(async (key: string) => { + map.delete(key) + }), + _map: map, + } +} + +const makeR2 = () => ({ + put: vi.fn(async () => undefined), + get: vi.fn(async () => null), +}) + +const makeDataSource = (): DataSource => + ({ source: 'internal', rpc: {} }) as unknown as DataSource + +const makeConfig = (): StarbaseDBConfiguration => ({ role: 'admin' }) + +const makeEngine = (overrides: Partial = {}) => { + const storage = makeStorage() + const r2 = makeR2() + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS, ...overrides } + ) + return { engine, storage, r2 } +} + +beforeEach(() => { + vi.clearAllMocks() +}) + +describe('identifier safety', () => { + it('accepts ordinary table names', () => { + expect(isSafeIdentifier('users')).toBe(true) + expect(isSafeIdentifier('order_items2')).toBe(true) + }) + + it('rejects injection attempts', () => { + expect(isSafeIdentifier('users; DROP TABLE users')).toBe(false) + expect(isSafeIdentifier('users"')).toBe(false) + expect(isSafeIdentifier('tmp_cache')).toBe(true) // syntactically safe + }) +}) + +describe('row serialization', () => { + it('escapes single quotes in strings', () => { + const { content } = serializeRows('users', [{ name: "O'Brien" }]) + expect(content).toContain("'O''Brien'") + expect(content).toContain('INSERT INTO "users"') + }) + + it('renders NULLs, numbers and booleans unquoted', () => { + const { content } = serializeRows('t', [ + { a: null, b: 1.5, c: true }, + ]) + expect(content).toContain('(NULL, 1.5, true)') + }) + + it('returns empty content for zero rows', () => { + const { content, rowCount } = serializeRows('t', []) + expect(content).toBe('') + expect(rowCount).toBe(0) + }) +}) + +describe('dump file naming', () => { + it('follows dump_YYYYMMDD-HHMMSS.sql in UTC', () => { + const name = makeDumpFileName(new Date('2024-01-01T17:00:00Z')) + expect(name).toBe('dump_20240101-170000.sql') + }) +}) + +describe('ChunkedDumpEngine cycles', () => { + const setupTables = (tables: string[], schemas: Record) => { + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes("type='table' AND name NOT LIKE")) { + return tables.map((name) => ({ name })) + } + if (sql.includes('SELECT sql FROM sqlite_master')) { + const match = /name='([^']+)'/.exec(sql) + const name = match?.[1] ?? '' + return schemas[name] ? [{ sql: schemas[name] }] : [] + } + // rowid-batched data fetch + return [] + }) + } + + it('completes a small dump in one cycle and writes chunks to R2', async () => { + setupTables(['users'], { users: 'CREATE TABLE users (id INTEGER)' }) + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes("name NOT LIKE 'tmp_%'")) { + return [{ name: 'users' }] + } + if (sql.includes('SELECT sql FROM sqlite_master')) { + return [{ sql: 'CREATE TABLE users (id INTEGER)' }] + } + if (sql.includes('WHERE rowid >')) { + const since = Number(/rowid > (\d+)/.exec(sql)?.[1] ?? 0) + return since === 0 + ? [ + { __rowid: 1, id: 1 }, + { __rowid: 2, id: 2 }, + ] + : [] + } + return [] + }) + + const { engine, storage, r2 } = makeEngine() + const state = await engine.startDump() + + expect(state.completedAt).toBeDefined() + expect(state.phase).toBe('complete') + expect(state.totalRows).toBe(2) + expect(state.chunkIndex).toBeGreaterThan(0) + expect(r2.put).toHaveBeenCalled() + // Progress persisted in DO storage for resumability. + const persisted = (await storage.get(DUMP_STATE_KEY)) as DumpState | undefined + expect(persisted?.completedAt).toBeDefined() + }) + + it('yields mid-dump when the cycle time budget is exhausted, then resumes', async () => { + setupTables(['users'], { users: 'CREATE TABLE users (id INTEGER)' }) + let call = 0 + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes("name NOT LIKE 'tmp_%'")) return [{ name: 'users' }] + if (sql.includes('SELECT sql FROM sqlite_master')) { + return [{ sql: 'CREATE TABLE users (id INTEGER)' }] + } + if (sql.includes('WHERE rowid >')) { + const since = Number(/rowid > (\d+)/.exec(sql)?.[1] ?? 0) + call++ + // Yield after every batch: each cycle emits exactly one row, + // and the stream is finite (3 rows total). + return since < 3 ? [{ __rowid: since + 1, id: since + 1 }] : [] + } + return [] + }) + + // 0ms budget: every data batch closes the cycle → resumable steps. + const { engine } = makeEngine({ cycleTimeBudgetMs: 0 }) + const first = await engine.startDump() + expect(first.completedAt).toBeUndefined() + expect(['schema', 'table-data']).toContain(first.phase) + + // Resume cycles until complete. + let state = first + for (let i = 0; i < 50 && !state.completedAt; i++) { + state = await engine.runCycle() + } + expect(state.completedAt).toBeDefined() + expect(state.totalRows).toBeGreaterThan(0) + }) + + it('filters unsafe table names out of the dump plan', async () => { + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes("name NOT LIKE 'tmp_%'")) { + return [ + { name: 'good_table' }, + { name: 'bad"; DROP TABLE users;--' }, + ] + } + if (sql.includes('SELECT sql FROM sqlite_master')) { + return [{ sql: 'CREATE TABLE good_table (id INTEGER)' }] + } + return [] + }) + + const { engine, storage } = makeEngine() + const state = await engine.startDump() + const persisted = (await storage.get(DUMP_STATE_KEY)) as DumpState | undefined + expect(persisted?.tables).toEqual(['good_table']) + expect(state.totalRows).toBe(0) + }) + + it('startDump resumes an in-progress dump instead of restarting', async () => { + setupTables(['users'], {}) + const { engine, storage } = makeEngine({ cycleTimeBudgetMs: 0 }) + // Seed an in-progress state. + const seed: DumpState = { + dumpId: 'dump_seed', + fileName: 'dump_seed.sql', + phase: 'table-data', + tables: ['users'], + tableIndex: 0, + lastFetchedRowId: null, + chunkRowOffset: 0, + bytesWritten: 0, + chunkIndex: 0, + startedAt: Date.now() - 1000, + updatedAt: Date.now() - 500, + totalRows: 0, + } + await storage.put(DUMP_STATE_KEY, seed) + + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes('WHERE rowid >')) return [] + return [] + }) + + const state = await engine.startDump() + // Must NOT have re-planned tables (phase kept, no new dumpId). + expect(state.dumpId).toBe('dump_seed') + expect(state.tables).toEqual(['users']) + }) + + it('assembleDump concatenates persisted chunks in order', async () => { + const { engine, storage } = makeEngine() + const state: DumpState = { + dumpId: 'dump_x', + fileName: 'dump_x.sql', + phase: 'complete', + tables: ['t'], + tableIndex: 1, + lastFetchedRowId: 3, + chunkRowOffset: 0, + bytesWritten: 10, + chunkIndex: 2, + startedAt: 1, + updatedAt: 2, + completedAt: 3, + totalRows: 3, + } + const c0: ChunkRecord = { + dumpId: 'dump_x', + chunkIndex: 0, + content: 'CREATE TABLE t (id INTEGER);\n', + bytes: 30, + createdAt: 1, + } + const c1: ChunkRecord = { + dumpId: 'dump_x', + chunkIndex: 1, + content: "INSERT INTO \"t\" (\"id\") VALUES (1);\n", + bytes: 35, + createdAt: 2, + } + await storage.put(`${DUMP_CHUNK_KEY}:0`, c0) + await storage.put(`${DUMP_CHUNK_KEY}:1`, c1) + + const stream = await engine.assembleDump(state) + expect(stream).not.toBeNull() + const text = await new Response(stream as ReadableStream).text() + expect(text).toContain('CREATE TABLE t') + expect(text).toContain('VALUES (1)') + }) + + it('respects custom rowsPerBatch from options', async () => { + let captured = '' + vi.mocked(executeOperation).mockImplementation(async (queries: any) => { + const sql: string = queries[0].sql + if (sql.includes('WHERE rowid >')) { + captured = sql + return [] + } + if (sql.includes("name NOT LIKE 'tmp_%'")) return [{ name: 'users' }] + if (sql.includes('SELECT sql FROM sqlite_master')) { + return [{ sql: 'CREATE TABLE users (id INTEGER)' }] + } + return [] + }) + const { engine } = makeEngine({ rowsPerBatch: 42 }) + await engine.startDump() + expect(captured).toContain('LIMIT 42') + }) +}) diff --git a/src/export/chunkedDump.ts b/src/export/chunkedDump.ts new file mode 100644 index 0000000..52b8d45 --- /dev/null +++ b/src/export/chunkedDump.ts @@ -0,0 +1,389 @@ +import { DataSource } from '../types' +import { StarbaseDBConfiguration } from '../handler' +import { executeOperation } from './index' + +/** + * Chunked, resumable database dumps with "breathing intervals". + * + * The legacy dump endpoint accumulated the entire database into a single + * in-memory string, which fails on large databases and blows through the + * 30s Workers request window. This module writes the dump in bounded chunks, + * yielding ("breathing") between chunks so queued requests are not starved, + * persists progress in DO storage so work can resume across the 30s window + * (driven by the DO alarm), and mirrors every chunk to R2 when a binding is + * available so dumps survive eviction and can be fetched after completion. + */ + +export const DUMP_STATE_KEY = 'tmp_dump_state' +export const DUMP_CHUNK_KEY = 'tmp_dump_chunk' + +/** Defaults tuned to stay far under the 30s request window per cycle. */ +export const DEFAULT_DUMP_OPTIONS = { + /** Wall-clock budget per work cycle (ms) before we breathe/yield. */ + cycleTimeBudgetMs: 5_000, + /** Minimum idle time between cycles when requests are waiting. */ + breathingIntervalMs: 5_000, + /** Maximum rows fetched per SELECT batch. */ + rowsPerBatch: 500, + /** Approximate serialized size (bytes) that closes a chunk. */ + chunkTargetBytes: 512 * 1024, +} as const + +export interface DumpOptions { + cycleTimeBudgetMs?: number + breathingIntervalMs?: number + rowsPerBatch?: number + chunkTargetBytes?: number +} + +export type DumpPhase = 'schema' | 'table-data' | 'complete' + +export interface DumpState { + dumpId: string + fileName: string + phase: DumpPhase + tables: string[] + tableIndex: number + lastFetchedRowId: number | null + /** Rowid of the last row written into the current chunk. */ + chunkRowOffset: number + /** Total bytes serialized so far (chunks flushed to R2). */ + bytesWritten: number + chunkIndex: number + startedAt: number + updatedAt: number + /** Set when the dump finished and the R2 object is ready. */ + completedAt?: number + /** Aggregate stats surfaced in status responses. */ + totalRows: number + callbackUrl?: string +} + +export interface ChunkRecord { + dumpId: string + chunkIndex: number + /** Serialized SQL statements for this chunk. */ + content: string + bytes: number + createdAt: number +} + +const IDENTIFIER_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/ + +/** Guard against SQL injection through table names coming from sqlite_master. */ +export function isSafeIdentifier(name: string): boolean { + return IDENTIFIER_PATTERN.test(name) +} + +function sqlQuote(value: unknown): string { + if (value === null || value === undefined) { + return 'NULL' + } + if (typeof value === 'number' || typeof value === 'boolean') { + return String(value) + } + if (value instanceof ArrayBuffer) { + return `X'${Array.from(new Uint8Array(value), (b) => + b.toString(16).padStart(2, '0') + ).join('')}'` + } + if (typeof value === 'string') { + return `'${value.replace(/'/g, "''")}'` + } + // Fallback: serialize deterministically rather than emitting "object". + return `'${JSON.stringify(value).replace(/'/g, "''")}'` +} + +export function serializeRows( + table: string, + rows: Record[] +): { content: string; rowCount: number } { + if (rows.length === 0) { + return { content: '', rowCount: 0 } + } + const columns = Object.keys(rows[0]) + const columnList = columns.map((c) => `"${c}"`).join(', ') + const lines = rows.map((row) => { + const values = columns.map((c) => sqlQuote(row[c])) + return `INSERT INTO "${table}" (${columnList}) VALUES (${values.join(', ')});` + }) + return { content: `${lines.join('\n')}\n`, rowCount: rows.length } +} + +export function makeDumpFileName(now = new Date()): string { + const pad = (n: number) => String(n).padStart(2, '0') + const stamp = + `${now.getUTCFullYear()}${pad(now.getUTCMonth() + 1)}${pad(now.getUTCDate())}` + + `-${pad(now.getUTCHours())}${pad(now.getUTCMinutes())}${pad(now.getUTCSeconds())}` + return `dump_${stamp}.sql` +} + +const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms)) + +/** + * Core engine. Designed to be driven by the Durable Object so each call + * performs at most one bounded cycle of work (respecting `cycleTimeBudgetMs`), + * then the DO decides whether to breathe and continue within this request or + * schedule its alarm for the next cycle. + */ +export class ChunkedDumpEngine { + constructor( + private readonly storage: DurableObjectStorage, + private readonly r2: R2Bucket | undefined, + private readonly dataSource: DataSource, + private readonly config: StarbaseDBConfiguration, + private readonly options: Required + ) {} + + /** Create or resume a dump. Returns the current state after one cycle. */ + async startDump(callbackUrl?: string): Promise { + const existing = await this.storage.get(DUMP_STATE_KEY) + if (existing && !existing.completedAt) { + // Resume in-progress dump instead of starting over. + return this.runCycle() + } + + const tablesResult = await executeOperation( + [{ sql: "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'tmp_%';" }], + this.dataSource, + this.config + ) + const tables = tablesResult + .map((row: Record) => String(row.name)) + .filter((name) => isSafeIdentifier(name)) + + const now = Date.now() + const state: DumpState = { + dumpId: `dump_${now.toString(36)}`, + fileName: makeDumpFileName(new Date(now)), + phase: 'schema', + tables, + tableIndex: 0, + lastFetchedRowId: null, + chunkRowOffset: 0, + bytesWritten: 0, + chunkIndex: 0, + startedAt: now, + updatedAt: now, + totalRows: 0, + ...(callbackUrl ? { callbackUrl } : {}), + } + await this.storage.put(DUMP_STATE_KEY, state) + return this.runCycle() + } + + async getState(): Promise { + return this.storage.get(DUMP_STATE_KEY) + } + + /** + * Run one bounded cycle: serialize schema/data chunks until the time + * budget is exhausted, flushing each chunk to R2 (when bound). Returns + * the updated state; `completedAt` set when the dump is done. + */ + async runCycle(): Promise { + const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState + if (!state) { + throw new Error('No dump in progress') + } + if (state.completedAt) { + return state + } + + const cycleStart = Date.now() + let content = '' + let contentBytes = 0 + let chunkDirty = false + + const flushChunk = async () => { + if (!chunkDirty) { + return + } + const record: ChunkRecord = { + dumpId: state.dumpId, + chunkIndex: state.chunkIndex, + content, + bytes: contentBytes, + createdAt: Date.now(), + } + if (this.r2) { + await this.r2.put( + `${state.dumpId}/${String(record.chunkIndex).padStart(8, '0')}.sql`, + record.content + ) + } + // Persist the chunk in DO storage so a boundless environment can + // still reassemble; keep only the tail window to bound memory. + await this.storage.put(`${DUMP_CHUNK_KEY}:${record.chunkIndex}`, record) + state.chunkIndex += 1 + state.bytesWritten += contentBytes + content = '' + contentBytes = 0 + chunkDirty = false + } + + // Phase 1: schema pass. tableIndex is advanced BEFORE the yield point + // so a resumed cycle never re-fetches (and duplicate-emits) a schema. + if (state.phase === 'schema') { + while (state.tableIndex < state.tables.length) { + const table = state.tables[state.tableIndex] + state.tableIndex++ + if (!isSafeIdentifier(table)) continue + const schemaResult = await executeOperation( + [{ + sql: `SELECT sql FROM sqlite_master WHERE type='table' AND name='${table}';`, + }], + this.dataSource, + this.config + ) + if (schemaResult.length) { + const schema = schemaResult[0].sql + const ddl = `\n-- Table: ${table}\n${schema};\n\n` + content += ddl + contentBytes += ddl.length + chunkDirty = true + } + if (Date.now() - cycleStart >= this.options.cycleTimeBudgetMs) { + await flushChunk() + state.updatedAt = Date.now() + await this.storage.put(DUMP_STATE_KEY, state) + return state + } + } + state.phase = 'table-data' + state.tableIndex = 0 + } + + // Phase 2: data pass in rowid-batched chunks. + while (state.tableIndex < state.tables.length) { + const table = state.tables[state.tableIndex] + if (!isSafeIdentifier(table)) { + state.tableIndex++ + state.lastFetchedRowId = null + continue + } + + const since = state.lastFetchedRowId ?? 0 + const rowsResult = (await executeOperation( + [{ + sql: `SELECT rowid AS __rowid, * FROM "${table}" WHERE rowid > ${since} ORDER BY rowid LIMIT ${this.options.rowsPerBatch};`, + }], + this.dataSource, + this.config + )) as Record[] + + if (rowsResult.length === 0) { + state.tableIndex++ + state.lastFetchedRowId = null + continue + } + + const withoutRowidCol = rowsResult.map((row) => { + const { __rowid, ...rest } = row + state.lastFetchedRowId = Number(__rowid ?? state.lastFetchedRowId) + return rest + }) + + const { content: batchSql, rowCount } = serializeRows(table, withoutRowidCol) + content += batchSql + contentBytes += batchSql.length + state.totalRows += rowCount + chunkDirty = true + + const chunkClosed = + contentBytes >= this.options.chunkTargetBytes || + Date.now() - cycleStart >= this.options.cycleTimeBudgetMs + + if (chunkClosed) { + await flushChunk() + state.updatedAt = Date.now() + await this.storage.put(DUMP_STATE_KEY, state) + return state + } + } + + // Phase 3: completion marker. + const done = `\n-- Dump complete: ${state.totalRows} rows, ${state.chunkIndex + (chunkDirty ? 1 : 0)} chunks.\n` + content += done + contentBytes += done.length + chunkDirty = true + await flushChunk() + + state.phase = 'complete' + state.completedAt = Date.now() + state.updatedAt = state.completedAt + await this.storage.put(DUMP_STATE_KEY, state) + return state + } + + /** True when another cycle should run right now (time still available). */ + shouldContinue(state: DumpState, requestStart: number): boolean { + if (state.completedAt) return false + return Date.now() - requestStart < this.options.cycleTimeBudgetMs + } + + /** Milliseconds to wait before the next cycle (breathing interval). */ + breathingDelayMs(): number { + return this.options.breathingIntervalMs + } + + /** Reassemble the dump from chunk records (or R2 when bound). */ + async assembleDump(state: DumpState): Promise | null> { + if (!state.completedAt) return null + + if (this.r2) { + const stream = await this.concatenateR2(state) + if (stream) return stream + } + + const storage = this.storage + return new ReadableStream({ + async start(controller) { + const encoder = new TextEncoder() + for (let i = 0; i < state.chunkIndex; i++) { + const record = await storage.get( + `${DUMP_CHUNK_KEY}:${i}` + ) + if (record?.content) { + controller.enqueue(encoder.encode(record.content)) + } + } + controller.close() + }, + }) + } + + private async concatenateR2(state: DumpState): Promise | null> { + if (!this.r2) return null + const head = await this.r2.get(`${state.dumpId}/00000000.sql`) + if (!head) return null + // R2 concatenated reads: stream each part sequentially. + const parts: ReadableStream[] = [] + for (let i = 0; i < state.chunkIndex; i++) { + const obj = await this.r2.get( + `${state.dumpId}/${String(i).padStart(8, '0')}.sql` + ) + if (obj?.body) { + parts.push(obj.body) + } + } + return concatStreams(parts) + } +} + +function concatStreams(streams: ReadableStream[]): ReadableStream { + return new ReadableStream({ + async start(controller) { + for (const stream of streams) { + const reader = stream.getReader() + for (;;) { + const { done, value } = await reader.read() + if (done) break + controller.enqueue(value) + } + reader.releaseLock() + } + controller.close() + }, + }) +} diff --git a/src/export/dump.ts b/src/export/dump.ts index 91a2e89..45029da 100644 --- a/src/export/dump.ts +++ b/src/export/dump.ts @@ -1,14 +1,64 @@ -import { executeOperation } from '.' +import { executeOperation } from './index' import { StarbaseDBConfiguration } from '../handler' import { DataSource } from '../types' import { createResponse } from '../utils' +import { + ChunkedDumpEngine, + DEFAULT_DUMP_OPTIONS, + type DumpOptions, + type DumpState, +} from './chunkedDump' -export async function dumpDatabaseRoute( +/** + * Dump route. + * + * Small databases keep the legacy behavior: the dump is fully serialized and + * returned inline as a downloadable file (well under the 30s window). + * + * Large databases exceed the 30s Workers request window, so `?job=1` opts + * into a resumable job flow driven inside the Durable Object: bounded work + * cycles with breathing intervals, progress persisted in DO storage, chunks + * mirrored to R2 when the binding exists, the DO alarm resuming work across + * window boundaries, and an optional `callbackUrl` invoked on completion. + */ + +export interface DumpJobEnv { + /** Optional R2 binding; when absent, chunks persist in DO storage only. */ + R2_DUMP_BUCKET?: R2Bucket +} + +export interface DumpEngineHost { + storage: DurableObjectStorage + env: DumpJobEnv + dataSource: DataSource + config: StarbaseDBConfiguration + setAlarm: (time: number, options?: DurableObjectSetAlarmOptions) => Promise +} + +/** Query param parsing: `cycleMs`, `breathMs`, `rows`, `chunkBytes`. */ +export function parseDumpOptions(searchParams: URLSearchParams): DumpOptions { + const options: DumpOptions = {} + const readNumber = (key: string, target: keyof DumpOptions) => { + const raw = searchParams.get(key) + if (raw === null) return + const value = Number(raw) + if (Number.isFinite(value) && value > 0) { + options[target] = value as never + } + } + readNumber('cycleMs', 'cycleTimeBudgetMs') + readNumber('breathMs', 'breathingIntervalMs') + readNumber('rows', 'rowsPerBatch') + readNumber('chunkBytes', 'chunkTargetBytes') + return options +} + +/** Legacy inline dump path, behavior-identical for small databases. */ +export async function legacyDump( dataSource: DataSource, config: StarbaseDBConfiguration ): Promise { try { - // Get all table names const tablesResult = await executeOperation( [{ sql: "SELECT name FROM sqlite_master WHERE type='table';" }], dataSource, @@ -18,15 +68,11 @@ export async function dumpDatabaseRoute( const tables = tablesResult.map((row: any) => row.name) let dumpContent = 'SQLite format 3\0' // SQLite file header - // Iterate through all tables for (const table of tables) { - // Get table schema const schemaResult = await executeOperation( - [ - { - sql: `SELECT sql FROM sqlite_master WHERE type='table' AND name='${table}';`, - }, - ], + [{ + sql: `SELECT sql FROM sqlite_master WHERE type='table' AND name='${table}';`, + }], dataSource, config ) @@ -36,7 +82,6 @@ export async function dumpDatabaseRoute( dumpContent += `\n-- Table: ${table}\n${schema};\n\n` } - // Get table data const dataResult = await executeOperation( [{ sql: `SELECT * FROM ${table};` }], dataSource, @@ -55,7 +100,6 @@ export async function dumpDatabaseRoute( dumpContent += '\n' } - // Create a Blob from the dump content const blob = new Blob([dumpContent], { type: 'application/x-sqlite3' }) const headers = new Headers({ @@ -69,3 +113,139 @@ export async function dumpDatabaseRoute( return createResponse(undefined, 'Failed to create database dump', 500) } } + +/** + * Chunked job dump, executed inside the Durable Object. Runs bounded cycles + * within the current request, schedules the alarm for the next breathing + * interval when work remains, and returns either the finished dump stream or + * a 202 progress payload. + */ +export async function runDumpJob( + host: DumpEngineHost, + searchParams: URLSearchParams, + requestStart: number = Date.now() +): Promise { + try { + const options = parseDumpOptions(searchParams) + const engine = new ChunkedDumpEngine( + host.storage, + host.env.R2_DUMP_BUCKET, + host.dataSource, + host.config, + { ...DEFAULT_DUMP_OPTIONS, ...options } + ) + + let state = await engine.getState() + if (!state || state.completedAt) { + state = await engine.startDump( + searchParams.get('callbackUrl') ?? undefined + ) + } else { + state = await engine.runCycle() + } + + // Keep working while this request still has budget (5s default cycle + // budget bounds each burst; breathing happens between bursts). + while ( + !state.completedAt && + Date.now() - requestStart < DEFAULT_DUMP_OPTIONS.cycleTimeBudgetMs + ) { + state = await engine.runCycle() + } + + if (state.completedAt) { + const stream = await engine.assembleDump(state) + if (stream) { + return new Response(stream, { + headers: { + 'Content-Type': 'application/x-sqlite3', + 'Content-Disposition': `attachment; filename="${state.fileName}"`, + }, + }) + } + } + + // Work remains: breathe, then let the DO alarm drive the next cycle. + const resumeAt = Date.now() + DEFAULT_DUMP_OPTIONS.breathingIntervalMs + await host.setAlarm(resumeAt) + + return createResponse( + { + dumpId: state.dumpId, + status: 'in-progress', + phase: state.phase, + progress: { + tablesTotal: state.tables.length, + tableIndex: state.tableIndex, + totalRows: state.totalRows, + bytesWritten: state.bytesWritten, + chunkIndex: state.chunkIndex, + }, + resumeAt, + fileName: state.fileName, + }, + undefined, + 202 + ) + } catch (error: any) { + console.error('Database Dump Error:', error) + return createResponse(undefined, 'Failed to create database dump', 500) + } +} + +/** + * Status + fetch endpoint for in-progress/completed chunked dumps. + * `GET /export/dump?job=1` while a job runs returns 202 progress; once the + * job is complete the assembled dump streams back. + */ +export async function dumpJobStatus( + host: DumpEngineHost +): Promise { + const engine = new ChunkedDumpEngine( + host.storage, + host.env.R2_DUMP_BUCKET, + host.dataSource, + host.config, + DEFAULT_DUMP_OPTIONS + ) + const state = await engine.getState() + if (!state) { + return createResponse(undefined, 'No dump job found', 404) + } + if (!state.completedAt) { + return createResponse( + { + dumpId: state.dumpId, + status: 'in-progress', + phase: state.phase, + progress: { + tablesTotal: state.tables.length, + tableIndex: state.tableIndex, + totalRows: state.totalRows, + bytesWritten: state.bytesWritten, + chunkIndex: state.chunkIndex, + }, + fileName: state.fileName, + }, + undefined, + 202 + ) + } + const stream = await engine.assembleDump(state) + if (!stream) { + return createResponse(undefined, 'Dump data unavailable', 410) + } + return new Response(stream, { + headers: { + 'Content-Type': 'application/x-sqlite3', + 'Content-Disposition': `attachment; filename="${state.fileName}"`, + }, + }) +} + +export async function dumpDatabaseRoute( + dataSource: DataSource, + config: StarbaseDBConfiguration +): Promise { + return legacyDump(dataSource, config) +} diff --git a/src/handler.ts b/src/handler.ts index 3fa0085..b69f7ce 100644 --- a/src/handler.ts +++ b/src/handler.ts @@ -120,10 +120,44 @@ export class StarbaseDB { } if (this.getFeature('export')) { - this.app.get('/export/dump', this.isInternalSource, async () => { + this.app.get('/export/dump', this.isInternalSource, async (c) => { + const url = new URL(c.req.raw.url) + const wantsJob = url.searchParams.get('job') === '1' + + if (wantsJob && this.dataSource.source === 'internal') { + // Chunked/resumable job flow runs inside the Durable + // Object via RPC; progress persists across the 30s window. + const searchParams: Record = {} + url.searchParams.forEach((value, key) => { + searchParams[key] = value + }) + // Narrow RPC view keeps hono's generic inference shallow. + const rpc = this.dataSource.rpc as unknown as { + startDumpJob: ( + config: StarbaseDBConfiguration, + searchParams: Record + ) => Promise + } + return await rpc.startDumpJob(this.config, searchParams) + } + return dumpDatabaseRoute(this.dataSource, this.config) }) + this.app.get('/export/dump/status', this.isInternalSource, async () => { + if (this.dataSource.source === 'internal') { + const rpc = this.dataSource.rpc as unknown as { + dumpJobStatus: (config: StarbaseDBConfiguration) => Promise + } + return await rpc.dumpJobStatus(this.config) + } + return createResponse( + undefined, + 'Chunked dump status requires the internal data source', + 400 + ) + }) + this.app.get( '/export/json/:tableName', this.isInternalSource, From 7eed275488a0c71ee8e23042571b12ba786191e3 Mon Sep 17 00:00:00 2001 From: Furox-Art <177975472+Furox-Art@users.noreply.github.com> Date: Sat, 19 Sep 2026 23:30:43 +0300 Subject: [PATCH 2/5] feat(export): consolidated R2 multipart finalize + presigned download URLs After a chunked dump completes, a DO-alarm-driven finalize now merges all chunk records into a single R2 object (dumps//) via a multipart upload, making large dumps downloadable as one object: - finalizeDump is time-budgeted and crash-resumable: uploaded parts and their etags are persisted after every part, interrupted finalizes resume via resumeMultipartUpload without re-uploading completed bytes, and a multi-GB consolidation progresses across several alarm invocations - getPresignedUrl feature-detects R2Bucket.createSignedUrl (newer workerd runtimes) and returns an expiring download URL; older bindings fall back to the existing streaming reassembly - /export/dump/status returns downloadUrl (+size, finalObjectKey) when a presigned URL is available; assembleDump prefers the consolidated object - Per-chunk R2 mirrors are cleaned up after a successful complete - parseDumpOptions: new partBytes/finalizeMs tuning params - 7 new unit tests (19 total): part sizing/order, byte-offset resume, budget-out-then-continue, no-R2 finalize, signed-URL presence/shape guards, consolidated-object preference --- src/do.ts | 18 +++ src/export/chunkedDump.test.ts | 260 +++++++++++++++++++++++++++++++++ src/export/chunkedDump.ts | 221 ++++++++++++++++++++++++++++ src/export/dump.ts | 53 +++++++ 4 files changed, 552 insertions(+) diff --git a/src/do.ts b/src/do.ts index 4d0bfb6..066ccb1 100644 --- a/src/do.ts +++ b/src/do.ts @@ -133,6 +133,24 @@ export class StarbaseDBDurableObject extends DurableObject { return } + // A finished dump that has not been consolidated yet: merge its + // chunk records into the single R2 object (multipart, resumable) + // so presigned download URLs become available. + if (dumpState && dumpState.completedAt && !dumpState.finalizedAt) { + const { runDumpFinalize } = await import('./export/dump') + await runDumpFinalize({ + storage: this.storage, + env: { + R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }) + .R2_DUMP_BUCKET, + }, + dataSource: this.dumpJobDataSource(), + config: { role: 'admin' } as StarbaseDBConfiguration, + setAlarm: (time, options) => this.setAlarm(time, options), + }) + return + } + // Fetch all the tasks that are marked to emit an event for this cycle. const task = (await this.executeQuery({ sql: 'SELECT * FROM tmp_cron_tasks WHERE is_active = 1;', diff --git a/src/export/chunkedDump.test.ts b/src/export/chunkedDump.test.ts index bd9f6ca..f471c75 100644 --- a/src/export/chunkedDump.test.ts +++ b/src/export/chunkedDump.test.ts @@ -303,3 +303,263 @@ describe('ChunkedDumpEngine cycles', () => { expect(captured).toContain('LIMIT 42') }) }) + +// --------------------------------------------------------------------------- +// R2 multipart finalize + presigned URL +// --------------------------------------------------------------------------- + +type UploadedPart = { partNumber: number; etag: string } + +const makeMultipartR2 = () => { + const uploaded: { partNumber: number; size: number }[] = [] + let completed: UploadedPart[] | null = null + let resumedWith: string | null = null + let createdFor: string | null = null + let signedUrlResult: { url?: string } | null | undefined = undefined + const deletedKeys: string[] = [] + const objects = new Map }>() + + const mpu = { + key: '', + uploadId: 'mpu-1', + uploadPart: async (partNumber: number, value: Uint8Array) => { + uploaded.push({ partNumber, size: value.length }) + return { partNumber, etag: `etag-${partNumber}` } + }, + abort: async () => undefined, + complete: async (parts: UploadedPart[]) => { + completed = parts + const total = uploaded.reduce((sum, p) => sum + p.size, 0) + return { key: mpu.key, size: total } + }, + } + + const r2 = { + createMultipartUpload: async (key: string) => { + createdFor = key + mpu.key = key + return mpu + }, + resumeMultipartUpload: (key: string, uploadId: string) => { + resumedWith = uploadId + mpu.key = key + return mpu + }, + put: async () => undefined, + get: async (key: string) => objects.get(key) ?? null, + delete: async (key: string) => { + deletedKeys.push(key) + }, + } + return { r2, uploaded, deleted: () => deletedKeys, completed: () => completed, resumedWith: () => resumedWith, createdFor: () => createdFor, objects, signedUrlResult: () => signedUrlResult, setSigned: (v: { url?: string } | null | undefined) => { signedUrlResult = v } } +} + +const seedCompletedState = async ( + storage: ReturnType, + chunks: string[], + extra: Partial = {} +) => { + const state: DumpState = { + dumpId: 'dump_fin', + fileName: 'dump_fin.sql', + phase: 'complete', + tables: ['t'], + tableIndex: 1, + lastFetchedRowId: null, + chunkRowOffset: 0, + bytesWritten: chunks.join('').length, + chunkIndex: chunks.length, + startedAt: 1, + updatedAt: 2, + completedAt: 3, + totalRows: 0, + ...extra, + } + for (let i = 0; i < chunks.length; i++) { + const record: ChunkRecord = { + dumpId: 'dump_fin', + chunkIndex: i, + content: chunks[i], + bytes: chunks[i].length, + createdAt: i, + } + await storage.put(`${DUMP_CHUNK_KEY}:${i}`, record) + } + await storage.put(DUMP_STATE_KEY, state) + return state +} + +describe('finalizeDump (R2 multipart upload + presigned URL)', () => { + it('uploads part-sized parts, completes the object and cleans chunk mirrors', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + // 4 chunks of 10 bytes each; partSize 20 → parts of 20, 20, 20(tail). + await seedCompletedState(storage, ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD']) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + m.r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const { done, state } = await engine.finalizeDump({ partSizeBytes: 20 }) + + expect(done).toBe(true) + expect(m.createdFor()).toBe('dumps/dump_fin/dump_fin.sql') + // 40 bytes at partSize 20 → exactly two full parts, no tail. + expect(m.uploaded.map((p) => p.size)).toEqual([20, 20]) + expect(m.uploaded.map((p) => p.partNumber)).toEqual([1, 2]) + expect(m.completed()?.length).toBe(2) + expect(state.finalObjectKey).toBe('dumps/dump_fin/dump_fin.sql') + expect(state.finalObjectSize).toBe(40) + expect(state.finalizedAt).toBeDefined() + // Per-chunk R2 mirrors are cleaned up after a successful complete. + expect(m.deleted().length).toBe(4) + }) + + it('resumes an interrupted finalize without re-uploading completed parts', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + const chunks = ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD'] + // First part (chunk 0 + chunk 1 = 20 bytes) already uploaded. + await seedCompletedState(storage, chunks, { + finalObjectKey: 'dumps/dump_fin/dump_fin.sql', + finalizeUploadId: 'mpu-9', + finalizeParts: [{ partNumber: 1, etag: 'etag-1' }], + finalizeBytes: 20, + }) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + m.r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const { done } = await engine.finalizeDump({ partSizeBytes: 20 }) + + expect(done).toBe(true) + expect(m.resumedWith()).toBe('mpu-9') + // Only the remaining 20 bytes upload, as part 2 — no re-upload. + expect(m.uploaded).toEqual([{ partNumber: 2, size: 20 }]) + expect(m.completed()?.length).toBe(2) + }) + + it('returns done:false when the time budget runs out mid-upload, then finishes on the next cycle', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + await seedCompletedState(storage, ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD']) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + m.r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const first = await engine.finalizeDump({ partSizeBytes: 20, timeBudgetMs: -1 }) + expect(first.done).toBe(false) + expect(first.state.finalizeUploadId).toBe('mpu-1') + + const second = await engine.finalizeDump({ partSizeBytes: 20 }) + expect(second.done).toBe(true) + expect(second.state.finalizedAt).toBeDefined() + }) + + it('finalizes without R2 binding (streaming-only environments)', async () => { + const storage = makeStorage() + await seedCompletedState(storage, ['data']) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + undefined as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const { done, state } = await engine.finalizeDump() + expect(done).toBe(true) + expect(state.finalObjectKey).toBeUndefined() + expect(state.finalizedAt).toBeDefined() + }) + + it('getPresignedUrl returns the signed URL when the runtime supports createSignedUrl', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + await seedCompletedState(storage, ['data'], { + finalObjectKey: 'dumps/dump_fin/dump_fin.sql', + finalizedAt: 9, + }) + m.setSigned({ url: 'https://signed.example/dump?sig=abc' }) + const r2 = Object.assign(m.r2, { + createSignedUrl: async () => (m.signedUrlResult() as { url?: string }), + }) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const url = await engine.getPresignedUrl(600) + expect(url).toBe('https://signed.example/dump?sig=abc') + }) + + it('getPresignedUrl returns null when the binding lacks createSignedUrl or the URL shape is unexpected', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + await seedCompletedState(storage, ['data'], { + finalObjectKey: 'dumps/dump_fin/dump_fin.sql', + finalizedAt: 9, + }) + // Older binding: no createSignedUrl at all. + const engineA = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + m.r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + await expect(engineA.getPresignedUrl()).resolves.toBeNull() + + // Unexpected result shape (missing url) must not throw. + m.setSigned({}) + const r2 = Object.assign(m.r2, { + createSignedUrl: async () => (m.signedUrlResult() as { url?: string }), + }) + const engineB = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + await expect(engineB.getPresignedUrl()).resolves.toBeNull() + }) + + it('assembleDump prefers the consolidated final object', async () => { + const storage = makeStorage() + const m = makeMultipartR2() + await seedCompletedState(storage, ['chunk-content'], { + finalObjectKey: 'dumps/dump_fin/dump_fin.sql', + finalizedAt: 9, + }) + const bytes = new TextEncoder().encode('CONSOLIDATED') + m.objects.set('dumps/dump_fin/dump_fin.sql', { + body: new ReadableStream({ + start(c) { + c.enqueue(bytes) + c.close() + }, + }), + }) + const engine = new ChunkedDumpEngine( + storage as unknown as DurableObjectStorage, + m.r2 as unknown as R2Bucket, + makeDataSource(), + makeConfig(), + { ...DEFAULT_DUMP_OPTIONS } + ) + const state = (await storage.get(DUMP_STATE_KEY)) as DumpState + const stream = await engine.assembleDump(state) + const text = await new Response(stream as ReadableStream).text() + expect(text).toBe('CONSOLIDATED') + }) +}) diff --git a/src/export/chunkedDump.ts b/src/export/chunkedDump.ts index 52b8d45..d1dc9d2 100644 --- a/src/export/chunkedDump.ts +++ b/src/export/chunkedDump.ts @@ -27,6 +27,11 @@ export const DEFAULT_DUMP_OPTIONS = { rowsPerBatch: 500, /** Approximate serialized size (bytes) that closes a chunk. */ chunkTargetBytes: 512 * 1024, + /** Part size for the consolidated R2 multipart upload. R2 (like S3) + * requires non-final parts to be at least 5 MiB. */ + finalizePartSizeBytes: 5 * 1024 * 1024, + /** Wall-clock budget per finalize cycle (ms) inside a DO alarm. */ + finalizeTimeBudgetMs: 20_000, } as const export interface DumpOptions { @@ -34,6 +39,8 @@ export interface DumpOptions { breathingIntervalMs?: number rowsPerBatch?: number chunkTargetBytes?: number + finalizePartSizeBytes?: number + finalizeTimeBudgetMs?: number } export type DumpPhase = 'schema' | 'table-data' | 'complete' @@ -57,6 +64,18 @@ export interface DumpState { /** Aggregate stats surfaced in status responses. */ totalRows: number callbackUrl?: string + + /** Consolidated R2 object produced by finalizeDump (multipart upload). + * Present once the per-chunk mirrors have been merged into a single + * `dumps//` object that supports presigned downloads. */ + finalObjectKey?: string + finalObjectSize?: number + finalizedAt?: number + /** In-progress multipart bookkeeping: survives eviction so a partially + * uploaded finalize can resume without re-uploading completed parts. */ + finalizeUploadId?: string + finalizeParts?: R2UploadedPart[] + finalizeBytes?: number } export interface ChunkRecord { @@ -327,10 +346,193 @@ export class ChunkedDumpEngine { return this.options.breathingIntervalMs } + /** + * Consolidate all chunk records into a single R2 object via a multipart + * upload (`dumps//`), bounded by a time budget so a + * multi-GB dump progresses across several DO alarm invocations instead of + * one 30s window. Partial progress (uploaded parts + their etags) is + * persisted after every part, so an interrupted finalize resumes with + * `resumeMultipartUpload` without re-uploading completed parts. + * + * Returns `{ done: false }` while parts remain; the DO alarm drives the + * remaining cycles. Idempotent once `state.finalizedAt` is set. + */ + async finalizeDump( + options: { partSizeBytes?: number; timeBudgetMs?: number } = {} + ): Promise<{ done: boolean; state: DumpState }> { + const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState + if (!state) { + throw new Error('No dump in progress') + } + if (state.finalizedAt) { + return { done: true, state } + } + if (!this.r2) { + // Without R2 there is nothing to consolidate; the DO-storage + // streaming reassembly remains the download path. + state.finalizedAt = Date.now() + await this.storage.put(DUMP_STATE_KEY, state) + return { done: true, state } + } + if (!state.completedAt) { + return { done: false, state } + } + + const partSize = options.partSizeBytes ?? this.options.finalizePartSizeBytes + const budget = options.timeBudgetMs ?? this.options.finalizeTimeBudgetMs + const key = state.finalObjectKey ?? `dumps/${state.dumpId}/${state.fileName}` + const cycleStart = Date.now() + + // Recover the in-flight multipart upload from a previous cycle, or + // start a fresh one and persist its uploadId immediately. + let mpu: R2MultipartUpload + if (state.finalizeUploadId) { + mpu = this.r2.resumeMultipartUpload(key, state.finalizeUploadId) + } else { + mpu = await this.r2.createMultipartUpload(key) + state.finalObjectKey = key + state.finalizeUploadId = mpu.uploadId + state.finalizeParts = [] + state.finalizeBytes = 0 + await this.storage.put(DUMP_STATE_KEY, state) + } + + const parts = state.finalizeParts ?? [] + const encoder = new TextEncoder() + const persist = async () => { + state.finalizeParts = parts + state.updatedAt = Date.now() + await this.storage.put(DUMP_STATE_KEY, state) + } + + // Stream chunk records into part-sized buffers, skipping the bytes + // already uploaded as parts in earlier cycles. + let skipped = state.finalizeBytes ?? 0 + let partNumber = parts.length + 1 + let buffer: Uint8Array[] = [] + let bufferLen = 0 + + for (let i = 0; i < state.chunkIndex; i++) { + const record = await this.storage.get( + `${DUMP_CHUNK_KEY}:${i}` + ) + if (!record?.content) continue + const encoded = encoder.encode(record.content) + let data = encoded + if (skipped > 0) { + if (skipped >= encoded.length) { + skipped -= encoded.length + continue + } + data = encoded.subarray(skipped) + skipped = 0 + } + buffer.push(data) + bufferLen += data.length + + // Non-final parts must be >= 5 MiB, so flush only at partSize. + if (bufferLen >= partSize) { + if (Date.now() - cycleStart >= budget) { + // Out of budget mid-upload: progress (parts + bytes) is + // already persisted, the next cycle continues cleanly. + await persist() + return { done: false, state } + } + const part = await mpu.uploadPart( + partNumber, + concatUint8(buffer, bufferLen) + ) + parts.push({ partNumber, etag: part.etag }) + state.finalizeBytes = (state.finalizeBytes ?? 0) + bufferLen + await persist() + partNumber++ + buffer = [] + bufferLen = 0 + } else if (Date.now() - cycleStart >= budget) { + // Budget exhausted while accumulating: the un-uploaded buffer + // is rebuilt next cycle from the persisted byte offset. + await persist() + return { done: false, state } + } + } + + if (parts.length === 0 && bufferLen === 0) { + // Empty dump: nothing to upload, finalize without an object. + state.finalizedAt = Date.now() + state.updatedAt = state.finalizedAt + await this.storage.put(DUMP_STATE_KEY, state) + return { done: true, state } + } + + // Tail part (allowed to be smaller than partSize). + if (bufferLen > 0) { + const part = await mpu.uploadPart( + partNumber, + concatUint8(buffer, bufferLen) + ) + parts.push({ partNumber, etag: part.etag }) + state.finalizeBytes = (state.finalizeBytes ?? 0) + bufferLen + } + + const object = await mpu.complete(parts) + state.finalObjectKey = key + state.finalObjectSize = object.size ?? state.finalizeBytes + state.finalizedAt = Date.now() + state.updatedAt = state.finalizedAt + state.finalizeUploadId = undefined + state.finalizeParts = undefined + state.finalizeBytes = undefined + await this.storage.put(DUMP_STATE_KEY, state) + + // Best-effort cleanup of the per-chunk R2 mirrors; DO storage records + // stay as the streaming fallback. + for (let i = 0; i < state.chunkIndex; i++) { + try { + await this.r2.delete( + `${state.dumpId}/${String(i).padStart(8, '0')}.sql` + ) + } catch { + // Cleanup is best-effort; leftover mirrors are harmless. + } + } + return { done: true, state } + } + + /** + * Presigned download URL for the finalized object. Presigned URL + * generation is feature-detected: newer workerd runtimes expose + * `R2Bucket.createSignedUrl`, older bindings do not — in that case the + * caller falls back to the streaming reassembly endpoint. + */ + async getPresignedUrl(expiresInSeconds = 3600): Promise { + const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState + if (!state?.finalObjectKey || !this.r2) return null + const creator = ( + this.r2 as R2Bucket & { createSignedUrl?: R2SignedUrlCreator } + ).createSignedUrl + if (typeof creator !== 'function') return null + try { + const signed = await creator.call( + this.r2, + state.finalObjectKey, + expiresInSeconds + ) + return signed?.url ?? null + } catch { + return null + } + } + /** Reassemble the dump from chunk records (or R2 when bound). */ async assembleDump(state: DumpState): Promise | null> { if (!state.completedAt) return null + // Prefer the consolidated multipart object when finalization ran. + if (state.finalObjectKey && this.r2) { + const final = await this.r2.get(state.finalObjectKey) + if (final?.body) return final.body + } + if (this.r2) { const stream = await this.concatenateR2(state) if (stream) return stream @@ -371,6 +573,25 @@ export class ChunkedDumpEngine { } } +/** + * Feature-detected presigned URL creator exposed by newer R2 runtime + * bindings (`R2Bucket.createSignedUrl`). + */ +type R2SignedUrlCreator = ( + key: string, + expiresInSeconds: number +) => Promise<{ url?: string } | null> + +function concatUint8(chunks: Uint8Array[], totalLength: number): Uint8Array { + const out = new Uint8Array(totalLength) + let offset = 0 + for (const chunk of chunks) { + out.set(chunk, offset) + offset += chunk.length + } + return out +} + function concatStreams(streams: ReadableStream[]): ReadableStream { return new ReadableStream({ async start(controller) { diff --git a/src/export/dump.ts b/src/export/dump.ts index 45029da..de32882 100644 --- a/src/export/dump.ts +++ b/src/export/dump.ts @@ -50,6 +50,8 @@ export function parseDumpOptions(searchParams: URLSearchParams): DumpOptions { readNumber('breathMs', 'breathingIntervalMs') readNumber('rows', 'rowsPerBatch') readNumber('chunkBytes', 'chunkTargetBytes') + readNumber('partBytes', 'finalizePartSizeBytes') + readNumber('finalizeMs', 'finalizeTimeBudgetMs') return options } @@ -154,6 +156,16 @@ export async function runDumpJob( } if (state.completedAt) { + // Kick off the R2 multipart consolidation (presigned-URL-ready + // single object) asynchronously via the DO alarm; the response + // below streams from the chunk records either way. + if (host.env.R2_DUMP_BUCKET && !state.finalizedAt) { + try { + await host.setAlarm(Date.now() + 1_000) + } catch (alarmError) { + console.error('Failed to schedule dump finalize:', alarmError) + } + } const stream = await engine.assembleDump(state) if (stream) { return new Response(stream, { @@ -231,6 +243,28 @@ export async function dumpJobStatus( 202 ) } + // Completed + consolidated: prefer a presigned, expiring download URL + // when the runtime supports it; otherwise fall back to streaming. + if (state.finalObjectKey) { + const downloadUrl = await engine.getPresignedUrl() + if (downloadUrl) { + return createResponse( + { + dumpId: state.dumpId, + status: 'complete', + downloadUrl, + downloadUrlExpiresInSeconds: 3600, + downloadType: 'presigned-url', + finalObjectKey: state.finalObjectKey, + size: state.finalObjectSize ?? state.bytesWritten, + totalRows: state.totalRows, + fileName: state.fileName, + }, + undefined, + 200 + ) + } + } const stream = await engine.assembleDump(state) if (!stream) { return createResponse(undefined, 'Dump data unavailable', 410) @@ -243,6 +277,25 @@ export async function dumpJobStatus( }) } +/** + * One bounded finalize cycle for the completed dump: consolidates chunks + * into the single R2 object via multipart upload. Driven by the DO alarm; + * reschedules itself while parts remain (`{ done: false }`). + */ +export async function runDumpFinalize(host: DumpEngineHost): Promise { + const engine = new ChunkedDumpEngine( + host.storage, + host.env.R2_DUMP_BUCKET, + host.dataSource, + host.config, + DEFAULT_DUMP_OPTIONS + ) + const { done } = await engine.finalizeDump() + if (!done) { + await host.setAlarm(Date.now() + 1_000) + } +} + export async function dumpDatabaseRoute( dataSource: DataSource, config: StarbaseDBConfiguration From 7019f2a6d95131f9dcd1eded66317c5ec0ac6fd9 Mon Sep 17 00:00:00 2001 From: Furox-Art <177975472+Furox-Art@users.noreply.github.com> Date: Wed, 23 Sep 2026 22:48:51 +0300 Subject: [PATCH 3/5] fix(export): harden resumable dump jobs --- README.md | 2 + src/do.ts | 31 +- src/export/chunkedDump.sqlite.test.ts | 140 ++++++++ src/export/chunkedDump.test.ts | 230 +++++++++--- src/export/chunkedDump.ts | 486 +++++++++++++++++--------- src/export/dump.job.test.ts | 134 +++++++ src/export/dump.ts | 151 ++++---- src/handler.dump.test.ts | 87 +++++ src/handler.ts | 43 ++- 9 files changed, 998 insertions(+), 306 deletions(-) create mode 100644 src/export/chunkedDump.sqlite.test.ts create mode 100644 src/export/dump.job.test.ts create mode 100644 src/handler.dump.test.ts diff --git a/README.md b/README.md index 5931b1c..43e17bc 100644 --- a/README.md +++ b/README.md @@ -243,6 +243,8 @@ curl --location 'https://starbasedb.YOUR-ID-HERE.workers.dev/export/dump' \ +Internal database dumps use one temporary-chunk retention policy. Chunks remain in Durable Object storage and, when configured, in R2 while a dump is active. After the R2 multipart object is finalized successfully, both copies of the temporary chunks are deleted. Without R2, the Durable Object chunk records remain the completed job's downloadable artifact. Large internal requests automatically use the resumable job flow; `?job=1` remains supported for explicit job requests. +

JSON Data Export

 
diff --git a/src/do.ts b/src/do.ts
index 066ccb1..816de25 100644
--- a/src/do.ts
+++ b/src/do.ts
@@ -121,12 +121,14 @@ export class StarbaseDBDurableObject extends DurableObject {
                     {
                         storage: this.storage,
                         env: {
-                            R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket })
-                                .R2_DUMP_BUCKET,
+                            R2_DUMP_BUCKET: (
+                                this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }
+                            ).R2_DUMP_BUCKET,
                         },
                         dataSource: this.dumpJobDataSource(),
                         config: { role: 'admin' } as StarbaseDBConfiguration,
-                        setAlarm: (time, options) => this.setAlarm(time, options),
+                        setAlarm: (time, options) =>
+                            this.setAlarm(time, options),
                     },
                     new URLSearchParams()
                 )
@@ -136,13 +138,20 @@ export class StarbaseDBDurableObject extends DurableObject {
             // A finished dump that has not been consolidated yet: merge its
             // chunk records into the single R2 object (multipart, resumable)
             // so presigned download URLs become available.
-            if (dumpState && dumpState.completedAt && !dumpState.finalizedAt) {
+            if (
+                dumpState &&
+                dumpState.completedAt &&
+                (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket })
+                    .R2_DUMP_BUCKET &&
+                (!dumpState.finalizedAt || !dumpState.temporaryChunksCleanedAt)
+            ) {
                 const { runDumpFinalize } = await import('./export/dump')
                 await runDumpFinalize({
                     storage: this.storage,
                     env: {
-                        R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket })
-                            .R2_DUMP_BUCKET,
+                        R2_DUMP_BUCKET: (
+                            this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }
+                        ).R2_DUMP_BUCKET,
                     },
                     dataSource: this.dumpJobDataSource(),
                     config: { role: 'admin' } as StarbaseDBConfiguration,
@@ -357,8 +366,9 @@ export class StarbaseDBDurableObject extends DurableObject {
                 env: {
                     // Optional binding; deployers add it to wrangler.toml when
                     // they want R2-backed dumps. Absent = storage-only mode.
-                    R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket })
-                        .R2_DUMP_BUCKET,
+                    R2_DUMP_BUCKET: (
+                        this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }
+                    ).R2_DUMP_BUCKET,
                 },
                 dataSource: this.dumpJobDataSource(),
                 config,
@@ -376,8 +386,9 @@ export class StarbaseDBDurableObject extends DurableObject {
         return dumpJobStatus({
             storage: this.storage,
             env: {
-                R2_DUMP_BUCKET: (this.env as Env & { R2_DUMP_BUCKET?: R2Bucket })
-                    .R2_DUMP_BUCKET,
+                R2_DUMP_BUCKET: (
+                    this.env as Env & { R2_DUMP_BUCKET?: R2Bucket }
+                ).R2_DUMP_BUCKET,
             },
             dataSource: this.dumpJobDataSource(),
             config,
diff --git a/src/export/chunkedDump.sqlite.test.ts b/src/export/chunkedDump.sqlite.test.ts
new file mode 100644
index 0000000..ec5bcd6
--- /dev/null
+++ b/src/export/chunkedDump.sqlite.test.ts
@@ -0,0 +1,140 @@
+import { createClient, type Client } from '@libsql/client'
+import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'
+import { executeOperation } from './index'
+import {
+    ChunkedDumpEngine,
+    DEFAULT_DUMP_OPTIONS,
+    DUMP_STATE_KEY,
+    type DumpState,
+} from './chunkedDump'
+import type { DataSource } from '../types'
+import type { StarbaseDBConfiguration } from '../handler'
+
+vi.mock('./index', () => ({
+    executeOperation: vi.fn(),
+}))
+
+const makeStorage = () => {
+    const values = new Map()
+    return {
+        get: vi.fn(async (key: string) => values.get(key) as T),
+        put: vi.fn(async (key: string, value: unknown) => {
+            values.set(key, value)
+        }),
+        delete: vi.fn(async (key: string) => {
+            values.delete(key)
+        }),
+    }
+}
+
+const makeDataSource = (): DataSource =>
+    ({ source: 'internal', rpc: {} }) as unknown as DataSource
+
+const makeConfig = (): StarbaseDBConfiguration => ({ role: 'admin' })
+
+describe('ChunkedDumpEngine with SQLite', () => {
+    let client: Client
+    let storage: ReturnType
+    let queries: { sql: string; params?: unknown[] }[]
+
+    beforeEach(async () => {
+        client = createClient({ url: 'file::memory:' })
+        storage = makeStorage()
+        queries = []
+        vi.mocked(executeOperation).mockImplementation(async (batch: any[]) => {
+            const query = batch[0]
+            queries.push(query)
+            const result = await client.execute({
+                sql: query.sql,
+                args: query.params ?? [],
+            })
+            return result.rows as Record[]
+        })
+    })
+
+    afterEach(() => {
+        client.close()
+        vi.clearAllMocks()
+    })
+
+    it('keeps negative, zero, and positive rowids and handles composite WITHOUT ROWID keys', async () => {
+        await client.execute(
+            'CREATE TABLE "odd names" (id INTEGER, value TEXT)'
+        )
+        await client.execute({
+            sql: 'INSERT INTO "odd names" (rowid, id, value) VALUES (?, ?, ?), (?, ?, ?), (?, ?, ?)',
+            args: [-7, 10, 'negative', 0, 20, 'zero', 9, 30, 'positive'],
+        })
+        await client.execute(
+            'CREATE TABLE "without rowid" ("key one" TEXT NOT NULL, part INTEGER NOT NULL, value TEXT, PRIMARY KEY ("key one", part)) WITHOUT ROWID'
+        )
+        await client.execute({
+            sql: 'INSERT INTO "without rowid" ("key one", part, value) VALUES (?, ?, ?), (?, ?, ?)',
+            args: ['a', 2, 'a2', 'a', 1, 'a1'],
+        })
+        await client.execute('CREATE TABLE "shadow" ("rowid" TEXT, value TEXT)')
+        await client.execute({
+            sql: 'INSERT INTO "shadow" ("rowid", value) VALUES (?, ?), (?, ?)',
+            args: ['first', 1, 'second', 2],
+        })
+
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            undefined,
+            makeDataSource(),
+            makeConfig(),
+            { ...DEFAULT_DUMP_OPTIONS, rowsPerBatch: 2, chunkTargetBytes: 80 }
+        )
+        let state = await engine.startDump()
+        for (let i = 0; i < 20 && !state.completedAt; i++) {
+            state = await engine.runCycle()
+        }
+        expect(state.completedAt).toBeDefined()
+        expect(state.totalRows).toBe(7)
+
+        const firstDataQuery = queries.find(
+            (query) =>
+                query.sql.includes('FROM "odd names"') &&
+                query.sql.includes('ORDER BY')
+        )
+        expect(firstDataQuery?.sql).not.toContain('WHERE "rowid" >')
+        expect(
+            queries.some(
+                (query) =>
+                    query.sql.includes('FROM "odd names"') &&
+                    query.sql.includes('WHERE "rowid" > ?') &&
+                    query.params?.[0] === 0
+            )
+        ).toBe(true)
+        expect(
+            queries.some(
+                (query) =>
+                    query.sql.includes('FROM "without rowid"') &&
+                    query.sql.includes('("key one", "part") > (?, ?)')
+            )
+        ).toBe(true)
+        expect(
+            queries.some(
+                (query) =>
+                    query.sql.includes('FROM "shadow"') &&
+                    query.sql.includes('ORDER BY "_rowid_"')
+            )
+        ).toBe(true)
+
+        const stream = await engine.assembleDump(state)
+        expect(stream).not.toBeNull()
+        const dump = await new Response(stream as ReadableStream).text()
+        expect(dump).toContain('INSERT INTO "odd names"')
+        expect(dump).toContain('negative')
+        expect(dump).toContain('zero')
+        expect(dump).toContain('positive')
+        expect(dump).toContain('INSERT INTO "without rowid"')
+        expect(dump).toContain('a1')
+        expect(dump).toContain('a2')
+        expect(dump).toContain('INSERT INTO "shadow"')
+        const persisted = (await storage.get(DUMP_STATE_KEY)) as
+            | DumpState
+            | undefined
+        expect(persisted?.completedAt).toBeDefined()
+    })
+})
diff --git a/src/export/chunkedDump.test.ts b/src/export/chunkedDump.test.ts
index f471c75..08692dd 100644
--- a/src/export/chunkedDump.test.ts
+++ b/src/export/chunkedDump.test.ts
@@ -3,6 +3,7 @@ import {
     ChunkedDumpEngine,
     DEFAULT_DUMP_OPTIONS,
     isSafeIdentifier,
+    quoteIdentifier,
     makeDumpFileName,
     serializeRows,
     DUMP_STATE_KEY,
@@ -83,9 +84,7 @@ describe('row serialization', () => {
     })
 
     it('renders NULLs, numbers and booleans unquoted', () => {
-        const { content } = serializeRows('t', [
-            { a: null, b: 1.5, c: true },
-        ])
+        const { content } = serializeRows('t', [{ a: null, b: 1.5, c: true }])
         expect(content).toContain('(NULL, 1.5, true)')
     })
 
@@ -94,6 +93,15 @@ describe('row serialization', () => {
         expect(content).toBe('')
         expect(rowCount).toBe(0)
     })
+
+    it('quotes identifiers and serializes null and binary values', () => {
+        const { content } = serializeRows('order "items"', [
+            { 'value "x"': null, payload: new Uint8Array([0, 15, 255]) },
+        ])
+        expect(content).toContain('INSERT INTO "order ""items"""')
+        expect(content).toContain('("value ""x""", "payload")')
+        expect(content).toContain("(NULL, X'000fff')")
+    })
 })
 
 describe('dump file naming', () => {
@@ -107,15 +115,43 @@ describe('ChunkedDumpEngine cycles', () => {
     const setupTables = (tables: string[], schemas: Record) => {
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
+            const params: unknown[] = queries[0].params ?? []
             if (sql.includes("type='table' AND name NOT LIKE")) {
                 return tables.map((name) => ({ name }))
             }
             if (sql.includes('SELECT sql FROM sqlite_master')) {
-                const match = /name='([^']+)'/.exec(sql)
-                const name = match?.[1] ?? ''
+                const name = String(params[0] ?? '')
                 return schemas[name] ? [{ sql: schemas[name] }] : []
             }
-            // rowid-batched data fetch
+            if (sql.includes('PRAGMA table_info')) {
+                return [{ name: 'id', pk: 1 }]
+            }
+            if (sql.includes('LIMIT 0')) {
+                return []
+            }
+            if (sql.includes('ORDER BY')) {
+                const since = Number(params[0] ?? -1)
+                if (since < 2) {
+                    return since < 0
+                        ? [
+                              {
+                                  __starbase_dump_cursor_0: 0,
+                                  id: 0,
+                              },
+                              {
+                                  __starbase_dump_cursor_0: 1,
+                                  id: 1,
+                              },
+                          ]
+                        : [
+                              {
+                                  __starbase_dump_cursor_0: since + 1,
+                                  id: since + 1,
+                              },
+                          ]
+                }
+                return []
+            }
             return []
         })
     }
@@ -124,20 +160,28 @@ describe('ChunkedDumpEngine cycles', () => {
         setupTables(['users'], { users: 'CREATE TABLE users (id INTEGER)' })
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
+            const params: unknown[] = queries[0].params ?? []
             if (sql.includes("name NOT LIKE 'tmp_%'")) {
                 return [{ name: 'users' }]
             }
             if (sql.includes('SELECT sql FROM sqlite_master')) {
                 return [{ sql: 'CREATE TABLE users (id INTEGER)' }]
             }
-            if (sql.includes('WHERE rowid >')) {
-                const since = Number(/rowid > (\d+)/.exec(sql)?.[1] ?? 0)
-                return since === 0
-                    ? [
-                          { __rowid: 1, id: 1 },
-                          { __rowid: 2, id: 2 },
-                      ]
-                    : []
+            if (sql.includes('PRAGMA table_info')) {
+                return [{ name: 'id', pk: 1 }]
+            }
+            if (sql.includes('LIMIT 0')) {
+                return []
+            }
+            if (sql.includes('ORDER BY')) {
+                const since = Number(params[0] ?? -1)
+                if (since < 0) {
+                    return [
+                        { __starbase_dump_cursor_0: -2, id: -2 },
+                        { __starbase_dump_cursor_0: 0, id: 0 },
+                    ]
+                }
+                return since < 1 ? [{ __starbase_dump_cursor_0: 1, id: 1 }] : []
             }
             return []
         })
@@ -147,11 +191,13 @@ describe('ChunkedDumpEngine cycles', () => {
 
         expect(state.completedAt).toBeDefined()
         expect(state.phase).toBe('complete')
-        expect(state.totalRows).toBe(2)
+        expect(state.totalRows).toBe(3)
         expect(state.chunkIndex).toBeGreaterThan(0)
         expect(r2.put).toHaveBeenCalled()
         // Progress persisted in DO storage for resumability.
-        const persisted = (await storage.get(DUMP_STATE_KEY)) as DumpState | undefined
+        const persisted = (await storage.get(DUMP_STATE_KEY)) as
+            | DumpState
+            | undefined
         expect(persisted?.completedAt).toBeDefined()
     })
 
@@ -160,16 +206,24 @@ describe('ChunkedDumpEngine cycles', () => {
         let call = 0
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
-            if (sql.includes("name NOT LIKE 'tmp_%'")) return [{ name: 'users' }]
+            const params: unknown[] = queries[0].params ?? []
+            if (sql.includes("name NOT LIKE 'tmp_%'"))
+                return [{ name: 'users' }]
             if (sql.includes('SELECT sql FROM sqlite_master')) {
                 return [{ sql: 'CREATE TABLE users (id INTEGER)' }]
             }
-            if (sql.includes('WHERE rowid >')) {
-                const since = Number(/rowid > (\d+)/.exec(sql)?.[1] ?? 0)
+            if (sql.includes('PRAGMA table_info')) {
+                return [{ name: 'id', pk: 1 }]
+            }
+            if (sql.includes('LIMIT 0')) {
+                return []
+            }
+            if (sql.includes('ORDER BY')) {
+                const since = Number(params[0] ?? -1)
                 call++
-                // Yield after every batch: each cycle emits exactly one row,
-                // and the stream is finite (3 rows total).
-                return since < 3 ? [{ __rowid: since + 1, id: since + 1 }] : []
+                return since < 3
+                    ? [{ __starbase_dump_cursor_0: since + 1, id: since + 1 }]
+                    : []
             }
             return []
         })
@@ -189,26 +243,44 @@ describe('ChunkedDumpEngine cycles', () => {
         expect(state.totalRows).toBeGreaterThan(0)
     })
 
-    it('filters unsafe table names out of the dump plan', async () => {
+    it('quotes valid and unusual table names instead of dropping them', async () => {
+        const dataSqls: string[] = []
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
+            const params: unknown[] = queries[0].params ?? []
             if (sql.includes("name NOT LIKE 'tmp_%'")) {
+                return [{ name: 'good_table' }, { name: 'order items "2026"' }]
+            }
+            if (sql.includes('SELECT sql FROM sqlite_master')) {
                 return [
-                    { name: 'good_table' },
-                    { name: 'bad"; DROP TABLE users;--' },
+                    { sql: 'CREATE TABLE "order items ""2026""" (id INTEGER)' },
                 ]
             }
-            if (sql.includes('SELECT sql FROM sqlite_master')) {
-                return [{ sql: 'CREATE TABLE good_table (id INTEGER)' }]
+            if (sql.includes('PRAGMA table_info')) {
+                return [{ name: 'id', pk: 1 }]
+            }
+            if (sql.includes('LIMIT 0')) return []
+            if (sql.includes('ORDER BY')) {
+                dataSqls.push(sql)
+                return Number(params[0] ?? -1) < 0
+                    ? [{ __starbase_dump_cursor_0: 1, id: 1 }]
+                    : []
             }
             return []
         })
 
         const { engine, storage } = makeEngine()
         const state = await engine.startDump()
-        const persisted = (await storage.get(DUMP_STATE_KEY)) as DumpState | undefined
-        expect(persisted?.tables).toEqual(['good_table'])
-        expect(state.totalRows).toBe(0)
+        const persisted = (await storage.get(DUMP_STATE_KEY)) as
+            | DumpState
+            | undefined
+        expect(persisted?.tables).toEqual(['good_table', 'order items "2026"'])
+        expect(
+            dataSqls.some((sql) =>
+                sql.includes(quoteIdentifier('order items "2026"'))
+            )
+        ).toBe(true)
+        expect(state.totalRows).toBe(2)
     })
 
     it('startDump resumes an in-progress dump instead of restarting', async () => {
@@ -233,7 +305,7 @@ describe('ChunkedDumpEngine cycles', () => {
 
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
-            if (sql.includes('WHERE rowid >')) return []
+            if (sql.includes('ORDER BY')) return []
             return []
         })
 
@@ -270,7 +342,7 @@ describe('ChunkedDumpEngine cycles', () => {
         const c1: ChunkRecord = {
             dumpId: 'dump_x',
             chunkIndex: 1,
-            content: "INSERT INTO \"t\" (\"id\") VALUES (1);\n",
+            content: 'INSERT INTO "t" ("id") VALUES (1);\n',
             bytes: 35,
             createdAt: 2,
         }
@@ -288,11 +360,12 @@ describe('ChunkedDumpEngine cycles', () => {
         let captured = ''
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
-            if (sql.includes('WHERE rowid >')) {
+            if (sql.includes('ORDER BY')) {
                 captured = sql
                 return []
             }
-            if (sql.includes("name NOT LIKE 'tmp_%'")) return [{ name: 'users' }]
+            if (sql.includes("name NOT LIKE 'tmp_%'"))
+                return [{ name: 'users' }]
             if (sql.includes('SELECT sql FROM sqlite_master')) {
                 return [{ sql: 'CREATE TABLE users (id INTEGER)' }]
             }
@@ -351,7 +424,19 @@ const makeMultipartR2 = () => {
             deletedKeys.push(key)
         },
     }
-    return { r2, uploaded, deleted: () => deletedKeys, completed: () => completed, resumedWith: () => resumedWith, createdFor: () => createdFor, objects, signedUrlResult: () => signedUrlResult, setSigned: (v: { url?: string } | null | undefined) => { signedUrlResult = v } }
+    return {
+        r2,
+        uploaded,
+        deleted: () => deletedKeys,
+        completed: () => completed,
+        resumedWith: () => resumedWith,
+        createdFor: () => createdFor,
+        objects,
+        signedUrlResult: () => signedUrlResult,
+        setSigned: (v: { url?: string } | null | undefined) => {
+            signedUrlResult = v
+        },
+    }
 }
 
 const seedCompletedState = async (
@@ -394,7 +479,12 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         const storage = makeStorage()
         const m = makeMultipartR2()
         // 4 chunks of 10 bytes each; partSize 20 → parts of 20, 20, 20(tail).
-        await seedCompletedState(storage, ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD'])
+        await seedCompletedState(storage, [
+            'AAAAAAAAAA',
+            'BBBBBBBBBB',
+            'CCCCCCCCCC',
+            'DDDDDDDDDD',
+        ])
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
             m.r2 as unknown as R2Bucket,
@@ -413,8 +503,37 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         expect(state.finalObjectKey).toBe('dumps/dump_fin/dump_fin.sql')
         expect(state.finalObjectSize).toBe(40)
         expect(state.finalizedAt).toBeDefined()
-        // Per-chunk R2 mirrors are cleaned up after a successful complete.
+        expect(state.temporaryChunksCleanedAt).toBeDefined()
         expect(m.deleted().length).toBe(4)
+        expect(await storage.get(`${DUMP_CHUNK_KEY}:0`)).toBeUndefined()
+    })
+
+    it('retries temporary chunk cleanup after a finalized upload', async () => {
+        const storage = makeStorage()
+        const m = makeMultipartR2()
+        await seedCompletedState(storage, ['data'])
+        const originalDelete = m.r2.delete
+        let attempts = 0
+        m.r2.delete = vi.fn(async (key: string) => {
+            attempts++
+            if (attempts === 1) throw new Error('temporary delete failure')
+            return originalDelete(key)
+        })
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            m.r2 as unknown as R2Bucket,
+            makeDataSource(),
+            makeConfig(),
+            DEFAULT_DUMP_OPTIONS
+        )
+
+        const first = await engine.finalizeDump()
+        expect(first.done).toBe(false)
+        expect(first.state.temporaryChunksCleanedAt).toBeUndefined()
+        expect(await storage.get(`${DUMP_CHUNK_KEY}:0`)).toBeDefined()
+        const second = await engine.finalizeDump()
+        expect(second.done).toBe(true)
+        expect(second.state.temporaryChunksCleanedAt).toBeDefined()
     })
 
     it('resumes an interrupted finalize without re-uploading completed parts', async () => {
@@ -447,7 +566,12 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
     it('returns done:false when the time budget runs out mid-upload, then finishes on the next cycle', async () => {
         const storage = makeStorage()
         const m = makeMultipartR2()
-        await seedCompletedState(storage, ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD'])
+        await seedCompletedState(storage, [
+            'AAAAAAAAAA',
+            'BBBBBBBBBB',
+            'CCCCCCCCCC',
+            'DDDDDDDDDD',
+        ])
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
             m.r2 as unknown as R2Bucket,
@@ -455,7 +579,10 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
             makeConfig(),
             { ...DEFAULT_DUMP_OPTIONS }
         )
-        const first = await engine.finalizeDump({ partSizeBytes: 20, timeBudgetMs: -1 })
+        const first = await engine.finalizeDump({
+            partSizeBytes: 20,
+            timeBudgetMs: -1,
+        })
         expect(first.done).toBe(false)
         expect(first.state.finalizeUploadId).toBe('mpu-1')
 
@@ -489,7 +616,8 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         })
         m.setSigned({ url: 'https://signed.example/dump?sig=abc' })
         const r2 = Object.assign(m.r2, {
-            createSignedUrl: async () => (m.signedUrlResult() as { url?: string }),
+            createSignedUrl: async () =>
+                m.signedUrlResult() as { url?: string },
         })
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
@@ -522,7 +650,8 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         // Unexpected result shape (missing url) must not throw.
         m.setSigned({})
         const r2 = Object.assign(m.r2, {
-            createSignedUrl: async () => (m.signedUrlResult() as { url?: string }),
+            createSignedUrl: async () =>
+                m.signedUrlResult() as { url?: string },
         })
         const engineB = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
@@ -534,6 +663,25 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         await expect(engineB.getPresignedUrl()).resolves.toBeNull()
     })
 
+    it('does not return an empty fallback after temporary chunks are cleaned', async () => {
+        const storage = makeStorage()
+        const m = makeMultipartR2()
+        const state = await seedCompletedState(storage, ['chunk-content'], {
+            finalObjectKey: 'dumps/dump_fin/dump_fin.sql',
+            finalizedAt: 9,
+            temporaryChunksCleanedAt: 10,
+        })
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            m.r2 as unknown as R2Bucket,
+            makeDataSource(),
+            makeConfig(),
+            DEFAULT_DUMP_OPTIONS
+        )
+
+        await expect(engine.assembleDump(state)).resolves.toBeNull()
+    })
+
     it('assembleDump prefers the consolidated final object', async () => {
         const storage = makeStorage()
         const m = makeMultipartR2()
diff --git a/src/export/chunkedDump.ts b/src/export/chunkedDump.ts
index d1dc9d2..2d352ac 100644
--- a/src/export/chunkedDump.ts
+++ b/src/export/chunkedDump.ts
@@ -2,35 +2,15 @@ import { DataSource } from '../types'
 import { StarbaseDBConfiguration } from '../handler'
 import { executeOperation } from './index'
 
-/**
- * Chunked, resumable database dumps with "breathing intervals".
- *
- * The legacy dump endpoint accumulated the entire database into a single
- * in-memory string, which fails on large databases and blows through the
- * 30s Workers request window. This module writes the dump in bounded chunks,
- * yielding ("breathing") between chunks so queued requests are not starved,
- * persists progress in DO storage so work can resume across the 30s window
- * (driven by the DO alarm), and mirrors every chunk to R2 when a binding is
- * available so dumps survive eviction and can be fetched after completion.
- */
-
 export const DUMP_STATE_KEY = 'tmp_dump_state'
 export const DUMP_CHUNK_KEY = 'tmp_dump_chunk'
 
-/** Defaults tuned to stay far under the 30s request window per cycle. */
 export const DEFAULT_DUMP_OPTIONS = {
-    /** Wall-clock budget per work cycle (ms) before we breathe/yield. */
     cycleTimeBudgetMs: 5_000,
-    /** Minimum idle time between cycles when requests are waiting. */
     breathingIntervalMs: 5_000,
-    /** Maximum rows fetched per SELECT batch. */
     rowsPerBatch: 500,
-    /** Approximate serialized size (bytes) that closes a chunk. */
     chunkTargetBytes: 512 * 1024,
-    /** Part size for the consolidated R2 multipart upload. R2 (like S3)
-     * requires non-final parts to be at least 5 MiB. */
     finalizePartSizeBytes: 5 * 1024 * 1024,
-    /** Wall-clock budget per finalize cycle (ms) inside a DO alarm. */
     finalizeTimeBudgetMs: 20_000,
 } as const
 
@@ -44,6 +24,8 @@ export interface DumpOptions {
 }
 
 export type DumpPhase = 'schema' | 'table-data' | 'complete'
+export type DumpCursorMode = 'rowid' | 'primary-key'
+export type DumpCursorValue = string | number | bigint | null
 
 export interface DumpState {
     dumpId: string
@@ -52,53 +34,58 @@ export interface DumpState {
     tables: string[]
     tableIndex: number
     lastFetchedRowId: number | null
-    /** Rowid of the last row written into the current chunk. */
     chunkRowOffset: number
-    /** Total bytes serialized so far (chunks flushed to R2). */
     bytesWritten: number
     chunkIndex: number
     startedAt: number
     updatedAt: number
-    /** Set when the dump finished and the R2 object is ready. */
     completedAt?: number
-    /** Aggregate stats surfaced in status responses. */
     totalRows: number
-    callbackUrl?: string
-
-    /** Consolidated R2 object produced by finalizeDump (multipart upload).
-     * Present once the per-chunk mirrors have been merged into a single
-     * `dumps//` object that supports presigned downloads. */
     finalObjectKey?: string
     finalObjectSize?: number
     finalizedAt?: number
-    /** In-progress multipart bookkeeping: survives eviction so a partially
-     * uploaded finalize can resume without re-uploading completed parts. */
     finalizeUploadId?: string
     finalizeParts?: R2UploadedPart[]
     finalizeBytes?: number
+    currentTable?: string
+    cursorMode?: DumpCursorMode
+    cursorColumns?: string[]
+    cursorAliases?: string[]
+    cursorValues?: DumpCursorValue[] | null
+    temporaryChunksCleanedAt?: number
 }
 
 export interface ChunkRecord {
     dumpId: string
     chunkIndex: number
-    /** Serialized SQL statements for this chunk. */
     content: string
     bytes: number
     createdAt: number
 }
 
 const IDENTIFIER_PATTERN = /^[A-Za-z_][A-Za-z0-9_]*$/
+const CURSOR_ALIAS_PREFIX = '__starbase_dump_cursor_'
 
-/** Guard against SQL injection through table names coming from sqlite_master. */
 export function isSafeIdentifier(name: string): boolean {
     return IDENTIFIER_PATTERN.test(name)
 }
 
+export function quoteIdentifier(name: string): string {
+    return `"${name.replace(/"/g, '""')}"`
+}
+
+export function sqlCommentLabel(value: string): string {
+    return value.replace(/[\r\n]/g, ' ')
+}
+
 function sqlQuote(value: unknown): string {
     if (value === null || value === undefined) {
         return 'NULL'
     }
-    if (typeof value === 'number' || typeof value === 'boolean') {
+    if (typeof value === 'number') {
+        return Number.isFinite(value) ? String(value) : 'NULL'
+    }
+    if (typeof value === 'bigint' || typeof value === 'boolean') {
         return String(value)
     }
     if (value instanceof ArrayBuffer) {
@@ -106,25 +93,32 @@ function sqlQuote(value: unknown): string {
             b.toString(16).padStart(2, '0')
         ).join('')}'`
     }
+    if (ArrayBuffer.isView(value)) {
+        return `X'${Array.from(
+            new Uint8Array(value.buffer, value.byteOffset, value.byteLength),
+            (b) => b.toString(16).padStart(2, '0')
+        ).join('')}'`
+    }
     if (typeof value === 'string') {
         return `'${value.replace(/'/g, "''")}'`
     }
-    // Fallback: serialize deterministically rather than emitting "object".
-    return `'${JSON.stringify(value).replace(/'/g, "''")}'`
+    const serialized = JSON.stringify(value)
+    return `'${(serialized ?? String(value)).replace(/'/g, "''")}'`
 }
 
 export function serializeRows(
     table: string,
-    rows: Record[]
+    rows: Record[],
+    columns?: string[]
 ): { content: string; rowCount: number } {
     if (rows.length === 0) {
         return { content: '', rowCount: 0 }
     }
-    const columns = Object.keys(rows[0])
-    const columnList = columns.map((c) => `"${c}"`).join(', ')
+    const selectedColumns = columns ?? Object.keys(rows[0])
+    const columnList = selectedColumns.map((c) => quoteIdentifier(c)).join(', ')
     const lines = rows.map((row) => {
-        const values = columns.map((c) => sqlQuote(row[c]))
-        return `INSERT INTO "${table}" (${columnList}) VALUES (${values.join(', ')});`
+        const values = selectedColumns.map((c) => sqlQuote(row[c]))
+        return `INSERT INTO ${quoteIdentifier(table)} (${columnList}) VALUES (${values.join(', ')});`
     })
     return { content: `${lines.join('\n')}\n`, rowCount: rows.length }
 }
@@ -137,14 +131,58 @@ export function makeDumpFileName(now = new Date()): string {
     return `dump_${stamp}.sql`
 }
 
-const sleep = (ms: number) => new Promise((resolve) => setTimeout(resolve, ms))
+type TableCursorPlan = {
+    mode: DumpCursorMode
+    columns: string[]
+    aliases: string[]
+}
+
+type CursorRow = Record
+
+function dumpChunkKey(dumpId: string, chunkIndex: number): string {
+    return `${dumpId}/${String(chunkIndex).padStart(8, '0')}.sql`
+}
+
+function normalizeCursorValue(value: unknown): DumpCursorValue {
+    if (value === null || value === undefined) {
+        return null
+    }
+    if (
+        typeof value === 'string' ||
+        typeof value === 'number' ||
+        typeof value === 'bigint'
+    ) {
+        return value
+    }
+    return String(value)
+}
+
+function stripCursorColumns(
+    row: CursorRow,
+    aliases: string[]
+): Record {
+    const result = { ...row }
+    for (const alias of aliases) {
+        delete result[alias]
+    }
+    return result
+}
+
+function rowCursorValues(row: CursorRow, aliases: string[]): DumpCursorValue[] {
+    return aliases.map((alias) => normalizeCursorValue(row[alias]))
+}
+
+function makeCursorAliases(columns: string[]): string[] {
+    const names = new Set(columns.map((column) => column.toLowerCase()))
+    return columns.map((_, index) => {
+        let alias = `${CURSOR_ALIAS_PREFIX}${index}`
+        while (names.has(alias.toLowerCase())) {
+            alias += '_'
+        }
+        return alias
+    })
+}
 
-/**
- * Core engine. Designed to be driven by the Durable Object so each call
- * performs at most one bounded cycle of work (respecting `cycleTimeBudgetMs`),
- * then the DO decides whether to breathe and continue within this request or
- * schedule its alarm for the next cycle.
- */
 export class ChunkedDumpEngine {
     constructor(
         private readonly storage: DurableObjectStorage,
@@ -154,22 +192,24 @@ export class ChunkedDumpEngine {
         private readonly options: Required
     ) {}
 
-    /** Create or resume a dump. Returns the current state after one cycle. */
-    async startDump(callbackUrl?: string): Promise {
+    async startDump(): Promise {
         const existing = await this.storage.get(DUMP_STATE_KEY)
-        if (existing && !existing.completedAt) {
-            // Resume in-progress dump instead of starting over.
+        if (existing) {
             return this.runCycle()
         }
 
         const tablesResult = await executeOperation(
-            [{ sql: "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'tmp_%';" }],
+            [
+                {
+                    sql: "SELECT name FROM sqlite_master WHERE type='table' AND name NOT LIKE 'tmp_%';",
+                },
+            ],
             this.dataSource,
             this.config
         )
         const tables = tablesResult
             .map((row: Record) => String(row.name))
-            .filter((name) => isSafeIdentifier(name))
+            .filter((name) => name.length > 0)
 
         const now = Date.now()
         const state: DumpState = {
@@ -185,7 +225,6 @@ export class ChunkedDumpEngine {
             startedAt: now,
             updatedAt: now,
             totalRows: 0,
-            ...(callbackUrl ? { callbackUrl } : {}),
         }
         await this.storage.put(DUMP_STATE_KEY, state)
         return this.runCycle()
@@ -195,13 +234,10 @@ export class ChunkedDumpEngine {
         return this.storage.get(DUMP_STATE_KEY)
     }
 
-    /**
-     * Run one bounded cycle: serialize schema/data chunks until the time
-     * budget is exhausted, flushing each chunk to R2 (when bound). Returns
-     * the updated state; `completedAt` set when the dump is done.
-     */
     async runCycle(): Promise {
-        const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState
+        const state = (await this.storage.get(
+            DUMP_STATE_KEY
+        )) as DumpState
         if (!state) {
             throw new Error('No dump in progress')
         }
@@ -227,13 +263,14 @@ export class ChunkedDumpEngine {
             }
             if (this.r2) {
                 await this.r2.put(
-                    `${state.dumpId}/${String(record.chunkIndex).padStart(8, '0')}.sql`,
+                    dumpChunkKey(state.dumpId, record.chunkIndex),
                     record.content
                 )
             }
-            // Persist the chunk in DO storage so a boundless environment can
-            // still reassemble; keep only the tail window to bound memory.
-            await this.storage.put(`${DUMP_CHUNK_KEY}:${record.chunkIndex}`, record)
+            await this.storage.put(
+                `${DUMP_CHUNK_KEY}:${record.chunkIndex}`,
+                record
+            )
             state.chunkIndex += 1
             state.bytesWritten += contentBytes
             content = ''
@@ -241,31 +278,37 @@ export class ChunkedDumpEngine {
             chunkDirty = false
         }
 
-        // Phase 1: schema pass. tableIndex is advanced BEFORE the yield point
-        // so a resumed cycle never re-fetches (and duplicate-emits) a schema.
+        const persist = async () => {
+            state.updatedAt = Date.now()
+            await this.storage.put(DUMP_STATE_KEY, state)
+        }
+
         if (state.phase === 'schema') {
             while (state.tableIndex < state.tables.length) {
                 const table = state.tables[state.tableIndex]
                 state.tableIndex++
-                if (!isSafeIdentifier(table)) continue
                 const schemaResult = await executeOperation(
-                    [{
-                        sql: `SELECT sql FROM sqlite_master WHERE type='table' AND name='${table}';`,
-                    }],
+                    [
+                        {
+                            sql: "SELECT sql FROM sqlite_master WHERE type='table' AND name=?;",
+                            params: [table],
+                        },
+                    ],
                     this.dataSource,
                     this.config
                 )
-                if (schemaResult.length) {
-                    const schema = schemaResult[0].sql
-                    const ddl = `\n-- Table: ${table}\n${schema};\n\n`
+                if (schemaResult.length && schemaResult[0]?.sql) {
+                    const ddl = `\n-- Table: ${sqlCommentLabel(table)}\n${String(schemaResult[0].sql)};\n\n`
                     content += ddl
                     contentBytes += ddl.length
                     chunkDirty = true
                 }
-                if (Date.now() - cycleStart >= this.options.cycleTimeBudgetMs) {
+                if (
+                    contentBytes >= this.options.chunkTargetBytes ||
+                    Date.now() - cycleStart >= this.options.cycleTimeBudgetMs
+                ) {
                     await flushChunk()
-                    state.updatedAt = Date.now()
-                    await this.storage.put(DUMP_STATE_KEY, state)
+                    await persist()
                     return state
                 }
             }
@@ -273,55 +316,92 @@ export class ChunkedDumpEngine {
             state.tableIndex = 0
         }
 
-        // Phase 2: data pass in rowid-batched chunks.
         while (state.tableIndex < state.tables.length) {
             const table = state.tables[state.tableIndex]
-            if (!isSafeIdentifier(table)) {
-                state.tableIndex++
+            if (state.currentTable !== table) {
+                state.currentTable = table
+                state.cursorMode = undefined
+                state.cursorColumns = undefined
+                state.cursorAliases = undefined
+                state.cursorValues = null
                 state.lastFetchedRowId = null
-                continue
             }
 
-            const since = state.lastFetchedRowId ?? 0
-            const rowsResult = (await executeOperation(
-                [{
-                    sql: `SELECT rowid AS __rowid, * FROM "${table}" WHERE rowid > ${since} ORDER BY rowid LIMIT ${this.options.rowsPerBatch};`,
-                }],
+            const plan = await this.resolveTableCursor(table, state)
+            const aliases = state.cursorAliases ?? plan.aliases
+            const cursorColumns = state.cursorColumns ?? plan.columns
+            const cursorValues = state.cursorValues
+            const limit = Math.max(1, Math.floor(this.options.rowsPerBatch))
+            const selectedCursorColumns = cursorColumns
+                .map(
+                    (column, index) =>
+                        `${quoteIdentifier(column)} AS ${quoteIdentifier(aliases[index])}`
+                )
+                .join(', ')
+            const orderColumns = cursorColumns
+                .map((column) => quoteIdentifier(column))
+                .join(', ')
+            const where =
+                cursorValues && cursorValues.length === cursorColumns.length
+                    ? state.cursorMode === 'rowid'
+                        ? ` WHERE ${quoteIdentifier(cursorColumns[0])} > ?`
+                        : ` WHERE (${cursorColumns.map((column) => quoteIdentifier(column)).join(', ')}) > (${cursorValues.map(() => '?').join(', ')})`
+                    : ''
+            const params = cursorValues ? [...cursorValues] : []
+            const rowsResult = ((await executeOperation(
+                [
+                    {
+                        sql: `SELECT ${selectedCursorColumns}, * FROM ${quoteIdentifier(table)}${where} ORDER BY ${orderColumns} LIMIT ${limit};`,
+                        params,
+                    },
+                ],
                 this.dataSource,
                 this.config
-            )) as Record[]
+            )) ?? []) as CursorRow[]
 
             if (rowsResult.length === 0) {
                 state.tableIndex++
+                state.currentTable = undefined
+                state.cursorMode = undefined
+                state.cursorColumns = undefined
+                state.cursorAliases = undefined
+                state.cursorValues = null
                 state.lastFetchedRowId = null
                 continue
             }
 
-            const withoutRowidCol = rowsResult.map((row) => {
-                const { __rowid, ...rest } = row
-                state.lastFetchedRowId = Number(__rowid ?? state.lastFetchedRowId)
-                return rest
-            })
+            const nextCursor = rowCursorValues(
+                rowsResult[rowsResult.length - 1],
+                aliases
+            )
+            state.cursorValues = nextCursor
+            if (state.cursorMode === 'rowid') {
+                const numeric = Number(nextCursor[0])
+                state.lastFetchedRowId = Number.isFinite(numeric)
+                    ? numeric
+                    : null
+            }
 
-            const { content: batchSql, rowCount } = serializeRows(table, withoutRowidCol)
+            const rows = rowsResult.map((row) =>
+                stripCursorColumns(row, aliases)
+            )
+            const { content: batchSql, rowCount } = serializeRows(table, rows)
             content += batchSql
             contentBytes += batchSql.length
             state.totalRows += rowCount
+            state.chunkRowOffset += rowCount
             chunkDirty = true
 
-            const chunkClosed =
+            if (
                 contentBytes >= this.options.chunkTargetBytes ||
                 Date.now() - cycleStart >= this.options.cycleTimeBudgetMs
-
-            if (chunkClosed) {
+            ) {
                 await flushChunk()
-                state.updatedAt = Date.now()
-                await this.storage.put(DUMP_STATE_KEY, state)
+                await persist()
                 return state
             }
         }
 
-        // Phase 3: completion marker.
         const done = `\n-- Dump complete: ${state.totalRows} rows, ${state.chunkIndex + (chunkDirty ? 1 : 0)} chunks.\n`
         content += done
         contentBytes += done.length
@@ -330,61 +410,56 @@ export class ChunkedDumpEngine {
 
         state.phase = 'complete'
         state.completedAt = Date.now()
-        state.updatedAt = state.completedAt
-        await this.storage.put(DUMP_STATE_KEY, state)
+        await persist()
         return state
     }
 
-    /** True when another cycle should run right now (time still available). */
     shouldContinue(state: DumpState, requestStart: number): boolean {
         if (state.completedAt) return false
         return Date.now() - requestStart < this.options.cycleTimeBudgetMs
     }
 
-    /** Milliseconds to wait before the next cycle (breathing interval). */
     breathingDelayMs(): number {
         return this.options.breathingIntervalMs
     }
 
-    /**
-     * Consolidate all chunk records into a single R2 object via a multipart
-     * upload (`dumps//`), bounded by a time budget so a
-     * multi-GB dump progresses across several DO alarm invocations instead of
-     * one 30s window. Partial progress (uploaded parts + their etags) is
-     * persisted after every part, so an interrupted finalize resumes with
-     * `resumeMultipartUpload` without re-uploading completed parts.
-     *
-     * Returns `{ done: false }` while parts remain; the DO alarm drives the
-     * remaining cycles. Idempotent once `state.finalizedAt` is set.
-     */
     async finalizeDump(
         options: { partSizeBytes?: number; timeBudgetMs?: number } = {}
     ): Promise<{ done: boolean; state: DumpState }> {
-        const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState
+        const state = (await this.storage.get(
+            DUMP_STATE_KEY
+        )) as DumpState
         if (!state) {
             throw new Error('No dump in progress')
         }
+        if (!state.completedAt) {
+            return { done: false, state }
+        }
         if (state.finalizedAt) {
+            if (this.r2 && !state.temporaryChunksCleanedAt) {
+                const cleaned = await this.cleanupTemporaryChunks(state)
+                return { done: cleaned, state }
+            }
             return { done: true, state }
         }
         if (!this.r2) {
-            // Without R2 there is nothing to consolidate; the DO-storage
-            // streaming reassembly remains the download path.
             state.finalizedAt = Date.now()
+            state.updatedAt = state.finalizedAt
             await this.storage.put(DUMP_STATE_KEY, state)
             return { done: true, state }
         }
-        if (!state.completedAt) {
-            return { done: false, state }
-        }
 
-        const partSize = options.partSizeBytes ?? this.options.finalizePartSizeBytes
+        const partSize = Math.max(
+            1,
+            Math.floor(
+                options.partSizeBytes ?? this.options.finalizePartSizeBytes
+            )
+        )
         const budget = options.timeBudgetMs ?? this.options.finalizeTimeBudgetMs
-        const key = state.finalObjectKey ?? `dumps/${state.dumpId}/${state.fileName}`
+        const key =
+            state.finalObjectKey ?? `dumps/${state.dumpId}/${state.fileName}`
         const cycleStart = Date.now()
 
-        // Recover the in-flight multipart upload from a previous cycle, or
-        // start a fresh one and persist its uploadId immediately.
         let mpu: R2MultipartUpload
         if (state.finalizeUploadId) {
             mpu = this.r2.resumeMultipartUpload(key, state.finalizeUploadId)
@@ -405,8 +480,6 @@ export class ChunkedDumpEngine {
             await this.storage.put(DUMP_STATE_KEY, state)
         }
 
-        // Stream chunk records into part-sized buffers, skipping the bytes
-        // already uploaded as parts in earlier cycles.
         let skipped = state.finalizeBytes ?? 0
         let partNumber = parts.length + 1
         let buffer: Uint8Array[] = []
@@ -430,11 +503,8 @@ export class ChunkedDumpEngine {
             buffer.push(data)
             bufferLen += data.length
 
-            // Non-final parts must be >= 5 MiB, so flush only at partSize.
             if (bufferLen >= partSize) {
                 if (Date.now() - cycleStart >= budget) {
-                    // Out of budget mid-upload: progress (parts + bytes) is
-                    // already persisted, the next cycle continues cleanly.
                     await persist()
                     return { done: false, state }
                 }
@@ -449,22 +519,19 @@ export class ChunkedDumpEngine {
                 buffer = []
                 bufferLen = 0
             } else if (Date.now() - cycleStart >= budget) {
-                // Budget exhausted while accumulating: the un-uploaded buffer
-                // is rebuilt next cycle from the persisted byte offset.
                 await persist()
                 return { done: false, state }
             }
         }
 
         if (parts.length === 0 && bufferLen === 0) {
-            // Empty dump: nothing to upload, finalize without an object.
             state.finalizedAt = Date.now()
             state.updatedAt = state.finalizedAt
             await this.storage.put(DUMP_STATE_KEY, state)
-            return { done: true, state }
+            const cleaned = await this.cleanupTemporaryChunks(state)
+            return { done: cleaned, state }
         }
 
-        // Tail part (allowed to be smaller than partSize).
         if (bufferLen > 0) {
             const part = await mpu.uploadPart(
                 partNumber,
@@ -476,7 +543,7 @@ export class ChunkedDumpEngine {
 
         const object = await mpu.complete(parts)
         state.finalObjectKey = key
-        state.finalObjectSize = object.size ?? state.finalizeBytes
+        state.finalObjectSize = object?.size ?? state.finalizeBytes
         state.finalizedAt = Date.now()
         state.updatedAt = state.finalizedAt
         state.finalizeUploadId = undefined
@@ -484,28 +551,14 @@ export class ChunkedDumpEngine {
         state.finalizeBytes = undefined
         await this.storage.put(DUMP_STATE_KEY, state)
 
-        // Best-effort cleanup of the per-chunk R2 mirrors; DO storage records
-        // stay as the streaming fallback.
-        for (let i = 0; i < state.chunkIndex; i++) {
-            try {
-                await this.r2.delete(
-                    `${state.dumpId}/${String(i).padStart(8, '0')}.sql`
-                )
-            } catch {
-                // Cleanup is best-effort; leftover mirrors are harmless.
-            }
-        }
-        return { done: true, state }
+        const cleaned = await this.cleanupTemporaryChunks(state)
+        return { done: cleaned, state }
     }
 
-    /**
-     * Presigned download URL for the finalized object. Presigned URL
-     * generation is feature-detected: newer workerd runtimes expose
-     * `R2Bucket.createSignedUrl`, older bindings do not — in that case the
-     * caller falls back to the streaming reassembly endpoint.
-     */
     async getPresignedUrl(expiresInSeconds = 3600): Promise {
-        const state = (await this.storage.get(DUMP_STATE_KEY)) as DumpState
+        const state = (await this.storage.get(
+            DUMP_STATE_KEY
+        )) as DumpState
         if (!state?.finalObjectKey || !this.r2) return null
         const creator = (
             this.r2 as R2Bucket & { createSignedUrl?: R2SignedUrlCreator }
@@ -523,16 +576,20 @@ export class ChunkedDumpEngine {
         }
     }
 
-    /** Reassemble the dump from chunk records (or R2 when bound). */
-    async assembleDump(state: DumpState): Promise | null> {
+    async assembleDump(
+        state: DumpState
+    ): Promise | null> {
         if (!state.completedAt) return null
 
-        // Prefer the consolidated multipart object when finalization ran.
         if (state.finalObjectKey && this.r2) {
             const final = await this.r2.get(state.finalObjectKey)
             if (final?.body) return final.body
         }
 
+        if (state.temporaryChunksCleanedAt) {
+            return null
+        }
+
         if (this.r2) {
             const stream = await this.concatenateR2(state)
             if (stream) return stream
@@ -555,16 +612,111 @@ export class ChunkedDumpEngine {
         })
     }
 
-    private async concatenateR2(state: DumpState): Promise | null> {
+    private async resolveTableCursor(
+        table: string,
+        state: DumpState
+    ): Promise {
+        if (state.cursorMode && state.cursorColumns && state.cursorAliases) {
+            return {
+                mode: state.cursorMode,
+                columns: state.cursorColumns,
+                aliases: state.cursorAliases,
+            }
+        }
+
+        const tableInfo = ((await executeOperation(
+            [{ sql: `PRAGMA table_info(${quoteIdentifier(table)});` }],
+            this.dataSource,
+            this.config
+        )) ?? []) as Record[]
+        const columns = tableInfo.map((row) => String(row.name))
+        const primaryKey = tableInfo
+            .filter((row) => Number(row.pk) > 0)
+            .sort((left, right) => Number(left.pk) - Number(right.pk))
+            .map((row) => String(row.name))
+
+        const rowidAliases = ['rowid', '_rowid_', 'oid'].filter(
+            (alias) => !columns.some((column) => column.toLowerCase() === alias)
+        )
+        for (const alias of rowidAliases) {
+            try {
+                await executeOperation(
+                    [
+                        {
+                            sql: `SELECT ${quoteIdentifier(alias)} AS ${quoteIdentifier(`${CURSOR_ALIAS_PREFIX}probe`)} FROM ${quoteIdentifier(table)} LIMIT 0;`,
+                        },
+                    ],
+                    this.dataSource,
+                    this.config
+                )
+                const plan = {
+                    mode: 'rowid' as const,
+                    columns: [alias],
+                    aliases: makeCursorAliases([alias]),
+                }
+                state.cursorMode = plan.mode
+                state.cursorColumns = plan.columns
+                state.cursorAliases = plan.aliases
+                state.cursorValues = null
+                return plan
+            } catch {}
+        }
+
+        if (primaryKey.length === 0) {
+            throw new Error(
+                `Cannot determine a stable cursor for table ${table}`
+            )
+        }
+        const plan = {
+            mode: 'primary-key' as const,
+            columns: primaryKey,
+            aliases: makeCursorAliases(primaryKey),
+        }
+        state.cursorMode = plan.mode
+        state.cursorColumns = plan.columns
+        state.cursorAliases = plan.aliases
+        state.cursorValues = null
+        return plan
+    }
+
+    private async cleanupTemporaryChunks(state: DumpState): Promise {
+        let complete = true
+        for (let i = 0; i < state.chunkIndex; i++) {
+            let r2Deleted = true
+            if (this.r2) {
+                try {
+                    await this.r2.delete(dumpChunkKey(state.dumpId, i))
+                } catch {
+                    r2Deleted = false
+                    complete = false
+                }
+            }
+            if (!r2Deleted) {
+                continue
+            }
+            try {
+                await this.storage.delete(`${DUMP_CHUNK_KEY}:${i}`)
+            } catch {
+                complete = false
+            }
+        }
+        if (complete) {
+            state.temporaryChunksCleanedAt = Date.now()
+        }
+        state.updatedAt = Date.now()
+        await this.storage.put(DUMP_STATE_KEY, state)
+        return complete
+    }
+
+    private async concatenateR2(
+        state: DumpState
+    ): Promise | null> {
         if (!this.r2) return null
-        const head = await this.r2.get(`${state.dumpId}/00000000.sql`)
+        const head = await this.r2.get(dumpChunkKey(state.dumpId, 0))
         if (!head) return null
-        // R2 concatenated reads: stream each part sequentially.
         const parts: ReadableStream[] = []
         for (let i = 0; i < state.chunkIndex; i++) {
-            const obj = await this.r2.get(
-                `${state.dumpId}/${String(i).padStart(8, '0')}.sql`
-            )
+            const obj = await this.r2.get(dumpChunkKey(state.dumpId, i))
             if (obj?.body) {
                 parts.push(obj.body)
             }
@@ -573,10 +725,6 @@ export class ChunkedDumpEngine {
     }
 }
 
-/**
- * Feature-detected presigned URL creator exposed by newer R2 runtime
- * bindings (`R2Bucket.createSignedUrl`).
- */
 type R2SignedUrlCreator = (
     key: string,
     expiresInSeconds: number
@@ -592,7 +740,9 @@ function concatUint8(chunks: Uint8Array[], totalLength: number): Uint8Array {
     return out
 }
 
-function concatStreams(streams: ReadableStream[]): ReadableStream {
+function concatStreams(
+    streams: ReadableStream[]
+): ReadableStream {
     return new ReadableStream({
         async start(controller) {
             for (const stream of streams) {
diff --git a/src/export/dump.job.test.ts b/src/export/dump.job.test.ts
new file mode 100644
index 0000000..b84167a
--- /dev/null
+++ b/src/export/dump.job.test.ts
@@ -0,0 +1,134 @@
+import { beforeEach, describe, expect, it, vi } from 'vitest'
+import { runDumpJob, dumpJobStatus, type DumpEngineHost } from './dump'
+import { executeOperation } from './index'
+import { DUMP_STATE_KEY, type DumpState } from './chunkedDump'
+import type { DataSource } from '../types'
+import type { StarbaseDBConfiguration } from '../handler'
+
+vi.mock('./index', () => ({
+    executeOperation: vi.fn(),
+}))
+
+const makeStorage = () => {
+    const values = new Map()
+    return {
+        get: vi.fn(async (key: string) => values.get(key) as T),
+        put: vi.fn(async (key: string, value: unknown) => {
+            values.set(key, value)
+        }),
+        delete: vi.fn(async (key: string) => {
+            values.delete(key)
+        }),
+    }
+}
+
+const makeDataSource = (): DataSource =>
+    ({ source: 'internal', rpc: {} }) as unknown as DataSource
+
+const makeConfig = (): StarbaseDBConfiguration => ({ role: 'admin' })
+
+const makeHost = () => {
+    const alarms: number[] = []
+    const host: DumpEngineHost = {
+        storage: makeStorage() as unknown as DurableObjectStorage,
+        env: {},
+        dataSource: makeDataSource(),
+        config: makeConfig(),
+        setAlarm: vi.fn(async (time: number) => {
+            alarms.push(time)
+        }),
+    }
+    return { host, alarms }
+}
+
+const setupCompleteQueries = () => {
+    vi.mocked(executeOperation).mockImplementation(async (queries: any[]) => {
+        const query = queries[0]
+        const sql: string = query.sql
+        const params: unknown[] = query.params ?? []
+        if (sql.includes('sqlite_master') && sql.includes('name NOT LIKE')) {
+            return [{ name: 't' }]
+        }
+        if (sql.includes('SELECT sql FROM sqlite_master')) {
+            return [{ sql: 'CREATE TABLE t (id INTEGER)' }]
+        }
+        if (sql.includes('PRAGMA table_info')) {
+            return [{ name: 'id', pk: 1 }]
+        }
+        if (sql.includes('LIMIT 0')) {
+            return []
+        }
+        if (sql.includes('ORDER BY')) {
+            return params.length === 0
+                ? [{ __starbase_dump_cursor_0: 1, id: 1 }]
+                : []
+        }
+        return []
+    })
+}
+
+describe('dump job lifecycle', () => {
+    beforeEach(() => {
+        vi.clearAllMocks()
+    })
+
+    it('reuses a completed job instead of planning a second dump', async () => {
+        setupCompleteQueries()
+        const { host } = makeHost()
+        const fetchSpy = vi.spyOn(globalThis, 'fetch')
+
+        const first = await runDumpJob(
+            host,
+            new URLSearchParams({
+                callbackUrl: 'https://example.invalid/callback',
+            })
+        )
+        const firstText = await first.text()
+        const firstState = (await host.storage.get(DUMP_STATE_KEY)) as DumpState
+        const planCallsAfterFirst = vi
+            .mocked(executeOperation)
+            .mock.calls.filter(([queries]) =>
+                String((queries as any[])[0].sql).includes('name NOT LIKE')
+            ).length
+
+        const second = await runDumpJob(host, new URLSearchParams())
+        const secondText = await second.text()
+        const secondState = (await host.storage.get(
+            DUMP_STATE_KEY
+        )) as DumpState
+        const planCallsAfterSecond = vi
+            .mocked(executeOperation)
+            .mock.calls.filter(([queries]) =>
+                String((queries as any[])[0].sql).includes('name NOT LIKE')
+            ).length
+        const status = await dumpJobStatus(host)
+
+        expect(first.status).toBe(200)
+        expect(second.status).toBe(200)
+        expect(secondText).toBe(firstText)
+        expect(secondState.dumpId).toBe(firstState.dumpId)
+        expect(planCallsAfterSecond).toBe(planCallsAfterFirst)
+        expect(status.status).toBe(200)
+        expect(await status.text()).toBe(firstText)
+        expect(fetchSpy).not.toHaveBeenCalled()
+        expect('callbackUrl' in firstState).toBe(false)
+        fetchSpy.mockRestore()
+    })
+
+    it('returns a resumable response and schedules the next cycle when work remains', async () => {
+        setupCompleteQueries()
+        const { host, alarms } = makeHost()
+        const response = await runDumpJob(
+            host,
+            new URLSearchParams({ chunkBytes: '1' }),
+            Date.now() - 10_000
+        )
+        const body = (await response.json()) as {
+            result: { status: string }
+        }
+
+        expect(response.status).toBe(202)
+        expect(body.result.status).toBe('in-progress')
+        expect(alarms.length).toBeGreaterThan(0)
+    })
+})
diff --git a/src/export/dump.ts b/src/export/dump.ts
index de32882..dc70cdc 100644
--- a/src/export/dump.ts
+++ b/src/export/dump.ts
@@ -5,25 +5,15 @@ import { createResponse } from '../utils'
 import {
     ChunkedDumpEngine,
     DEFAULT_DUMP_OPTIONS,
+    quoteIdentifier,
+    sqlCommentLabel,
+    isSafeIdentifier,
     type DumpOptions,
-    type DumpState,
 } from './chunkedDump'
 
-/**
- * Dump route.
- *
- * Small databases keep the legacy behavior: the dump is fully serialized and
- * returned inline as a downloadable file (well under the 30s window).
- *
- * Large databases exceed the 30s Workers request window, so `?job=1` opts
- * into a resumable job flow driven inside the Durable Object: bounded work
- * cycles with breathing intervals, progress persisted in DO storage, chunks
- * mirrored to R2 when the binding exists, the DO alarm resuming work across
- * window boundaries, and an optional `callbackUrl` invoked on completion.
- */
+export const AUTO_JOB_THRESHOLD_BYTES = 8 * 1024 * 1024
 
 export interface DumpJobEnv {
-    /** Optional R2 binding; when absent, chunks persist in DO storage only. */
     R2_DUMP_BUCKET?: R2Bucket
 }
 
@@ -32,10 +22,12 @@ export interface DumpEngineHost {
     env: DumpJobEnv
     dataSource: DataSource
     config: StarbaseDBConfiguration
-    setAlarm: (time: number, options?: DurableObjectSetAlarmOptions) => Promise
+    setAlarm: (
+        time: number,
+        options?: DurableObjectSetAlarmOptions
+    ) => Promise
 }
 
-/** Query param parsing: `cycleMs`, `breathMs`, `rows`, `chunkBytes`. */
 export function parseDumpOptions(searchParams: URLSearchParams): DumpOptions {
     const options: DumpOptions = {}
     const readNumber = (key: string, target: keyof DumpOptions) => {
@@ -55,7 +47,26 @@ export function parseDumpOptions(searchParams: URLSearchParams): DumpOptions {
     return options
 }
 
-/** Legacy inline dump path, behavior-identical for small databases. */
+function legacyIdentifier(name: string): string {
+    return isSafeIdentifier(name) ? name : quoteIdentifier(name)
+}
+
+function legacyValue(value: unknown): string {
+    if (value === null || value === undefined) return 'NULL'
+    if (typeof value === 'number')
+        return Number.isFinite(value) ? String(value) : 'NULL'
+    if (typeof value === 'bigint' || typeof value === 'boolean')
+        return String(value)
+    if (value instanceof ArrayBuffer) {
+        return `X'${Array.from(new Uint8Array(value), (b) => b.toString(16).padStart(2, '0')).join('')}'`
+    }
+    if (ArrayBuffer.isView(value)) {
+        return `X'${Array.from(new Uint8Array(value.buffer, value.byteOffset, value.byteLength), (b) => b.toString(16).padStart(2, '0')).join('')}'`
+    }
+    if (typeof value === 'string') return `'${value.replace(/'/g, "''")}'`
+    return `'${(JSON.stringify(value) ?? String(value)).replace(/'/g, "''")}'`
+}
+
 export async function legacyDump(
     dataSource: DataSource,
     config: StarbaseDBConfiguration
@@ -67,48 +78,48 @@ export async function legacyDump(
             config
         )
 
-        const tables = tablesResult.map((row: any) => row.name)
-        let dumpContent = 'SQLite format 3\0' // SQLite file header
+        const tables = tablesResult
+            .map((row: Record) => String(row.name))
+            .filter((name: string) => name.length > 0)
+        let dumpContent = 'SQLite format 3\0'
 
         for (const table of tables) {
             const schemaResult = await executeOperation(
-                [{
-                    sql: `SELECT sql FROM sqlite_master WHERE type='table' AND name='${table}';`,
-                }],
+                [
+                    {
+                        sql: "SELECT sql FROM sqlite_master WHERE type='table' AND name=?;",
+                        params: [table],
+                    },
+                ],
                 dataSource,
                 config
             )
 
-            if (schemaResult.length) {
-                const schema = schemaResult[0].sql
-                dumpContent += `\n-- Table: ${table}\n${schema};\n\n`
+            if (schemaResult.length && schemaResult[0]?.sql) {
+                dumpContent += `\n-- Table: ${sqlCommentLabel(table)}\n${String(schemaResult[0].sql)};\n\n`
             }
 
             const dataResult = await executeOperation(
-                [{ sql: `SELECT * FROM ${table};` }],
+                [{ sql: `SELECT * FROM ${quoteIdentifier(table)};` }],
                 dataSource,
                 config
             )
 
             for (const row of dataResult) {
                 const values = Object.values(row).map((value) =>
-                    typeof value === 'string'
-                        ? `'${value.replace(/'/g, "''")}'`
-                        : value
+                    legacyValue(value)
                 )
-                dumpContent += `INSERT INTO ${table} VALUES (${values.join(', ')});\n`
+                dumpContent += `INSERT INTO ${legacyIdentifier(table)} VALUES (${values.join(', ')});\n`
             }
 
             dumpContent += '\n'
         }
 
         const blob = new Blob([dumpContent], { type: 'application/x-sqlite3' })
-
         const headers = new Headers({
             'Content-Type': 'application/x-sqlite3',
             'Content-Disposition': 'attachment; filename="database_dump.sql"',
         })
-
         return new Response(blob, { headers })
     } catch (error: any) {
         console.error('Database Dump Error:', error)
@@ -116,54 +127,68 @@ export async function legacyDump(
     }
 }
 
-/**
- * Chunked job dump, executed inside the Durable Object. Runs bounded cycles
- * within the current request, schedules the alarm for the next breathing
- * interval when work remains, and returns either the finished dump stream or
- * a 202 progress payload.
- */
+export async function shouldUseResumableDump(
+    dataSource: DataSource,
+    config: StarbaseDBConfiguration
+): Promise {
+    try {
+        const readPragma = async (sql: string): Promise => {
+            const result = await executeOperation([{ sql }], dataSource, config)
+            const row = result[0] as Record | undefined
+            const value = row ? Object.values(row)[0] : undefined
+            const number = Number(value)
+            return Number.isFinite(number) ? number : 0
+        }
+        const pageCount = await readPragma('PRAGMA page_count;')
+        const pageSize = await readPragma('PRAGMA page_size;')
+        return pageCount * pageSize >= AUTO_JOB_THRESHOLD_BYTES
+    } catch {
+        return true
+    }
+}
+
 export async function runDumpJob(
     host: DumpEngineHost,
     searchParams: URLSearchParams,
     requestStart: number = Date.now()
 ): Promise {
     try {
-        const options = parseDumpOptions(searchParams)
+        const parsedOptions = parseDumpOptions(searchParams)
+        const options = { ...DEFAULT_DUMP_OPTIONS, ...parsedOptions }
         const engine = new ChunkedDumpEngine(
             host.storage,
             host.env.R2_DUMP_BUCKET,
             host.dataSource,
             host.config,
-            { ...DEFAULT_DUMP_OPTIONS, ...options }
+            options
         )
 
         let state = await engine.getState()
-        if (!state || state.completedAt) {
-            state = await engine.startDump(
-                searchParams.get('callbackUrl') ?? undefined
-            )
-        } else {
+        if (!state) {
+            state = await engine.startDump()
+        } else if (!state.completedAt) {
             state = await engine.runCycle()
         }
 
-        // Keep working while this request still has budget (5s default cycle
-        // budget bounds each burst; breathing happens between bursts).
         while (
             !state.completedAt &&
-            Date.now() - requestStart < DEFAULT_DUMP_OPTIONS.cycleTimeBudgetMs
+            Date.now() - requestStart < options.cycleTimeBudgetMs
         ) {
             state = await engine.runCycle()
         }
 
         if (state.completedAt) {
-            // Kick off the R2 multipart consolidation (presigned-URL-ready
-            // single object) asynchronously via the DO alarm; the response
-            // below streams from the chunk records either way.
-            if (host.env.R2_DUMP_BUCKET && !state.finalizedAt) {
+            if (
+                host.env.R2_DUMP_BUCKET &&
+                (!state.finalizedAt || !state.temporaryChunksCleanedAt)
+            ) {
                 try {
                     await host.setAlarm(Date.now() + 1_000)
                 } catch (alarmError) {
-                    console.error('Failed to schedule dump finalize:', alarmError)
+                    console.error(
+                        'Failed to schedule dump finalize:',
+                        alarmError
+                    )
                 }
             }
             const stream = await engine.assembleDump(state)
@@ -177,10 +202,8 @@ export async function runDumpJob(
             }
         }
 
-        // Work remains: breathe, then let the DO alarm drive the next cycle.
-        const resumeAt = Date.now() + DEFAULT_DUMP_OPTIONS.breathingIntervalMs
+        const resumeAt = Date.now() + options.breathingIntervalMs
         await host.setAlarm(resumeAt)
-
         return createResponse(
             {
                 dumpId: state.dumpId,
@@ -205,14 +228,7 @@ export async function runDumpJob(
     }
 }
 
-/**
- * Status + fetch endpoint for in-progress/completed chunked dumps.
- * `GET /export/dump?job=1` while a job runs returns 202 progress; once the
- * job is complete the assembled dump streams back.
- */
-export async function dumpJobStatus(
-    host: DumpEngineHost
-): Promise {
+export async function dumpJobStatus(host: DumpEngineHost): Promise {
     const engine = new ChunkedDumpEngine(
         host.storage,
         host.env.R2_DUMP_BUCKET,
@@ -243,8 +259,6 @@ export async function dumpJobStatus(
             202
         )
     }
-    // Completed + consolidated: prefer a presigned, expiring download URL
-    // when the runtime supports it; otherwise fall back to streaming.
     if (state.finalObjectKey) {
         const downloadUrl = await engine.getPresignedUrl()
         if (downloadUrl) {
@@ -277,11 +291,6 @@ export async function dumpJobStatus(
     })
 }
 
-/**
- * One bounded finalize cycle for the completed dump: consolidates chunks
- * into the single R2 object via multipart upload. Driven by the DO alarm;
- * reschedules itself while parts remain (`{ done: false }`).
- */
 export async function runDumpFinalize(host: DumpEngineHost): Promise {
     const engine = new ChunkedDumpEngine(
         host.storage,
diff --git a/src/handler.dump.test.ts b/src/handler.dump.test.ts
new file mode 100644
index 0000000..5ebc7db
--- /dev/null
+++ b/src/handler.dump.test.ts
@@ -0,0 +1,87 @@
+import { beforeEach, describe, expect, it, vi } from 'vitest'
+import { StarbaseDB } from './handler'
+import type { DataSource } from './types'
+
+const makeContext = () =>
+    ({
+        waitUntil: vi.fn(),
+    }) as unknown as ExecutionContext
+
+const makeSource = (large: boolean) => {
+    const startDumpJob = vi.fn(async () => new Response('job', { status: 202 }))
+    const executeQuery = vi.fn(async ({ sql }: { sql: string }) => {
+        if (sql.includes('PRAGMA page_count')) {
+            return [{ page_count: large ? 4096 : 1 }]
+        }
+        if (sql.includes('PRAGMA page_size')) {
+            return [{ page_size: 4096 }]
+        }
+        if (sql.includes('SELECT name FROM sqlite_master')) {
+            return [{ name: 't' }]
+        }
+        if (sql.includes('SELECT sql FROM sqlite_master')) {
+            return [{ sql: 'CREATE TABLE t (id INTEGER)' }]
+        }
+        return []
+    })
+    const dataSource = {
+        source: 'internal',
+        rpc: {
+            executeQuery,
+            startDumpJob,
+            dumpJobStatus: vi.fn(async () => new Response('status')),
+        },
+    } as unknown as DataSource
+    return { dataSource, startDumpJob, executeQuery }
+}
+
+describe('StarbaseDB dump routing', () => {
+    beforeEach(() => {
+        vi.clearAllMocks()
+    })
+
+    it('automatically routes a large internal dump to a resumable job', async () => {
+        const { dataSource, startDumpJob } = makeSource(true)
+        const app = new StarbaseDB({
+            dataSource,
+            config: {
+                role: 'admin',
+                features: { export: true, rls: false, allowlist: false },
+            },
+        })
+        const response = await app.handle(
+            new Request('https://example.test/export/dump'),
+            makeContext()
+        )
+
+        expect(response.status).toBe(202)
+        expect(await response.text()).toBe('job')
+        expect(startDumpJob).toHaveBeenCalledOnce()
+    })
+
+    it('keeps small internal dumps synchronous and preserves explicit job requests', async () => {
+        const { dataSource, startDumpJob } = makeSource(false)
+        const app = new StarbaseDB({
+            dataSource,
+            config: {
+                role: 'admin',
+                features: { export: true, rls: false, allowlist: false },
+            },
+        })
+
+        const small = await app.handle(
+            new Request('https://example.test/export/dump'),
+            makeContext()
+        )
+        expect(small.status).toBe(200)
+        expect(await small.text()).toContain('CREATE TABLE t')
+        expect(startDumpJob).not.toHaveBeenCalled()
+
+        const explicit = await app.handle(
+            new Request('https://example.test/export/dump?job=1'),
+            makeContext()
+        )
+        expect(explicit.status).toBe(202)
+        expect(startDumpJob).toHaveBeenCalledOnce()
+    })
+})
diff --git a/src/handler.ts b/src/handler.ts
index b69f7ce..5dc471e 100644
--- a/src/handler.ts
+++ b/src/handler.ts
@@ -6,7 +6,7 @@ import { DataSource } from './types'
 import { LiteREST } from './literest'
 import { executeQuery, executeTransaction } from './operation'
 import { createResponse, QueryRequest, QueryTransactionRequest } from './utils'
-import { dumpDatabaseRoute } from './export/dump'
+import { dumpDatabaseRoute, shouldUseResumableDump } from './export/dump'
 import { exportTableToJsonRoute } from './export/json'
 import { exportTableToCsvRoute } from './export/csv'
 import { importDumpRoute } from './import/dump'
@@ -122,16 +122,21 @@ export class StarbaseDB {
         if (this.getFeature('export')) {
             this.app.get('/export/dump', this.isInternalSource, async (c) => {
                 const url = new URL(c.req.raw.url)
-                const wantsJob = url.searchParams.get('job') === '1'
+                let wantsJob = url.searchParams.get('job') === '1'
+
+                if (
+                    !wantsJob &&
+                    this.dataSource.source === 'internal' &&
+                    (await shouldUseResumableDump(this.dataSource, this.config))
+                ) {
+                    wantsJob = true
+                }
 
                 if (wantsJob && this.dataSource.source === 'internal') {
-                    // Chunked/resumable job flow runs inside the Durable
-                    // Object via RPC; progress persists across the 30s window.
                     const searchParams: Record = {}
                     url.searchParams.forEach((value, key) => {
                         searchParams[key] = value
                     })
-                    // Narrow RPC view keeps hono's generic inference shallow.
                     const rpc = this.dataSource.rpc as unknown as {
                         startDumpJob: (
                             config: StarbaseDBConfiguration,
@@ -144,19 +149,25 @@ export class StarbaseDB {
                 return dumpDatabaseRoute(this.dataSource, this.config)
             })
 
-            this.app.get('/export/dump/status', this.isInternalSource, async () => {
-                if (this.dataSource.source === 'internal') {
-                    const rpc = this.dataSource.rpc as unknown as {
-                        dumpJobStatus: (config: StarbaseDBConfiguration) => Promise
+            this.app.get(
+                '/export/dump/status',
+                this.isInternalSource,
+                async () => {
+                    if (this.dataSource.source === 'internal') {
+                        const rpc = this.dataSource.rpc as unknown as {
+                            dumpJobStatus: (
+                                config: StarbaseDBConfiguration
+                            ) => Promise
+                        }
+                        return await rpc.dumpJobStatus(this.config)
                     }
-                    return await rpc.dumpJobStatus(this.config)
+                    return createResponse(
+                        undefined,
+                        'Chunked dump status requires the internal data source',
+                        400
+                    )
                 }
-                return createResponse(
-                    undefined,
-                    'Chunked dump status requires the internal data source',
-                    400
-                )
-            })
+            )
 
             this.app.get(
                 '/export/json/:tableName',

From 3de8b882dbb30e76a48a3c829a7d2c6385193fd7 Mon Sep 17 00:00:00 2001
From: Furox-Art <177975472+Furox-Art@users.noreply.github.com>
Date: Wed, 23 Sep 2026 22:55:29 +0300
Subject: [PATCH 4/5] fix(export): resume legacy rowid cursors

---
 src/export/chunkedDump.test.ts | 13 ++++++++++---
 src/export/chunkedDump.ts      | 17 +++++++++++++++--
 2 files changed, 25 insertions(+), 5 deletions(-)

diff --git a/src/export/chunkedDump.test.ts b/src/export/chunkedDump.test.ts
index 08692dd..afd659c 100644
--- a/src/export/chunkedDump.test.ts
+++ b/src/export/chunkedDump.test.ts
@@ -293,7 +293,7 @@ describe('ChunkedDumpEngine cycles', () => {
             phase: 'table-data',
             tables: ['users'],
             tableIndex: 0,
-            lastFetchedRowId: null,
+            lastFetchedRowId: 3,
             chunkRowOffset: 0,
             bytesWritten: 0,
             chunkIndex: 0,
@@ -303,16 +303,23 @@ describe('ChunkedDumpEngine cycles', () => {
         }
         await storage.put(DUMP_STATE_KEY, seed)
 
+        let dataSql = ''
+        let dataParams: unknown[] = []
         vi.mocked(executeOperation).mockImplementation(async (queries: any) => {
             const sql: string = queries[0].sql
-            if (sql.includes('ORDER BY')) return []
+            if (sql.includes('ORDER BY')) {
+                dataSql = sql
+                dataParams = queries[0].params ?? []
+                return []
+            }
             return []
         })
 
         const state = await engine.startDump()
-        // Must NOT have re-planned tables (phase kept, no new dumpId).
         expect(state.dumpId).toBe('dump_seed')
         expect(state.tables).toEqual(['users'])
+        expect(dataSql).toContain('WHERE "rowid" > ?')
+        expect(dataParams).toEqual([3])
     })
 
     it('assembleDump concatenates persisted chunks in order', async () => {
diff --git a/src/export/chunkedDump.ts b/src/export/chunkedDump.ts
index 2d352ac..c132ebf 100644
--- a/src/export/chunkedDump.ts
+++ b/src/export/chunkedDump.ts
@@ -318,19 +318,32 @@ export class ChunkedDumpEngine {
 
         while (state.tableIndex < state.tables.length) {
             const table = state.tables[state.tableIndex]
+            const hasLegacyRowIdCursor =
+                state.currentTable === undefined &&
+                state.cursorMode === undefined &&
+                state.lastFetchedRowId !== null
             if (state.currentTable !== table) {
                 state.currentTable = table
                 state.cursorMode = undefined
                 state.cursorColumns = undefined
                 state.cursorAliases = undefined
                 state.cursorValues = null
-                state.lastFetchedRowId = null
+                if (!hasLegacyRowIdCursor) {
+                    state.lastFetchedRowId = null
+                }
             }
 
             const plan = await this.resolveTableCursor(table, state)
             const aliases = state.cursorAliases ?? plan.aliases
             const cursorColumns = state.cursorColumns ?? plan.columns
-            const cursorValues = state.cursorValues
+            const cursorValues =
+                state.cursorValues ??
+                (state.cursorMode === 'rowid' && state.lastFetchedRowId !== null
+                    ? [state.lastFetchedRowId]
+                    : null)
+            if (cursorValues && !state.cursorValues) {
+                state.cursorValues = cursorValues
+            }
             const limit = Math.max(1, Math.floor(this.options.rowsPerBatch))
             const selectedCursorColumns = cursorColumns
                 .map(

From 25fa20f2adec20862aaf6557ff38e28de9b99ff9 Mon Sep 17 00:00:00 2001
From: Furox-Art <177975472+Furox-Art@users.noreply.github.com>
Date: Thu, 24 Sep 2026 11:26:03 +0300
Subject: [PATCH 5/5] fix(export): enforce R2 multipart part sizes

---
 src/export/chunkedDump.test.ts | 226 ++++++++++++++++++++++++++------
 src/export/chunkedDump.ts      | 228 ++++++++++++++++++++++++---------
 src/export/dump.job.test.ts    |  19 ++-
 src/export/dump.ts             |   8 +-
 4 files changed, 381 insertions(+), 100 deletions(-)

diff --git a/src/export/chunkedDump.test.ts b/src/export/chunkedDump.test.ts
index afd659c..f410705 100644
--- a/src/export/chunkedDump.test.ts
+++ b/src/export/chunkedDump.test.ts
@@ -2,6 +2,8 @@ import { describe, it, expect, vi, beforeEach } from 'vitest'
 import {
     ChunkedDumpEngine,
     DEFAULT_DUMP_OPTIONS,
+    MIN_R2_PART_SIZE_BYTES,
+    normalizePartSizeBytes,
     isSafeIdentifier,
     quoteIdentifier,
     makeDumpFileName,
@@ -13,6 +15,7 @@ import {
     type DumpState,
 } from './chunkedDump'
 import { executeOperation } from './index'
+import { parseDumpOptions } from './dump'
 import type { DataSource } from '../types'
 import type { StarbaseDBConfiguration } from '../handler'
 
@@ -392,11 +395,14 @@ type UploadedPart = { partNumber: number; etag: string }
 
 const makeMultipartR2 = () => {
     const uploaded: { partNumber: number; size: number }[] = []
+    const uploadedContents: { partNumber: number; content: string }[] = []
     let completed: UploadedPart[] | null = null
     let resumedWith: string | null = null
     let createdFor: string | null = null
+    let aborted = 0
     let signedUrlResult: { url?: string } | null | undefined = undefined
     const deletedKeys: string[] = []
+    const puts: { key: string; size: number; content: string }[] = []
     const objects = new Map }>()
 
     const mpu = {
@@ -404,9 +410,15 @@ const makeMultipartR2 = () => {
         uploadId: 'mpu-1',
         uploadPart: async (partNumber: number, value: Uint8Array) => {
             uploaded.push({ partNumber, size: value.length })
+            uploadedContents.push({
+                partNumber,
+                content: new TextDecoder().decode(value),
+            })
             return { partNumber, etag: `etag-${partNumber}` }
         },
-        abort: async () => undefined,
+        abort: async () => {
+            aborted++
+        },
         complete: async (parts: UploadedPart[]) => {
             completed = parts
             const total = uploaded.reduce((sum, p) => sum + p.size, 0)
@@ -425,7 +437,23 @@ const makeMultipartR2 = () => {
             mpu.key = key
             return mpu
         },
-        put: async () => undefined,
+        put: async (key: string, value: string | Uint8Array) => {
+            const bytes =
+                typeof value === 'string'
+                    ? new TextEncoder().encode(value)
+                    : value
+            const content = new TextDecoder().decode(bytes)
+            puts.push({ key, size: bytes.length, content })
+            objects.set(key, {
+                body: new ReadableStream({
+                    start(controller) {
+                        if (bytes.length > 0) controller.enqueue(bytes)
+                        controller.close()
+                    },
+                }),
+            })
+            return { key, size: bytes.length }
+        },
         get: async (key: string) => objects.get(key) ?? null,
         delete: async (key: string) => {
             deletedKeys.push(key)
@@ -434,6 +462,9 @@ const makeMultipartR2 = () => {
     return {
         r2,
         uploaded,
+        uploadedContents: () => uploadedContents,
+        puts: () => puts,
+        aborted: () => aborted,
         deleted: () => deletedKeys,
         completed: () => completed,
         resumedWith: () => resumedWith,
@@ -465,6 +496,7 @@ const seedCompletedState = async (
         updatedAt: 2,
         completedAt: 3,
         totalRows: 0,
+        finalizePartSizeBytes: MIN_R2_PART_SIZE_BYTES,
         ...extra,
     }
     for (let i = 0; i < chunks.length; i++) {
@@ -482,36 +514,42 @@ const seedCompletedState = async (
 }
 
 describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
-    it('uploads part-sized parts, completes the object and cleans chunk mirrors', async () => {
+    it('aggregates uneven chunks into fixed-size parts with a small tail', async () => {
         const storage = makeStorage()
         const m = makeMultipartR2()
-        // 4 chunks of 10 bytes each; partSize 20 → parts of 20, 20, 20(tail).
-        await seedCompletedState(storage, [
-            'AAAAAAAAAA',
-            'BBBBBBBBBB',
-            'CCCCCCCCCC',
-            'DDDDDDDDDD',
-        ])
+        const partSize = MIN_R2_PART_SIZE_BYTES
+        const chunks = ['A'.repeat(partSize + 1), 'B'.repeat(partSize - 1), 'C']
+        await seedCompletedState(storage, chunks)
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
             m.r2 as unknown as R2Bucket,
             makeDataSource(),
             makeConfig(),
-            { ...DEFAULT_DUMP_OPTIONS }
+            DEFAULT_DUMP_OPTIONS
         )
-        const { done, state } = await engine.finalizeDump({ partSizeBytes: 20 })
+        const { done, state } = await engine.finalizeDump({ partSizeBytes: 1 })
 
         expect(done).toBe(true)
         expect(m.createdFor()).toBe('dumps/dump_fin/dump_fin.sql')
-        // 40 bytes at partSize 20 → exactly two full parts, no tail.
-        expect(m.uploaded.map((p) => p.size)).toEqual([20, 20])
-        expect(m.uploaded.map((p) => p.partNumber)).toEqual([1, 2])
-        expect(m.completed()?.length).toBe(2)
+        expect(m.uploaded.map((p) => p.size)).toEqual([partSize, partSize, 1])
+        expect(m.uploaded.map((p) => p.partNumber)).toEqual([1, 2, 3])
+        expect(
+            m
+                .uploadedContents()
+                .map((p) => p.content)
+                .join('')
+        ).toBe(chunks.join(''))
+        expect(m.completed()).toEqual([
+            { partNumber: 1, etag: 'etag-1' },
+            { partNumber: 2, etag: 'etag-2' },
+            { partNumber: 3, etag: 'etag-3' },
+        ])
         expect(state.finalObjectKey).toBe('dumps/dump_fin/dump_fin.sql')
-        expect(state.finalObjectSize).toBe(40)
+        expect(state.finalObjectSize).toBe(partSize * 2 + 1)
+        expect(state.finalizePartSizeBytes).toBe(MIN_R2_PART_SIZE_BYTES)
         expect(state.finalizedAt).toBeDefined()
         expect(state.temporaryChunksCleanedAt).toBeDefined()
-        expect(m.deleted().length).toBe(4)
+        expect(m.deleted().length).toBe(3)
         expect(await storage.get(`${DUMP_CHUNK_KEY}:0`)).toBeUndefined()
     })
 
@@ -543,58 +581,166 @@ describe('finalizeDump (R2 multipart upload + presigned URL)', () => {
         expect(second.state.temporaryChunksCleanedAt).toBeDefined()
     })
 
-    it('resumes an interrupted finalize without re-uploading completed parts', async () => {
+    it('resumes after multiple compliant parts without re-uploading bytes', async () => {
         const storage = makeStorage()
         const m = makeMultipartR2()
-        const chunks = ['AAAAAAAAAA', 'BBBBBBBBBB', 'CCCCCCCCCC', 'DDDDDDDDDD']
-        // First part (chunk 0 + chunk 1 = 20 bytes) already uploaded.
+        const partSize = MIN_R2_PART_SIZE_BYTES
+        const chunks = [
+            'A'.repeat(partSize + 1),
+            'B'.repeat(partSize - 1),
+            'C'.repeat(partSize + 1),
+            'D'.repeat(partSize),
+        ]
         await seedCompletedState(storage, chunks, {
             finalObjectKey: 'dumps/dump_fin/dump_fin.sql',
             finalizeUploadId: 'mpu-9',
-            finalizeParts: [{ partNumber: 1, etag: 'etag-1' }],
-            finalizeBytes: 20,
+            finalizeParts: [
+                { partNumber: 1, etag: 'etag-1' },
+                { partNumber: 2, etag: 'etag-2' },
+            ],
+            finalizeBytes: partSize * 2,
+            finalizePartSizeBytes: partSize,
         })
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
             m.r2 as unknown as R2Bucket,
             makeDataSource(),
             makeConfig(),
-            { ...DEFAULT_DUMP_OPTIONS }
+            DEFAULT_DUMP_OPTIONS
         )
-        const { done } = await engine.finalizeDump({ partSizeBytes: 20 })
+        const { done } = await engine.finalizeDump({ partSizeBytes: 1 })
 
         expect(done).toBe(true)
         expect(m.resumedWith()).toBe('mpu-9')
-        // Only the remaining 20 bytes upload, as part 2 — no re-upload.
-        expect(m.uploaded).toEqual([{ partNumber: 2, size: 20 }])
-        expect(m.completed()?.length).toBe(2)
+        expect(m.uploaded).toEqual([
+            { partNumber: 3, size: partSize },
+            { partNumber: 4, size: partSize },
+            { partNumber: 5, size: 1 },
+        ])
+        expect(m.completed()).toEqual([
+            { partNumber: 1, etag: 'etag-1' },
+            { partNumber: 2, etag: 'etag-2' },
+            { partNumber: 3, etag: 'etag-3' },
+            { partNumber: 4, etag: 'etag-4' },
+            { partNumber: 5, etag: 'etag-5' },
+        ])
     })
 
-    it('returns done:false when the time budget runs out mid-upload, then finishes on the next cycle', async () => {
+    it('normalizes under-minimum API and persisted part sizes', async () => {
+        expect(normalizePartSizeBytes(1)).toBe(MIN_R2_PART_SIZE_BYTES)
+        expect(
+            parseDumpOptions(new URLSearchParams('partBytes=1'))
+                .finalizePartSizeBytes
+        ).toBe(MIN_R2_PART_SIZE_BYTES)
+        expect(
+            parseDumpOptions(new URLSearchParams('partBytes=0'))
+                .finalizePartSizeBytes
+        ).toBe(MIN_R2_PART_SIZE_BYTES)
+        expect(
+            parseDumpOptions(new URLSearchParams('partBytes=invalid'))
+                .finalizePartSizeBytes
+        ).toBe(MIN_R2_PART_SIZE_BYTES)
+
         const storage = makeStorage()
         const m = makeMultipartR2()
-        await seedCompletedState(storage, [
-            'AAAAAAAAAA',
-            'BBBBBBBBBB',
-            'CCCCCCCCCC',
-            'DDDDDDDDDD',
-        ])
+        await seedCompletedState(storage, ['x'], {
+            finalizePartSizeBytes: 1,
+        })
         const engine = new ChunkedDumpEngine(
             storage as unknown as DurableObjectStorage,
             m.r2 as unknown as R2Bucket,
             makeDataSource(),
             makeConfig(),
-            { ...DEFAULT_DUMP_OPTIONS }
+            DEFAULT_DUMP_OPTIONS
         )
-        const first = await engine.finalizeDump({
-            partSizeBytes: 20,
-            timeBudgetMs: -1,
+        const result = await engine.finalizeDump({ partSizeBytes: 1 })
+
+        expect(result.done).toBe(true)
+        expect(result.state.finalizePartSizeBytes).toBe(MIN_R2_PART_SIZE_BYTES)
+        expect(m.uploaded).toEqual([{ partNumber: 1, size: 1 }])
+    })
+
+    it('aborts an invalid persisted multipart before restarting it', async () => {
+        const storage = makeStorage()
+        const m = makeMultipartR2()
+        await seedCompletedState(storage, ['x'], {
+            finalizeUploadId: 'old-upload',
+            finalizeParts: [{ partNumber: 1, etag: 'old-etag' }],
+            finalizeBytes: 1,
+            finalizePartSizeBytes: 1,
         })
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            m.r2 as unknown as R2Bucket,
+            makeDataSource(),
+            makeConfig(),
+            DEFAULT_DUMP_OPTIONS
+        )
+        const result = await engine.finalizeDump({ partSizeBytes: 1 })
+
+        expect(result.done).toBe(true)
+        expect(m.aborted()).toBe(1)
+        expect(m.uploaded).toEqual([{ partNumber: 1, size: 1 }])
+        expect(m.completed()?.length).toBe(1)
+        expect(result.state.finalizePartSizeBytes).toBe(MIN_R2_PART_SIZE_BYTES)
+    })
+
+    it('writes an empty downloadable object without creating a multipart upload', async () => {
+        const storage = makeStorage()
+        const m = makeMultipartR2()
+        await seedCompletedState(storage, [])
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            m.r2 as unknown as R2Bucket,
+            makeDataSource(),
+            makeConfig(),
+            DEFAULT_DUMP_OPTIONS
+        )
+        const result = await engine.finalizeDump({ partSizeBytes: 1 })
+
+        expect(result.done).toBe(true)
+        expect(m.createdFor()).toBeNull()
+        expect(m.uploaded).toEqual([])
+        expect(m.completed()).toBeNull()
+        expect(m.aborted()).toBe(0)
+        expect(m.puts()).toEqual([
+            {
+                key: 'dumps/dump_fin/dump_fin.sql',
+                size: 0,
+                content: '',
+            },
+        ])
+        expect(result.state.finalObjectKey).toBe('dumps/dump_fin/dump_fin.sql')
+        expect(result.state.finalObjectSize).toBe(0)
+        expect(result.state.finalizeUploadId).toBeUndefined()
+        expect(result.state.finalizeParts).toBeUndefined()
+        expect(result.state.finalizedAt).toBeDefined()
+
+        const stream = await engine.assembleDump(result.state)
+        expect(stream).not.toBeNull()
+        expect(await new Response(stream as ReadableStream).text()).toBe('')
+    })
+
+    it('returns done:false when the time budget runs out, then finishes compliant parts', async () => {
+        const storage = makeStorage()
+        const m = makeMultipartR2()
+        const partSize = MIN_R2_PART_SIZE_BYTES
+        const chunks = ['A'.repeat(partSize + 1), 'B'.repeat(partSize - 1), 'C']
+        await seedCompletedState(storage, chunks)
+        const engine = new ChunkedDumpEngine(
+            storage as unknown as DurableObjectStorage,
+            m.r2 as unknown as R2Bucket,
+            makeDataSource(),
+            makeConfig(),
+            DEFAULT_DUMP_OPTIONS
+        )
+        const first = await engine.finalizeDump({ timeBudgetMs: -1 })
         expect(first.done).toBe(false)
         expect(first.state.finalizeUploadId).toBe('mpu-1')
 
-        const second = await engine.finalizeDump({ partSizeBytes: 20 })
+        const second = await engine.finalizeDump()
         expect(second.done).toBe(true)
+        expect(m.uploaded.map((p) => p.size)).toEqual([partSize, partSize, 1])
         expect(second.state.finalizedAt).toBeDefined()
     })
 
diff --git a/src/export/chunkedDump.ts b/src/export/chunkedDump.ts
index c132ebf..5c81e72 100644
--- a/src/export/chunkedDump.ts
+++ b/src/export/chunkedDump.ts
@@ -4,13 +4,14 @@ import { executeOperation } from './index'
 
 export const DUMP_STATE_KEY = 'tmp_dump_state'
 export const DUMP_CHUNK_KEY = 'tmp_dump_chunk'
+export const MIN_R2_PART_SIZE_BYTES = 5 * 1024 * 1024
 
 export const DEFAULT_DUMP_OPTIONS = {
     cycleTimeBudgetMs: 5_000,
     breathingIntervalMs: 5_000,
     rowsPerBatch: 500,
     chunkTargetBytes: 512 * 1024,
-    finalizePartSizeBytes: 5 * 1024 * 1024,
+    finalizePartSizeBytes: MIN_R2_PART_SIZE_BYTES,
     finalizeTimeBudgetMs: 20_000,
 } as const
 
@@ -47,6 +48,7 @@ export interface DumpState {
     finalizeUploadId?: string
     finalizeParts?: R2UploadedPart[]
     finalizeBytes?: number
+    finalizePartSizeBytes?: number
     currentTable?: string
     cursorMode?: DumpCursorMode
     cursorColumns?: string[]
@@ -183,14 +185,31 @@ function makeCursorAliases(columns: string[]): string[] {
     })
 }
 
+export function normalizePartSizeBytes(value: unknown): number {
+    const numeric = typeof value === 'number' ? value : Number(value)
+    if (!Number.isFinite(numeric)) {
+        return MIN_R2_PART_SIZE_BYTES
+    }
+    return Math.max(MIN_R2_PART_SIZE_BYTES, Math.floor(numeric))
+}
+
 export class ChunkedDumpEngine {
     constructor(
         private readonly storage: DurableObjectStorage,
         private readonly r2: R2Bucket | undefined,
         private readonly dataSource: DataSource,
         private readonly config: StarbaseDBConfiguration,
-        private readonly options: Required
-    ) {}
+        options: Required
+    ) {
+        this.options = {
+            ...options,
+            finalizePartSizeBytes: normalizePartSizeBytes(
+                options.finalizePartSizeBytes
+            ),
+        }
+    }
+
+    private readonly options: Required
 
     async startDump(): Promise {
         const existing = await this.storage.get(DUMP_STATE_KEY)
@@ -225,6 +244,7 @@ export class ChunkedDumpEngine {
             startedAt: now,
             updatedAt: now,
             totalRows: 0,
+            finalizePartSizeBytes: this.options.finalizePartSizeBytes,
         }
         await this.storage.put(DUMP_STATE_KEY, state)
         return this.runCycle()
@@ -445,10 +465,32 @@ export class ChunkedDumpEngine {
         if (!state) {
             throw new Error('No dump in progress')
         }
+
+        const persistedPartSize = state.finalizePartSizeBytes
+        const requestedPartSize =
+            persistedPartSize ??
+            options.partSizeBytes ??
+            this.options.finalizePartSizeBytes
+        const partSize = normalizePartSizeBytes(requestedPartSize)
+        const partSizeChanged = state.finalizePartSizeBytes !== partSize
+        const resetMultipart =
+            Boolean(state.finalizeUploadId) &&
+            (persistedPartSize === undefined || persistedPartSize !== partSize)
+        if (partSizeChanged) {
+            state.finalizePartSizeBytes = partSize
+        }
+        const persistPartSize = async () => {
+            if (!partSizeChanged) return
+            state.updatedAt = Date.now()
+            await this.storage.put(DUMP_STATE_KEY, state)
+        }
+
         if (!state.completedAt) {
+            await persistPartSize()
             return { done: false, state }
         }
         if (state.finalizedAt) {
+            await persistPartSize()
             if (this.r2 && !state.temporaryChunksCleanedAt) {
                 const cleaned = await this.cleanupTemporaryChunks(state)
                 return { done: cleaned, state }
@@ -456,23 +498,45 @@ export class ChunkedDumpEngine {
             return { done: true, state }
         }
         if (!this.r2) {
+            await persistPartSize()
             state.finalizedAt = Date.now()
             state.updatedAt = state.finalizedAt
             await this.storage.put(DUMP_STATE_KEY, state)
             return { done: true, state }
         }
 
-        const partSize = Math.max(
-            1,
-            Math.floor(
-                options.partSizeBytes ?? this.options.finalizePartSizeBytes
-            )
-        )
         const budget = options.timeBudgetMs ?? this.options.finalizeTimeBudgetMs
         const key =
             state.finalObjectKey ?? `dumps/${state.dumpId}/${state.fileName}`
-        const cycleStart = Date.now()
+        let multipartReset = false
+        if (resetMultipart && state.finalizeUploadId) {
+            const upload = this.r2.resumeMultipartUpload(
+                key,
+                state.finalizeUploadId
+            )
+            try {
+                await upload.abort()
+            } catch {
+                return { done: false, state }
+            }
+            state.finalizeUploadId = undefined
+            state.finalizeParts = undefined
+            state.finalizeBytes = undefined
+            multipartReset = true
+        }
+        if (partSizeChanged || multipartReset) {
+            state.updatedAt = Date.now()
+            await this.storage.put(DUMP_STATE_KEY, state)
+        }
+        const hasContent = await this.hasChunkContent(state)
+        if (!hasContent) {
+            if (state.bytesWritten > 0) {
+                throw new Error('Dump chunk data is missing')
+            }
+            return this.finalizeEmptyDump(state, key)
+        }
 
+        const cycleStart = Date.now()
         let mpu: R2MultipartUpload
         if (state.finalizeUploadId) {
             mpu = this.r2.resumeMultipartUpload(key, state.finalizeUploadId)
@@ -493,10 +557,10 @@ export class ChunkedDumpEngine {
             await this.storage.put(DUMP_STATE_KEY, state)
         }
 
-        let skipped = state.finalizeBytes ?? 0
+        let skipped = Math.max(0, state.finalizeBytes ?? 0)
         let partNumber = parts.length + 1
-        let buffer: Uint8Array[] = []
-        let bufferLen = 0
+        const buffer = new Uint8Array(partSize)
+        let bufferLength = 0
 
         for (let i = 0; i < state.chunkIndex; i++) {
             const record = await this.storage.get(
@@ -504,54 +568,63 @@ export class ChunkedDumpEngine {
             )
             if (!record?.content) continue
             const encoded = encoder.encode(record.content)
-            let data = encoded
+            let offset = 0
+            if (skipped >= encoded.length) {
+                skipped -= encoded.length
+                continue
+            }
             if (skipped > 0) {
-                if (skipped >= encoded.length) {
-                    skipped -= encoded.length
-                    continue
-                }
-                data = encoded.subarray(skipped)
+                offset = skipped
                 skipped = 0
             }
-            buffer.push(data)
-            bufferLen += data.length
 
-            if (bufferLen >= partSize) {
-                if (Date.now() - cycleStart >= budget) {
+            while (offset < encoded.length) {
+                const length = Math.min(
+                    partSize - bufferLength,
+                    encoded.length - offset
+                )
+                buffer.set(
+                    encoded.subarray(offset, offset + length),
+                    bufferLength
+                )
+                bufferLength += length
+                offset += length
+
+                if (bufferLength === partSize) {
+                    if (Date.now() - cycleStart >= budget) {
+                        await persist()
+                        return { done: false, state }
+                    }
+                    const part = await mpu.uploadPart(
+                        partNumber,
+                        buffer.slice(0, partSize)
+                    )
+                    parts.push({ partNumber, etag: part.etag })
+                    state.finalizeBytes = (state.finalizeBytes ?? 0) + partSize
+                    bufferLength = 0
                     await persist()
-                    return { done: false, state }
+                    partNumber++
                 }
-                const part = await mpu.uploadPart(
-                    partNumber,
-                    concatUint8(buffer, bufferLen)
-                )
-                parts.push({ partNumber, etag: part.etag })
-                state.finalizeBytes = (state.finalizeBytes ?? 0) + bufferLen
-                await persist()
-                partNumber++
-                buffer = []
-                bufferLen = 0
-            } else if (Date.now() - cycleStart >= budget) {
-                await persist()
-                return { done: false, state }
             }
         }
 
-        if (parts.length === 0 && bufferLen === 0) {
-            state.finalizedAt = Date.now()
-            state.updatedAt = state.finalizedAt
-            await this.storage.put(DUMP_STATE_KEY, state)
-            const cleaned = await this.cleanupTemporaryChunks(state)
-            return { done: cleaned, state }
-        }
-
-        if (bufferLen > 0) {
+        if (bufferLength > 0) {
+            if (Date.now() - cycleStart >= budget) {
+                await persist()
+                return { done: false, state }
+            }
             const part = await mpu.uploadPart(
                 partNumber,
-                concatUint8(buffer, bufferLen)
+                buffer.slice(0, bufferLength)
             )
             parts.push({ partNumber, etag: part.etag })
-            state.finalizeBytes = (state.finalizeBytes ?? 0) + bufferLen
+            state.finalizeBytes = (state.finalizeBytes ?? 0) + bufferLength
+            bufferLength = 0
+            await persist()
+        }
+
+        if (parts.length === 0) {
+            return { done: false, state }
         }
 
         const object = await mpu.complete(parts)
@@ -692,6 +765,55 @@ export class ChunkedDumpEngine {
         return plan
     }
 
+    private async hasChunkContent(state: DumpState): Promise {
+        for (let i = 0; i < state.chunkIndex; i++) {
+            const record = await this.storage.get(
+                `${DUMP_CHUNK_KEY}:${i}`
+            )
+            if (record?.content) {
+                return true
+            }
+        }
+        return false
+    }
+
+    private async finalizeEmptyDump(
+        state: DumpState,
+        key: string
+    ): Promise<{ done: boolean; state: DumpState }> {
+        const r2 = this.r2
+        if (!r2) {
+            throw new Error(
+                'R2 binding is required for empty dump finalization'
+            )
+        }
+        if (state.finalizeUploadId) {
+            const upload = r2.resumeMultipartUpload(key, state.finalizeUploadId)
+            try {
+                await upload.abort()
+            } catch {
+                return { done: false, state }
+            }
+            state.finalizeUploadId = undefined
+            state.finalizeParts = undefined
+            state.finalizeBytes = undefined
+            state.updatedAt = Date.now()
+            await this.storage.put(DUMP_STATE_KEY, state)
+        }
+
+        await r2.put(key, new Uint8Array(0))
+        state.finalObjectKey = key
+        state.finalObjectSize = 0
+        state.finalizedAt = Date.now()
+        state.updatedAt = state.finalizedAt
+        state.finalizeUploadId = undefined
+        state.finalizeParts = undefined
+        state.finalizeBytes = undefined
+        await this.storage.put(DUMP_STATE_KEY, state)
+        const cleaned = await this.cleanupTemporaryChunks(state)
+        return { done: cleaned, state }
+    }
+
     private async cleanupTemporaryChunks(state: DumpState): Promise {
         let complete = true
         for (let i = 0; i < state.chunkIndex; i++) {
@@ -743,16 +865,6 @@ type R2SignedUrlCreator = (
     expiresInSeconds: number
 ) => Promise<{ url?: string } | null>
 
-function concatUint8(chunks: Uint8Array[], totalLength: number): Uint8Array {
-    const out = new Uint8Array(totalLength)
-    let offset = 0
-    for (const chunk of chunks) {
-        out.set(chunk, offset)
-        offset += chunk.length
-    }
-    return out
-}
-
 function concatStreams(
     streams: ReadableStream[]
 ): ReadableStream {
diff --git a/src/export/dump.job.test.ts b/src/export/dump.job.test.ts
index b84167a..bc35f7e 100644
--- a/src/export/dump.job.test.ts
+++ b/src/export/dump.job.test.ts
@@ -1,7 +1,11 @@
 import { beforeEach, describe, expect, it, vi } from 'vitest'
 import { runDumpJob, dumpJobStatus, type DumpEngineHost } from './dump'
 import { executeOperation } from './index'
-import { DUMP_STATE_KEY, type DumpState } from './chunkedDump'
+import {
+    DUMP_STATE_KEY,
+    MIN_R2_PART_SIZE_BYTES,
+    type DumpState,
+} from './chunkedDump'
 import type { DataSource } from '../types'
 import type { StarbaseDBConfiguration } from '../handler'
 
@@ -115,6 +119,19 @@ describe('dump job lifecycle', () => {
         fetchSpy.mockRestore()
     })
 
+    it('normalizes the part size in persisted job state', async () => {
+        setupCompleteQueries()
+        const { host } = makeHost()
+        const response = await runDumpJob(
+            host,
+            new URLSearchParams({ partBytes: '1' })
+        )
+        await response.arrayBuffer()
+        const state = (await host.storage.get(DUMP_STATE_KEY)) as DumpState
+
+        expect(state.finalizePartSizeBytes).toBe(MIN_R2_PART_SIZE_BYTES)
+    })
+
     it('returns a resumable response and schedules the next cycle when work remains', async () => {
         setupCompleteQueries()
         const { host, alarms } = makeHost()
diff --git a/src/export/dump.ts b/src/export/dump.ts
index dc70cdc..c14fa3b 100644
--- a/src/export/dump.ts
+++ b/src/export/dump.ts
@@ -5,6 +5,7 @@ import { createResponse } from '../utils'
 import {
     ChunkedDumpEngine,
     DEFAULT_DUMP_OPTIONS,
+    normalizePartSizeBytes,
     quoteIdentifier,
     sqlCommentLabel,
     isSafeIdentifier,
@@ -42,7 +43,12 @@ export function parseDumpOptions(searchParams: URLSearchParams): DumpOptions {
     readNumber('breathMs', 'breathingIntervalMs')
     readNumber('rows', 'rowsPerBatch')
     readNumber('chunkBytes', 'chunkTargetBytes')
-    readNumber('partBytes', 'finalizePartSizeBytes')
+    const rawPartBytes = searchParams.get('partBytes')
+    if (rawPartBytes !== null) {
+        options.finalizePartSizeBytes = normalizePartSizeBytes(
+            Number(rawPartBytes)
+        )
+    }
     readNumber('finalizeMs', 'finalizeTimeBudgetMs')
     return options
 }