From 8ff56b48372a013d5cbadc86e47d91cfbce800f2 Mon Sep 17 00:00:00 2001 From: onenewcode Date: Sat, 12 Sep 2026 21:59:29 +0800 Subject: [PATCH] feat(mongodb): add Compass-style collection import/export --- apps/desktop/src/App.vue | 1 + .../components/document/DocumentBrowser.vue | 10 + .../components/document/MongoImportDialog.vue | 453 +++ .../__tests__/MongoImportDialog.spec.ts | 131 + .../components/import/TableImportDialog.vue | 30 +- .../src/components/layout/AppDialogs.vue | 2 + .../sidebar/SidebarTreeRuntimeHost.vue | 28 +- .../mongoCollectionImportExportMenu.spec.ts | 39 + .../__tests__/useDialogSources.mongo.spec.ts | 64 + .../useSidebarTreeExportRuntime.spec.ts | 36 + .../useSidebarTreeToolRuntime.docs.spec.ts | 20 + .../src/composables/useDialogSources.ts | 21 + .../useSidebarTreeExportRuntime.ts | 63 +- .../composables/useSidebarTreeToolRuntime.ts | 11 + apps/desktop/src/i18n/locales/az.ts | 41 + apps/desktop/src/i18n/locales/en.ts | 41 + apps/desktop/src/i18n/locales/es.ts | 41 + apps/desktop/src/i18n/locales/it.ts | 41 + apps/desktop/src/i18n/locales/ja.ts | 41 + apps/desktop/src/i18n/locales/ko.ts | 41 + apps/desktop/src/i18n/locales/pt-BR.ts | 41 + apps/desktop/src/i18n/locales/tr.ts | 41 + apps/desktop/src/i18n/locales/zh-CN.ts | 41 + apps/desktop/src/i18n/locales/zh-TW.ts | 41 + .../lib/__tests__/import/importSource.spec.ts | 24 + apps/desktop/src/lib/backend/api.ts | 19 + apps/desktop/src/lib/backend/http.ts | 124 + apps/desktop/src/lib/backend/tauri.ts | 184 + apps/desktop/src/lib/import/importSource.ts | 25 + apps/desktop/src/lib/table/tableImport.ts | 8 + apps/desktop/src/stores/connectionStore.ts | 13 + crates/dbx-core/src/db/mongo_driver.rs | 128 +- crates/dbx-core/src/lib.rs | 1 + crates/dbx-core/src/mongodb_import_export.rs | 3085 +++++++++++++++++ crates/dbx-core/src/table_import.rs | 16 +- crates/dbx-web/src/main.rs | 15 + crates/dbx-web/src/routes/mod.rs | 1 + .../src/routes/mongodb_import_export.rs | 528 +++ docs/content/docs/mongodb.cn.mdx | 28 +- docs/content/docs/mongodb.mdx | 28 +- docs/content/docs/mongodb.tr.mdx | 28 +- docs/content/docs/table-import.cn.mdx | 2 + docs/content/docs/table-import.mdx | 2 + docs/content/docs/table-import.tr.mdx | 2 + src-tauri/src/commands/mod.rs | 1 + .../src/commands/mongodb_import_export.rs | 92 + src-tauri/src/lib.rs | 5 + 47 files changed, 5626 insertions(+), 52 deletions(-) create mode 100644 apps/desktop/src/components/document/MongoImportDialog.vue create mode 100644 apps/desktop/src/components/document/__tests__/MongoImportDialog.spec.ts create mode 100644 apps/desktop/src/components/sidebar/__tests__/mongoCollectionImportExportMenu.spec.ts create mode 100644 apps/desktop/src/composables/__tests__/useDialogSources.mongo.spec.ts create mode 100644 apps/desktop/src/lib/__tests__/import/importSource.spec.ts create mode 100644 apps/desktop/src/lib/import/importSource.ts create mode 100644 crates/dbx-core/src/mongodb_import_export.rs create mode 100644 crates/dbx-web/src/routes/mongodb_import_export.rs create mode 100644 src-tauri/src/commands/mongodb_import_export.rs diff --git a/apps/desktop/src/App.vue b/apps/desktop/src/App.vue index 36be5ac48..38d52ba73 100644 --- a/apps/desktop/src/App.vue +++ b/apps/desktop/src/App.vue @@ -271,6 +271,7 @@ async function initializeUpdatePreparation() { dialogs.showTransferDialog, dialogs.showSqlFileDialog, dialogs.showTableImportDialog, + dialogs.showMongoImportDialog, dialogs.showTableDataGenerateDialog, dialogs.showDatabaseExportDialog, dialogs.showSchemaDiffDialog, diff --git a/apps/desktop/src/components/document/DocumentBrowser.vue b/apps/desktop/src/components/document/DocumentBrowser.vue index 93f3bce2e..be32fb517 100644 --- a/apps/desktop/src/components/document/DocumentBrowser.vue +++ b/apps/desktop/src/components/document/DocumentBrowser.vue @@ -96,6 +96,7 @@ import { TABLE_FONT_SIZE_MAX, TABLE_FONT_SIZE_MIN, useSettingsStore } from "@/st import { useToast } from "@/composables/useToast"; import { copyToClipboard } from "@/lib/common/clipboard"; import JsonEditNode from "./JsonEditNode.vue"; + import type { EditNode } from "@/types/editor"; import type { ColumnInfo, DatabaseType, QueryResult, QueryTab } from "@/types/database"; import type { CustomSaveHandler } from "@/composables/useDataGridEditor"; @@ -2227,6 +2228,15 @@ function focusSearch(): boolean { return documentJsonEditorRef.value?.openSearch() ?? false; } +watch( + () => connectionStore.mongoImportCompleted, + (completed) => { + if (!completed) return; + if (completed.connectionId !== props.connectionId || completed.database !== props.database || completed.collection !== props.collection) return; + void refreshDocuments(); + }, +); + watch([viewMode, isEditing, selectedIdx], ([mode, editing, index]) => { if (mode === "document" && !editing && index !== null) return; documentViewerSearchActive.value = false; diff --git a/apps/desktop/src/components/document/MongoImportDialog.vue b/apps/desktop/src/components/document/MongoImportDialog.vue new file mode 100644 index 000000000..39686e8e7 --- /dev/null +++ b/apps/desktop/src/components/document/MongoImportDialog.vue @@ -0,0 +1,453 @@ + + + diff --git a/apps/desktop/src/components/document/__tests__/MongoImportDialog.spec.ts b/apps/desktop/src/components/document/__tests__/MongoImportDialog.spec.ts new file mode 100644 index 000000000..282f951a7 --- /dev/null +++ b/apps/desktop/src/components/document/__tests__/MongoImportDialog.spec.ts @@ -0,0 +1,131 @@ +// @vitest-environment happy-dom + +import { defineComponent, h, nextTick, ref } from "vue"; +import { createApp } from "vue"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { MongoImportPreview } from "@/lib/backend/api"; + +const api = vi.hoisted(() => ({ + previewMongodbImportFile: vi.fn(), + importMongodbFile: vi.fn(), + cancelMongodbImport: vi.fn(), + releaseMongodbImportSource: vi.fn(), +})); + +vi.mock("vue-i18n", async (importOriginal) => ({ + ...(await importOriginal()), + useI18n: () => ({ t: (key: string) => key }), +})); + +vi.mock("@/lib/backend/api", () => api); +vi.mock("@/lib/backend/tauriRuntime", () => ({ isTauriRuntime: () => false })); +vi.mock("@/stores/connectionStore", () => ({ + useConnectionStore: () => ({ getConfig: () => ({ id: "c1", name: "mongo", read_only: false }), mongoImportCompleted: null }), +})); +vi.mock("@/composables/useToast", () => ({ useToast: () => ({ toast: vi.fn() }) })); +vi.mock("@/lib/database/productionExecutionGuard", () => ({ + executeWithProductionContextGuard: async ({ execute }: { execute: () => Promise }) => execute(), +})); + +import MongoImportDialog from "../MongoImportDialog.vue"; + +function previewResult(overrides: Partial = {}): MongoImportPreview { + return { + sourceRef: "src-1", + format: "csv", + fileName: "orders.csv", + filePath: "/tmp/orders.csv", + sizeBytes: 12, + columns: [{ name: "id", inferredType: "string" }], + rows: [{ id: "1" }], + warnings: [], + errors: [], + estimatedRows: 1, + estimatedRowsExact: true, + ...overrides, + }; +} + +describe("MongoImportDialog", () => { + beforeEach(() => { + vi.useFakeTimers(); + api.previewMongodbImportFile.mockReset().mockResolvedValue(previewResult()); + api.releaseMongodbImportSource.mockReset().mockResolvedValue(true); + }); + + afterEach(() => { + vi.useRealTimers(); + document.body.innerHTML = ""; + }); + + it("keeps the collection target in the wizard", async () => { + const app = createApp( + defineComponent({ + setup() { + return () => + h(MongoImportDialog, { + open: true, + connectionId: "c1", + database: "shop", + collection: "orders", + }); + }, + }), + ); + app.mount(document.body); + await nextTick(); + expect(document.body.textContent).toContain("shop.orders"); + app.unmount(); + }); + + it("reuses the uploaded sourceRef on later previews and releases it on close", async () => { + const open = ref(true); + const app = createApp( + defineComponent({ + setup() { + return () => + h(MongoImportDialog, { + open: open.value, + "onUpdate:open": (value: boolean) => { + open.value = value; + }, + connectionId: "c1", + database: "shop", + collection: "orders", + }); + }, + }), + ); + app.mount(document.body); + await nextTick(); + + const input = document.querySelector("input[type='file']") as HTMLInputElement; + const file = new File(["id\n1"], "orders.csv", { type: "text/csv" }); + Object.defineProperty(input, "files", { value: [file] }); + input.dispatchEvent(new Event("change")); + await vi.advanceTimersByTimeAsync(200); + await Promise.resolve(); + await nextTick(); + + expect(api.previewMongodbImportFile).toHaveBeenCalledTimes(1); + const firstSource = api.previewMongodbImportFile.mock.calls[0]?.[0]; + expect(firstSource).toBeInstanceOf(File); + expect((firstSource as File).name).toBe("orders.csv"); + expect(api.previewMongodbImportFile.mock.calls[0]?.[1]).toMatchObject({ sourceRef: null }); + + const header = document.querySelector("input[type='checkbox']") as HTMLInputElement; + header.click(); + await vi.advanceTimersByTimeAsync(200); + await Promise.resolve(); + await nextTick(); + + expect(api.previewMongodbImportFile).toHaveBeenCalledTimes(2); + expect(api.previewMongodbImportFile.mock.calls[1]?.[0]).toBe("/tmp/orders.csv"); + expect(api.previewMongodbImportFile.mock.calls[1]?.[1]).toMatchObject({ sourceRef: "src-1" }); + + open.value = false; + await nextTick(); + expect(api.releaseMongodbImportSource).toHaveBeenCalledWith("src-1"); + app.unmount(); + }); +}); diff --git a/apps/desktop/src/components/import/TableImportDialog.vue b/apps/desktop/src/components/import/TableImportDialog.vue index 42e16166f..7c71bc739 100644 --- a/apps/desktop/src/components/import/TableImportDialog.vue +++ b/apps/desktop/src/components/import/TableImportDialog.vue @@ -24,10 +24,12 @@ import { requiredImportTargetColumns, resolveTableImportElapsed, suggestImportTargetDataTypes, + TABLE_IMPORT_ENCODING_OPTIONS, tableImportProgressPercent, validateImportMappings, type TableImportWizardStep, } from "@/lib/table/tableImport"; +import { importPreviewInput, importSourceDisplayName, uploadedImportSourceFromPreview } from "@/lib/import/importSource"; import { getDataTypeOptions } from "@/lib/table/tableStructureEditorState"; import { metadataSchemaForConnection, tableStructureDatabaseTypeForConnection } from "@/lib/database/jdbcDialect"; import type { ColumnInfo } from "@/types/database"; @@ -139,13 +141,7 @@ const formatOptions: Array<{ value: api.TableImportSourceFormat; icon: any; labe { value: "sql", icon: FileCode, labelKey: "tableImport.formatSql", descriptionKey: "tableImport.formatSqlDescription" }, ]; -const encodingOptions: Array<{ value: api.TableImportTextEncoding; labelKey: string }> = [ - { value: "auto", labelKey: "tableImport.encodingAuto" }, - { value: "utf8", labelKey: "tableImport.encodingUtf8" }, - { value: "gbk", labelKey: "tableImport.encodingGbk" }, - { value: "utf16Le", labelKey: "tableImport.encodingUtf16Le" }, - { value: "utf16Be", labelKey: "tableImport.encodingUtf16Be" }, -]; +const encodingOptions = TABLE_IMPORT_ENCODING_OPTIONS; const wizardSteps: Array<{ value: TableImportWizardStep; labelKey: string }> = [ { value: "source", labelKey: "tableImport.stepSource" }, @@ -331,7 +327,7 @@ function suggestedTableName(name: string) { } function sourceName(source: ImportSource): string { - return typeof source === "string" ? source.split(/[\\/]/).pop() || source : source.name; + return importSourceDisplayName(source); } function uniqueTableName(baseName: string, usedNames: Set): string { @@ -556,9 +552,9 @@ async function loadTargetColumns() { } async function previewSelectedImportFile(fileOrPath: string | File) { - const reusablePreview = preview.value?.sourceRef ? preview.value : null; - return api.previewTableImportFile(reusablePreview?.filePath || fileOrPath, { - sourceRef: reusablePreview?.sourceRef || null, + const input = importPreviewInput(uploadedImportSourceFromPreview(preview.value), fileOrPath); + return api.previewTableImportFile(input.fileOrPath, { + sourceRef: input.sourceRef, sourceFormat: sourceFormat.value, parseOptions: parseOptions.value, previewLimit: Math.max(1, Number(previewLimit.value) || 50), @@ -648,11 +644,11 @@ async function prepareBatchSources(sources: ImportSource[]) { }); const sheets = format === "excel" && initialPreview.sheets?.length ? initialPreview.sheets : [""]; for (const sheetName of sheets) { - const reusableSource = initialPreview.sourceRef ? initialPreview.filePath : source; + const input = importPreviewInput(uploadedImportSourceFromPreview(initialPreview), source); const effectiveSheetName = sheetName && sheetName === initialPreview.sheets?.[0] ? "" : sheetName; const taskPreview = effectiveSheetName - ? await api.previewTableImportFile(reusableSource, { - sourceRef: initialPreview.sourceRef || null, + ? await api.previewTableImportFile(input.fileOrPath, { + sourceRef: input.sourceRef, sourceFormat: format, parseOptions: taskParseOptions(format, effectiveSheetName), previewLimit: Math.max(1, Number(previewLimit.value) || 50), @@ -1025,9 +1021,9 @@ async function reloadBatchPreviewsForEncoding() { for (const task of batchTasks.value) { // SQL 脚本同样是文本源,编码变化时需要重新预览 if (!isDelimitedFormat(task.format) && task.format !== "sql") continue; - const reusableSource = task.preview.sourceRef ? task.preview.filePath : task.source; - const nextPreview = await api.previewTableImportFile(reusableSource, { - sourceRef: task.preview.sourceRef || null, + const input = importPreviewInput(uploadedImportSourceFromPreview(task.preview), task.source); + const nextPreview = await api.previewTableImportFile(input.fileOrPath, { + sourceRef: input.sourceRef, sourceFormat: task.format, parseOptions: taskParseOptions(task.format, task.sheetName), previewLimit: Math.max(1, Number(previewLimit.value) || 50), diff --git a/apps/desktop/src/components/layout/AppDialogs.vue b/apps/desktop/src/components/layout/AppDialogs.vue index a7b6107f9..599fcd0cc 100644 --- a/apps/desktop/src/components/layout/AppDialogs.vue +++ b/apps/desktop/src/components/layout/AppDialogs.vue @@ -13,6 +13,7 @@ const SqlFileExecutionDialog = defineAsyncComponent(() => import("@/components/s const SchemaDiagramDialog = defineAsyncComponent(() => import("@/components/diagram/SchemaDiagramDialog.vue")); const DatabaseDocsDialog = defineAsyncComponent(() => import("@/components/docs/DatabaseDocsDialog.vue")); const TableImportDialog = defineAsyncComponent(() => import("@/components/import/TableImportDialog.vue")); +const MongoImportDialog = defineAsyncComponent(() => import("@/components/document/MongoImportDialog.vue")); const FieldLineageDialog = defineAsyncComponent(() => import("@/components/lineage/FieldLineageDialog.vue")); const ConfigPassphraseDialog = defineAsyncComponent(() => import("@/components/config/ConfigPassphraseDialog.vue")); const ConfigConnectionSelectDialog = defineAsyncComponent(() => import("@/components/config/ConfigConnectionSelectDialog.vue")); @@ -295,6 +296,7 @@ watch( :prefill-schema="dialogs.tableImportPrefillSchema.value" :prefill-table="dialogs.tableImportPrefillTable.value" /> + acceptedSelectionIds, }); -const { openAllDatabasesExport, openDataCompare, openDatabaseExport, openDatabaseSearch, openDiagram, openDocs, openFieldLineage, openScheduledBackups, openSchemaDiff, openSchemaDiffForRoutine, openSqlFileExecution, openStructureEditor, openTableImport, openTransfer } = useSidebarTreeToolRuntime({ - activeNode, - connectionStore, - queryStore, - settingsStore, - tableChildObjectName: tableChildDropObjectName, - acceptedSelectionIds: () => acceptedSelectionIds, -}); +const { openAllDatabasesExport, openDataCompare, openDatabaseExport, openDatabaseSearch, openDiagram, openDocs, openFieldLineage, openMongoImport, openScheduledBackups, openSchemaDiff, openSchemaDiffForRoutine, openSqlFileExecution, openStructureEditor, openTableImport, openTransfer } = + useSidebarTreeToolRuntime({ + activeNode, + connectionStore, + queryStore, + settingsStore, + tableChildObjectName: tableChildDropObjectName, + acceptedSelectionIds: () => acceptedSelectionIds, + }); const emit = defineEmits<{ "rename-started": []; @@ -5932,6 +5933,15 @@ function buildSpecialSidebarMenu(context: SidebarMenuFactoryContext): boolean { if (canCloneMongoCollection.value) { items.push({ label: t("contextMenu.cloneCollection"), action: openCloneMongoCollectionDialog, icon: CopyPlus }); } + items.push({ label: t("contextMenu.importData"), action: openMongoImport, icon: Download }); + items.push({ + label: t("contextMenu.exportData"), + icon: Upload, + children: [ + { label: "CSV", action: () => void exportMongoCollection("csv") }, + { label: "NDJSON", action: () => void exportMongoCollection("ndjson") }, + ], + }); if (canDropMongoCollection.value) { items.push({ label: "", separator: true }); items.push({ label: t("contextMenu.dropCollection"), action: dropMongoCollection, icon: Trash2, shortcut: shortcutDelete, variant: "destructive" as const }); diff --git a/apps/desktop/src/components/sidebar/__tests__/mongoCollectionImportExportMenu.spec.ts b/apps/desktop/src/components/sidebar/__tests__/mongoCollectionImportExportMenu.spec.ts new file mode 100644 index 000000000..8cf0c8bab --- /dev/null +++ b/apps/desktop/src/components/sidebar/__tests__/mongoCollectionImportExportMenu.spec.ts @@ -0,0 +1,39 @@ +import { readFileSync } from "node:fs"; +import { dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import { describe, expect, it } from "vitest"; + +const here = dirname(fileURLToPath(import.meta.url)); + +function read(relativePath: string) { + return readFileSync(resolve(here, relativePath), "utf8"); +} + +describe("mongo collection import/export UI", () => { + it("adds table-style import/export to the original mongo-collection menu", () => { + const menu = read("../SidebarTreeRuntimeHost.vue"); + const start = menu.indexOf('if (node.type === "mongo-collection") {\n items.push({ label: t("contextMenu.copyName")'); + expect(start).toBeGreaterThan(-1); + const end = menu.indexOf("return true;", start); + const block = menu.slice(start, end); + expect(block).toContain('t("contextMenu.viewData")'); + expect(block).toContain('t("contextMenu.cloneCollection")'); + expect(block.indexOf('t("contextMenu.cloneCollection")')).toBeLessThan(block.indexOf('t("contextMenu.importData")')); + expect(block).toContain("openMongoImport"); + expect(block).toContain('t("contextMenu.exportData")'); + expect(block).toContain('exportMongoCollection("csv")'); + expect(block).toContain('exportMongoCollection("ndjson")'); + expect(block).not.toContain("openMongoExport"); + }); + + it("opens import through AppDialogs like table import, without an export setup dialog", () => { + const dialogs = read("../../layout/AppDialogs.vue"); + expect(dialogs).toContain("MongoImportDialog"); + expect(dialogs).not.toContain("MongoExportDialog"); + expect(dialogs).toContain("dialogs.showMongoImportDialog.value"); + + const sources = read("../../../composables/useDialogSources.ts"); + expect(sources).toContain("connectionStore.mongoImportSource"); + expect(sources).not.toContain("connectionStore.mongoExportSource"); + }); +}); diff --git a/apps/desktop/src/composables/__tests__/useDialogSources.mongo.spec.ts b/apps/desktop/src/composables/__tests__/useDialogSources.mongo.spec.ts new file mode 100644 index 000000000..63938ae74 --- /dev/null +++ b/apps/desktop/src/composables/__tests__/useDialogSources.mongo.spec.ts @@ -0,0 +1,64 @@ +// @vitest-environment happy-dom + +import { createApp, defineComponent, h, nextTick, reactive, type App } from "vue"; +import { afterAll, beforeAll, describe, expect, it, vi } from "vitest"; +import i18n from "@/i18n"; + +const store = reactive({ + connections: [], + mongoImportSource: null as { connectionId: string; database: string; collection: string } | null, + transferSource: null, + schemaDiffSource: null, + dataCompareSource: null, + sqlFileSource: null, + diagramSource: null, + docsSource: null, + tableImportSource: null, + tableDataGenerateSource: null, + fieldLineageSource: null, + databaseSearchSource: null, + databaseExportSource: null, +}); + +vi.mock("@/stores/connectionStore", () => ({ useConnectionStore: () => store })); +vi.mock("@/composables/useToast", () => ({ useToast: () => ({ toast: vi.fn() }) })); + +import { useDialogSources } from "@/composables/useDialogSources"; + +const mountedApps: App[] = []; +let dialogs: ReturnType; + +async function mountDialogs() { + const container = document.createElement("div"); + document.body.append(container); + const app = createApp( + defineComponent({ + setup() { + dialogs = useDialogSources(); + return () => h("div"); + }, + }), + ); + mountedApps.push(app); + app.use(i18n); + app.mount(container); + await nextTick(); +} + +describe("useDialogSources mongo import/export", () => { + beforeAll(async () => { + await mountDialogs(); + }); + + afterAll(() => { + for (const app of mountedApps.splice(0)) app.unmount(); + }); + + it("opens the mongo import dialog from a collection source", async () => { + store.mongoImportSource = { connectionId: "c1", database: "shop", collection: "orders" }; + await nextTick(); + expect(dialogs.showMongoImportDialog.value).toBe(true); + expect(dialogs.mongoImportPrefillCollection.value).toBe("orders"); + expect(store.mongoImportSource).toBeNull(); + }); +}); diff --git a/apps/desktop/src/composables/__tests__/useSidebarTreeExportRuntime.spec.ts b/apps/desktop/src/composables/__tests__/useSidebarTreeExportRuntime.spec.ts index adabd2991..a33e85387 100644 --- a/apps/desktop/src/composables/__tests__/useSidebarTreeExportRuntime.spec.ts +++ b/apps/desktop/src/composables/__tests__/useSidebarTreeExportRuntime.spec.ts @@ -16,6 +16,7 @@ const apiMock = vi.hoisted(() => ({ getColumns: vi.fn(), getTableDdl: vi.fn(), startTableExport: vi.fn(), + exportMongodbQuery: vi.fn(), })); vi.mock("@/lib/backend/api", () => apiMock); @@ -122,6 +123,41 @@ describe("useSidebarTreeExportRuntime", () => { expect(toastMock).toHaveBeenCalledWith("导出失败:上一个 DuckDB 查询仍在停止中,请稍后重试。", 5000); }); + it("exports a mongo collection through the save-file path without a setup dialog", async () => { + apiMock.exportMongodbQuery.mockImplementation(async (_request, onProgress) => { + onProgress({ exportId: "export-1", status: "done", documentsRead: 3, bytesWritten: 12, elapsedMs: 4 }); + return { exportId: "export-1", documentsExported: 3, filePath: "orders.ndjson", elapsedMs: 4 }; + }); + const activeNode = shallowRef({ id: "col-1", type: "mongo-collection", label: "orders", connectionId: "conn-1", database: "shop", children: [] } as TreeNode); + const connectionStore = { + ensureConnected: vi.fn(), + getConfig: vi.fn(() => ({ db_type: "mongodb" })), + treeNodes: [], + selectedTreeNodeIds: [], + }; + const runtime = useSidebarTreeExportRuntime({ + activeNode, + connectionStore: connectionStore as never, + settingsStore: exportSettings() as never, + acceptedSelectionIds: () => null, + }); + + await runtime.exportMongoCollection("ndjson"); + + expect(connectionStore.ensureConnected).toHaveBeenCalledWith("conn-1"); + expect(apiMock.exportMongodbQuery).toHaveBeenCalledWith( + expect.objectContaining({ + connectionId: "conn-1", + database: "shop", + collection: "orders", + format: "ndjson", + filePath: "orders.ndjson", + }), + expect.any(Function), + ); + expect(toastMock).toHaveBeenCalledWith("grid.exported"); + }); + it("loads and joins every selected DDL in tree order", async () => { apiMock.getTableDdl.mockResolvedValueOnce("CREATE TABLE one (id INT)").mockResolvedValueOnce("CREATE VIEW two AS SELECT 1;"); const first = { id: "table-1", type: "table", label: "one", connectionId: "conn-1", database: "db", schema: "main" } as TreeNode; diff --git a/apps/desktop/src/composables/__tests__/useSidebarTreeToolRuntime.docs.spec.ts b/apps/desktop/src/composables/__tests__/useSidebarTreeToolRuntime.docs.spec.ts index a19eda617..62b03d5bb 100644 --- a/apps/desktop/src/composables/__tests__/useSidebarTreeToolRuntime.docs.spec.ts +++ b/apps/desktop/src/composables/__tests__/useSidebarTreeToolRuntime.docs.spec.ts @@ -10,6 +10,7 @@ function setup(node: Partial, options: { treeNodes?: TreeNode[]; selec docsSource: null as unknown, diagramSource: null as unknown, databaseExportSource: null as unknown, + mongoImportSource: undefined as unknown, schemaDiffSource: null as unknown, treeNodes: options.treeNodes ?? [], selectedTreeNodeIds: options.selectedTreeNodeIds ?? [], @@ -109,6 +110,25 @@ describe("useSidebarTreeToolRuntime diagram and database export", () => { }); }); +describe("useSidebarTreeToolRuntime mongo import", () => { + it("opens collection import from a mongo-collection node", () => { + const { connectionStore, runtime } = setup({ + type: "mongo-collection", + label: "orders", + connectionId: "conn-1", + database: "shop", + }); + + runtime.openMongoImport(); + + expect(connectionStore.mongoImportSource).toEqual({ + connectionId: "conn-1", + database: "shop", + collection: "orders", + }); + }); +}); + describe("useSidebarTreeToolRuntime openSchemaDiffForRoutine", () => { it("prefills schema diff with a signature-aware routine key", () => { const { connectionStore, runtime } = setup({ diff --git a/apps/desktop/src/composables/useDialogSources.ts b/apps/desktop/src/composables/useDialogSources.ts index 22d8ad30f..0de72e98c 100644 --- a/apps/desktop/src/composables/useDialogSources.ts +++ b/apps/desktop/src/composables/useDialogSources.ts @@ -14,6 +14,7 @@ const showSqlFileDialog = ref(false); const showDiagramDialog = ref(false); const showDocsDialog = ref(false); const showTableImportDialog = ref(false); +const showMongoImportDialog = ref(false); const showTableDataGenerateDialog = ref(false); const showFieldLineageDialog = ref(false); const showDatabaseSearchDialog = ref(false); @@ -68,6 +69,9 @@ const tableImportPrefillConnectionId = ref(""); const tableImportPrefillDatabase = ref(""); const tableImportPrefillSchema = ref(""); const tableImportPrefillTable = ref(""); +const mongoImportPrefillConnectionId = ref(""); +const mongoImportPrefillDatabase = ref(""); +const mongoImportPrefillCollection = ref(""); const tableDataGeneratePrefillConnectionId = ref(""); const tableDataGeneratePrefillDatabase = ref(""); const tableDataGeneratePrefillSchema = ref(""); @@ -246,6 +250,19 @@ export function useDialogSources() { }, ); + watch( + () => connectionStore.mongoImportSource, + (v) => { + if (v) { + mongoImportPrefillConnectionId.value = v.connectionId; + mongoImportPrefillDatabase.value = v.database; + mongoImportPrefillCollection.value = v.collection; + showMongoImportDialog.value = true; + connectionStore.mongoImportSource = null; + } + }, + ); + watch( () => connectionStore.tableDataGenerateSource, (v) => { @@ -501,6 +518,7 @@ export function useDialogSources() { showDiagramDialog, showDocsDialog, showTableImportDialog, + showMongoImportDialog, showTableDataGenerateDialog, showFieldLineageDialog, showDatabaseSearchDialog, @@ -551,6 +569,9 @@ export function useDialogSources() { tableImportPrefillDatabase, tableImportPrefillSchema, tableImportPrefillTable, + mongoImportPrefillConnectionId, + mongoImportPrefillDatabase, + mongoImportPrefillCollection, tableDataGeneratePrefillConnectionId, tableDataGeneratePrefillDatabase, tableDataGeneratePrefillSchema, diff --git a/apps/desktop/src/composables/useSidebarTreeExportRuntime.ts b/apps/desktop/src/composables/useSidebarTreeExportRuntime.ts index 973a2d4bd..cc8ecd1de 100644 --- a/apps/desktop/src/composables/useSidebarTreeExportRuntime.ts +++ b/apps/desktop/src/composables/useSidebarTreeExportRuntime.ts @@ -270,23 +270,34 @@ export function useSidebarTreeExportRuntime(options: SidebarTreeExportRuntimeOpt return typeof selected === "string" ? selected : null; } - async function resolveTableExportOutputPath(target: SidebarTableExportTarget, format: string, outputDirectory?: string): Promise { - const fileName = `${target.fileNameBase ?? target.tableName}.${format}`; + function exportFilterName(format: string): string { + if (format === "csv") return "CSV"; + if (format === "json") return "JSON"; + if (format === "ndjson") return "NDJSON"; + if (format === "xlsx") return "Excel"; + return "SQL"; + } + + async function resolveExportOutputPath(fileNameBase: string, format: string, outputDirectory?: string): Promise { + const fileName = `${fileNameBase}.${format}`; if (outputDirectory !== undefined) { return outputDirectory ? joinExportFilePath(outputDirectory, fileName) : fileName; } if (isTauriRuntime()) { const { save } = await import("@tauri-apps/plugin-dialog"); - const filterName = format === "csv" ? "CSV" : format === "json" ? "JSON" : format === "xlsx" ? "Excel" : "SQL"; const path = await save({ defaultPath: fileName, - filters: [{ name: filterName, extensions: [format] }], + filters: [{ name: exportFilterName(format), extensions: [format] }], }); return path ? String(path) : null; } return fileName; } + async function resolveTableExportOutputPath(target: SidebarTableExportTarget, format: string, outputDirectory?: string): Promise { + return resolveExportOutputPath(target.fileNameBase ?? target.tableName, format, outputDirectory); + } + async function exportDataLegacyForTarget(target: SidebarTableExportTarget, outputDirectory?: string, suppressDoneToast = false) { const { connectionId, database } = target; const config = connectionStore.getConfig(connectionId); @@ -461,6 +472,49 @@ export function useSidebarTreeExportRuntime(options: SidebarTreeExportRuntimeOpt return pickTableExportDirectory(); } + async function exportMongoCollection(format: "csv" | "ndjson") { + const node = activeNode.value; + if (node.type !== "mongo-collection" || !node.connectionId || !node.database) return; + const outputPath = await resolveExportOutputPath(node.label, format); + if (!outputPath) return; + let task: ExportTask | null = null; + try { + await connectionStore.ensureConnected(node.connectionId); + task = addExportTask(node.label, format, outputPath); + const currentTask = task; + await api.exportMongodbQuery( + { + exportId: currentTask.exportId, + connectionId: node.connectionId, + database: node.database, + collection: node.label, + format, + includeHeader: true, + filePath: outputPath, + }, + (progress) => { + currentTask.rowsExported = progress.documentsRead; + currentTask.totalRows = progress.totalDocuments ?? null; + if (progress.status === "running") currentTask.status = "Writing"; + else if (progress.status === "done") { + currentTask.status = "Done"; + currentTask.finishedAt = Date.now(); + } else if (progress.status === "error") { + currentTask.status = "Error"; + currentTask.errorMessage = progress.errorMessage ?? null; + } else if (progress.status === "cancelled") currentTask.status = "Cancelled"; + }, + ); + toast(t("grid.exported")); + } catch (error: unknown) { + if (task) { + task.status = "Error"; + task.errorMessage = error instanceof Error ? error.message : String(error); + } + toast(t("grid.exportFailed", { message: translateBackendError(t, error) }), 5000); + } + } + async function exportData(format: "csv" | "json" | "sql") { const targets = currentTableExportTargets(); if (!targets.length) return; @@ -531,6 +585,7 @@ export function useSidebarTreeExportRuntime(options: SidebarTreeExportRuntimeOpt copyStructurePreview, exportData, exportDataXlsx, + exportMongoCollection, exportStructure, saveStructurePreview, selectTextareaContent, diff --git a/apps/desktop/src/composables/useSidebarTreeToolRuntime.ts b/apps/desktop/src/composables/useSidebarTreeToolRuntime.ts index eb0d4a1ff..000e5edde 100644 --- a/apps/desktop/src/composables/useSidebarTreeToolRuntime.ts +++ b/apps/desktop/src/composables/useSidebarTreeToolRuntime.ts @@ -139,6 +139,16 @@ export function useSidebarTreeToolRuntime(options: SidebarTreeToolRuntimeOptions }; } + function openMongoImport() { + const node = activeNode.value; + if (!node.connectionId || !node.database || node.type !== "mongo-collection") return; + connectionStore.mongoImportSource = { + connectionId: node.connectionId, + database: node.database, + collection: node.label, + }; + } + function openStructureEditor() { const node = activeNode.value; if (!node.connectionId || !node.database) return; @@ -180,6 +190,7 @@ export function useSidebarTreeToolRuntime(options: SidebarTreeToolRuntimeOptions openDiagram, openDocs, openFieldLineage, + openMongoImport, openScheduledBackups, openSchemaDiff, openSchemaDiffForRoutine, diff --git a/apps/desktop/src/i18n/locales/az.ts b/apps/desktop/src/i18n/locales/az.ts index 8e737aef7..449ea2f85 100644 --- a/apps/desktop/src/i18n/locales/az.ts +++ b/apps/desktop/src/i18n/locales/az.ts @@ -5353,6 +5353,46 @@ export default withEnglishFallback({ tableView: "Cədvəl görünüşü", filterPlaceholder: "Süzgəc...", sortPlaceholder: "Sıralama...", + import: { + title: "Məlumat idxal et", + target: "Hədəf kolleksiya", + readonly: "Bu bağlantı yalnız oxuma üçündür. İdxaldan əvvəl yazmanı açın.", + chooseFile: "Fayl seçin", + fileFilter: "MongoDB məlumatı", + noFile: "Fayl seçilməyib", + back: "Geri", + format: "Format", + encoding: "Kodlama", + delimiter: "Ayırıcı", + typeMode: "Tip rejimi", + typeString: "Hamısı sətir", + typeAuto: "Avtomatik aşkarlama", + typeExtendedJson: "Extended JSON", + hasHeader: "Birinci sətir başlıqdır", + trim: "Dəyərləri kəs", + emptyAsNull: "Boş xanaları null yaz", + recognizeObjectIdHex: "24 simvollu onaltılığı ObjectId say", + skipErrorRows: "Xətalı sətirləri keç", + previewing: "Önizləmə yenilənir…", + estimatedRows: "Təxminən {count} sənəd", + continue: "Davam et", + confirmAppend: "Sənədlər əlavə olunacaq. Mövcud sənədlər silinməz və əvəz olunmaz.", + batchSize: "Paket ölçüsü", + duplicateIdPolicy: "Təkrar _id idxalı dayandırır. Artıq göndərilmiş paketlər geri qaytarılmır.", + start: "İdxal et", + success: "{count} sənəd idxal olundu", + stillRunning: "İdxal hələ işləyir. Yeni paketləri dayandırmaq üçün Ləğv et düyməsini istifadə edin.", + rowsRead: "Oxunan sətirlər", + rowsInserted: "Əlavə olunan sətirlər", + rowsFailed: "Uğursuz sətirlər", + batchesCommitted: "Göndərilmiş paketlər", + phase: { + preparing: "Hazırlanır", + parsing: "Ayrışdırılır", + writing: "Yazılır", + done: "Bitdi", + }, + }, }, meilisearch: { ...meilisearchManagementAz, @@ -5652,6 +5692,7 @@ export default withEnglishFallback({ lockNow: "Yazmanı indi kilidlə", unlockAction: "Yazma kilidini aç…", sourceImport: "Cədvəl idxalı", + sourceMongoImport: "MongoDB idxalı", sourceTransfer: "Məlumatların köçürülməsi", sourceSqlFile: "SQL faylının icrası", sourceStatus: "Yalnız oxuma rejiminin idarəsi", diff --git a/apps/desktop/src/i18n/locales/en.ts b/apps/desktop/src/i18n/locales/en.ts index e48d0a1de..34ce14dad 100644 --- a/apps/desktop/src/i18n/locales/en.ts +++ b/apps/desktop/src/i18n/locales/en.ts @@ -5353,6 +5353,46 @@ export default { tableView: "Table View", filterPlaceholder: "Filter...", sortPlaceholder: "Sort...", + import: { + title: "Import data", + target: "Target collection", + readonly: "This connection is read-only. Unlock writes before importing.", + chooseFile: "Choose file", + fileFilter: "MongoDB data", + noFile: "No file selected", + back: "Back", + format: "Format", + encoding: "Encoding", + delimiter: "Delimiter", + typeMode: "Type mode", + typeString: "All strings", + typeAuto: "Auto detect", + typeExtendedJson: "Extended JSON", + hasHeader: "First row is header", + trim: "Trim values", + emptyAsNull: "Empty cells as null", + recognizeObjectIdHex: "Treat 24-character hex as ObjectId", + skipErrorRows: "Skip error rows", + previewing: "Refreshing preview…", + estimatedRows: "{count} documents estimated", + continue: "Continue", + confirmAppend: "Documents will be appended. Existing documents are not truncated or upserted.", + batchSize: "Batch size", + duplicateIdPolicy: "Duplicate _id values stop the import. Already committed batches are not rolled back.", + start: "Import", + success: "Imported {count} documents", + stillRunning: "Import is still running. Use Cancel to stop new batches.", + rowsRead: "Rows read", + rowsInserted: "Rows inserted", + rowsFailed: "Rows failed", + batchesCommitted: "Batches committed", + phase: { + preparing: "Preparing", + parsing: "Parsing", + writing: "Writing", + done: "Done", + }, + }, }, meilisearch: { ...meilisearchManagementEn, @@ -5652,6 +5692,7 @@ export default { lockNow: "Lock writes now", unlockAction: "Unlock writes…", sourceImport: "Table import", + sourceMongoImport: "MongoDB import", sourceTransfer: "Data transfer", sourceSqlFile: "SQL file execution", sourceStatus: "Read-only control", diff --git a/apps/desktop/src/i18n/locales/es.ts b/apps/desktop/src/i18n/locales/es.ts index ae6d36076..1e4930388 100644 --- a/apps/desktop/src/i18n/locales/es.ts +++ b/apps/desktop/src/i18n/locales/es.ts @@ -5064,6 +5064,46 @@ export default withEnglishFallback({ indexNoExpiry: "Nunca expira", indexOtherOptions: "Otras opciones", indexPropertiesUnavailable: "El controlador Legacy de MongoDB no puede informar disperso, expiración, background o tamaño de bucket. Conéctese con el controlador nativo para verlos.", + import: { + title: "Importar datos", + target: "Colección de destino", + readonly: "Esta conexión es de solo lectura. Desbloquee la escritura antes de importar.", + chooseFile: "Elegir archivo", + fileFilter: "Datos de MongoDB", + noFile: "Ningún archivo seleccionado", + back: "Atrás", + format: "Formato", + encoding: "Codificación", + delimiter: "Delimitador", + typeMode: "Modo de tipos", + typeString: "Todo como texto", + typeAuto: "Detectar automáticamente", + typeExtendedJson: "Extended JSON", + hasHeader: "La primera fila es el encabezado", + trim: "Recortar valores", + emptyAsNull: "Celdas vacías como null", + recognizeObjectIdHex: "Tratar hexadecimal de 24 caracteres como ObjectId", + skipErrorRows: "Omitir filas con error", + previewing: "Actualizando la vista previa…", + estimatedRows: "{count} documentos estimados", + continue: "Continuar", + confirmAppend: "Los documentos se añadirán. No se truncará ni se hará upsert de los existentes.", + batchSize: "Tamaño de lote", + duplicateIdPolicy: "Un _id duplicado detiene la importación. Los lotes ya confirmados no se revierten.", + start: "Importar", + success: "Se importaron {count} documentos", + stillRunning: "La importación sigue en curso. Use Cancelar para detener nuevos lotes.", + rowsRead: "Filas leídas", + rowsInserted: "Filas insertadas", + rowsFailed: "Filas fallidas", + batchesCommitted: "Lotes confirmados", + phase: { + preparing: "Preparando", + parsing: "Analizando", + writing: "Escribiendo", + done: "Hecho", + }, + }, }, meilisearch: { ...meilisearchManagementEs, @@ -8927,6 +8967,7 @@ export default withEnglishFallback({ lockNow: "Volver a bloquear ahora", unlockAction: "Desbloquear escrituras…", sourceImport: "Importación de tabla", + sourceMongoImport: "Importación de MongoDB", sourceTransfer: "Transferencia de datos", sourceSqlFile: "Ejecución de archivo SQL", sourceStatus: "Control de solo lectura", diff --git a/apps/desktop/src/i18n/locales/it.ts b/apps/desktop/src/i18n/locales/it.ts index 0016e18d3..11b2acf9f 100644 --- a/apps/desktop/src/i18n/locales/it.ts +++ b/apps/desktop/src/i18n/locales/it.ts @@ -5062,6 +5062,46 @@ export default withEnglishFallback({ indexNoExpiry: "Mai", indexOtherOptions: "Altre opzioni", indexPropertiesUnavailable: "Il driver Legacy di MongoDB non può riportare sparsità, scadenza, background o dimensione bucket. Connettersi con il driver nativo per vederli.", + import: { + title: "Importa dati", + target: "Collezione di destinazione", + readonly: "Questa connessione è in sola lettura. Sblocca la scrittura prima di importare.", + chooseFile: "Scegli file", + fileFilter: "Dati MongoDB", + noFile: "Nessun file selezionato", + back: "Indietro", + format: "Formato", + encoding: "Codifica", + delimiter: "Delimitatore", + typeMode: "Modalità tipi", + typeString: "Tutto come stringa", + typeAuto: "Rilevamento automatico", + typeExtendedJson: "Extended JSON", + hasHeader: "La prima riga è l'intestazione", + trim: "Taglia gli spazi", + emptyAsNull: "Celle vuote come null", + recognizeObjectIdHex: "Tratta l'esadecimale a 24 caratteri come ObjectId", + skipErrorRows: "Salta le righe con errore", + previewing: "Aggiornamento anteprima…", + estimatedRows: "{count} documenti stimati", + continue: "Continua", + confirmAppend: "I documenti verranno aggiunti. Nessun truncate né upsert.", + batchSize: "Dimensione batch", + duplicateIdPolicy: "Un _id duplicato interrompe l'importazione. I batch già confermati non vengono annullati.", + start: "Importa", + success: "Importati {count} documenti", + stillRunning: "L'importazione è ancora in esecuzione. Usa Annulla per fermare i nuovi batch.", + rowsRead: "Righe lette", + rowsInserted: "Righe inserite", + rowsFailed: "Righe non riuscite", + batchesCommitted: "Batch confermati", + phase: { + preparing: "Preparazione", + parsing: "Analisi", + writing: "Scrittura", + done: "Completato", + }, + }, }, meilisearch: { ...meilisearchManagementIt, @@ -8928,6 +8968,7 @@ export default withEnglishFallback({ lockNow: "Blocca scritture ora", unlockAction: "Sblocca scritture…", sourceImport: "Importazione tabella", + sourceMongoImport: "Importazione MongoDB", sourceTransfer: "Trasferimento dati", sourceSqlFile: "Esecuzione file SQL", sourceStatus: "Controllo sola lettura", diff --git a/apps/desktop/src/i18n/locales/ja.ts b/apps/desktop/src/i18n/locales/ja.ts index 925a06394..f2fdfe007 100644 --- a/apps/desktop/src/i18n/locales/ja.ts +++ b/apps/desktop/src/i18n/locales/ja.ts @@ -5091,6 +5091,46 @@ export default withEnglishFallback({ indexNoExpiry: "期限なし", indexOtherOptions: "その他のオプション", indexPropertiesUnavailable: "MongoDB Legacy ドライバーはスパース、有効期限、バックグラウンド、バケットサイズを報告できません。ネイティブドライバーで接続して確認してください。", + import: { + title: "データをインポート", + target: "対象コレクション", + readonly: "この接続は読み取り専用です。インポート前に書き込みを解除してください。", + chooseFile: "ファイルを選択", + fileFilter: "MongoDB データ", + noFile: "ファイル未選択", + back: "戻る", + format: "形式", + encoding: "エンコーディング", + delimiter: "区切り文字", + typeMode: "型モード", + typeString: "すべて文字列", + typeAuto: "自動検出", + typeExtendedJson: "Extended JSON", + hasHeader: "先頭行をヘッダーにする", + trim: "値をトリム", + emptyAsNull: "空セルを null にする", + recognizeObjectIdHex: "24 桁の 16 進数を ObjectId として扱う", + skipErrorRows: "エラー行をスキップ", + previewing: "プレビューを更新中…", + estimatedRows: "推定 {count} 件", + continue: "続行", + confirmAppend: "ドキュメントは追加されます。既存ドキュメントの切り捨てや upsert は行いません。", + batchSize: "バッチサイズ", + duplicateIdPolicy: "重複した _id でインポートは停止します。すでに確定したバッチはロールバックされません。", + start: "インポート", + success: "{count} 件のドキュメントをインポートしました", + stillRunning: "インポートは実行中です。以降のバッチを止めるにはキャンセルしてください。", + rowsRead: "読み取り行数", + rowsInserted: "挿入行数", + rowsFailed: "失敗行数", + batchesCommitted: "確定バッチ数", + phase: { + preparing: "準備中", + parsing: "解析中", + writing: "書き込み中", + done: "完了", + }, + }, }, meilisearch: { ...meilisearchManagementJa, @@ -8979,6 +9019,7 @@ export default withEnglishFallback({ lockNow: "今すぐ再ロック", unlockAction: "書き込みのロックを解除…", sourceImport: "テーブルのインポート", + sourceMongoImport: "MongoDB のインポート", sourceTransfer: "データ転送", sourceSqlFile: "SQL ファイル実行", sourceStatus: "読み取り専用コントロール", diff --git a/apps/desktop/src/i18n/locales/ko.ts b/apps/desktop/src/i18n/locales/ko.ts index 55778215a..ada457d12 100644 --- a/apps/desktop/src/i18n/locales/ko.ts +++ b/apps/desktop/src/i18n/locales/ko.ts @@ -4690,6 +4690,46 @@ export default withEnglishFallback({ indexNoExpiry: "만료 안 함", indexOtherOptions: "기타 옵션", indexPropertiesUnavailable: "MongoDB Legacy 드라이버는 희소, 만료, 백그라운드 또는 버킷 크기를 보고할 수 없습니다. 네이티브 드라이버로 연결하여 확인하세요.", + import: { + title: "데이터 가져오기", + target: "대상 컬렉션", + readonly: "이 연결은 읽기 전용입니다. 가져오기 전에 쓰기를 잠금 해제하세요.", + chooseFile: "파일 선택", + fileFilter: "MongoDB 데이터", + noFile: "선택된 파일 없음", + back: "뒤로", + format: "형식", + encoding: "인코딩", + delimiter: "구분 기호", + typeMode: "유형 모드", + typeString: "모두 문자열", + typeAuto: "자동 감지", + typeExtendedJson: "Extended JSON", + hasHeader: "첫 행을 헤더로 사용", + trim: "값 앞뒤 공백 제거", + emptyAsNull: "빈 셀을 null로 기록", + recognizeObjectIdHex: "24자 16진수를 ObjectId로 인식", + skipErrorRows: "오류 행 건너뛰기", + previewing: "미리보기 새로 고치는 중…", + estimatedRows: "약 {count}개 문서", + continue: "계속", + confirmAppend: "문서는 추가됩니다. 기존 문서는 비우거나 upsert하지 않습니다.", + batchSize: "배치 크기", + duplicateIdPolicy: "중복 _id는 가져오기를 중지합니다. 이미 커밋된 배치는 롤백되지 않습니다.", + start: "가져오기", + success: "{count}개 문서를 가져왔습니다", + stillRunning: "가져오기가 계속 실행 중입니다. 이후 배치를 멈추려면 취소를 사용하세요.", + rowsRead: "읽은 행", + rowsInserted: "삽입된 행", + rowsFailed: "실패한 행", + batchesCommitted: "커밋된 배치", + phase: { + preparing: "준비 중", + parsing: "구문 분석 중", + writing: "쓰는 중", + done: "완료", + }, + }, }, meilisearch: { ...meilisearchManagementKo, @@ -4988,6 +5028,7 @@ export default withEnglishFallback({ lockNow: "지금 다시 잠금", unlockAction: "쓰기 잠금 해제…", sourceImport: "테이블 가져오기", + sourceMongoImport: "MongoDB 가져오기", sourceTransfer: "데이터 이전", sourceSqlFile: "SQL 파일 실행", sourceStatus: "읽기 전용 컨트롤", diff --git a/apps/desktop/src/i18n/locales/pt-BR.ts b/apps/desktop/src/i18n/locales/pt-BR.ts index 688b29360..728a632db 100644 --- a/apps/desktop/src/i18n/locales/pt-BR.ts +++ b/apps/desktop/src/i18n/locales/pt-BR.ts @@ -5064,6 +5064,46 @@ export default withEnglishFallback({ indexNoExpiry: "Nunca", indexOtherOptions: "Outras opções", indexPropertiesUnavailable: "O driver Legacy do MongoDB não pode informar esparso, expiração, background ou tamanho do bucket. Conecte-se com o driver nativo para vê-los.", + import: { + title: "Importar dados", + target: "Coleção de destino", + readonly: "Esta conexão é somente leitura. Desbloqueie a gravação antes de importar.", + chooseFile: "Escolher arquivo", + fileFilter: "Dados do MongoDB", + noFile: "Nenhum arquivo selecionado", + back: "Voltar", + format: "Formato", + encoding: "Codificação", + delimiter: "Delimitador", + typeMode: "Modo de tipos", + typeString: "Tudo como texto", + typeAuto: "Detectar automaticamente", + typeExtendedJson: "Extended JSON", + hasHeader: "A primeira linha é o cabeçalho", + trim: "Remover espaços", + emptyAsNull: "Células vazias como null", + recognizeObjectIdHex: "Tratar hexadecimal de 24 caracteres como ObjectId", + skipErrorRows: "Ignorar linhas com erro", + previewing: "Atualizando a prévia…", + estimatedRows: "{count} documentos estimados", + continue: "Continuar", + confirmAppend: "Os documentos serão acrescentados. Não há truncate nem upsert.", + batchSize: "Tamanho do lote", + duplicateIdPolicy: "Um _id duplicado interrompe a importação. Lotes já confirmados não são revertidos.", + start: "Importar", + success: "{count} documentos importados", + stillRunning: "A importação ainda está em execução. Use Cancelar para interromper novos lotes.", + rowsRead: "Linhas lidas", + rowsInserted: "Linhas inseridas", + rowsFailed: "Linhas com falha", + batchesCommitted: "Lotes confirmados", + phase: { + preparing: "Preparando", + parsing: "Analisando", + writing: "Gravando", + done: "Concluído", + }, + }, }, meilisearch: { ...meilisearchManagementPtBR, @@ -8929,6 +8969,7 @@ export default withEnglishFallback({ lockNow: "Bloquear gravações agora", unlockAction: "Desbloquear gravações…", sourceImport: "Importação de tabela", + sourceMongoImport: "Importação do MongoDB", sourceTransfer: "Transferência de dados", sourceSqlFile: "Execução de arquivo SQL", sourceStatus: "Controle de somente leitura", diff --git a/apps/desktop/src/i18n/locales/tr.ts b/apps/desktop/src/i18n/locales/tr.ts index 27b78e131..785743e6e 100644 --- a/apps/desktop/src/i18n/locales/tr.ts +++ b/apps/desktop/src/i18n/locales/tr.ts @@ -5247,6 +5247,46 @@ export default withEnglishFallback({ tableView: "Tablo Görünümü", filterPlaceholder: "Filtrele...", sortPlaceholder: "Sırala...", + import: { + title: "Veri içe aktar", + target: "Hedef koleksiyon", + readonly: "Bu bağlantı salt okunur. İçe aktarmadan önce yazmayı açın.", + chooseFile: "Dosya seç", + fileFilter: "MongoDB verisi", + noFile: "Dosya seçilmedi", + back: "Geri", + format: "Biçim", + encoding: "Kodlama", + delimiter: "Ayırıcı", + typeMode: "Tür kipi", + typeString: "Tümü metin", + typeAuto: "Otomatik algıla", + typeExtendedJson: "Extended JSON", + hasHeader: "İlk satır başlıktır", + trim: "Değerleri kırp", + emptyAsNull: "Boş hücreleri null yaz", + recognizeObjectIdHex: "24 karakterlik onaltılığı ObjectId say", + skipErrorRows: "Hatalı satırları atla", + previewing: "Önizleme yenileniyor…", + estimatedRows: "Yaklaşık {count} belge", + continue: "Devam", + confirmAppend: "Belgeler eklenecek. Mevcut belgeler silinmez veya üzerine yazılmaz.", + batchSize: "Toplu boyut", + duplicateIdPolicy: "Yinelenen _id içe aktarmayı durdurur. Gönderilmiş toplu işler geri alınmaz.", + start: "İçe aktar", + success: "{count} belge içe aktarıldı", + stillRunning: "İçe aktarma hâlâ çalışıyor. Yeni toplu işleri durdurmak için İptal’i kullanın.", + rowsRead: "Okunan satırlar", + rowsInserted: "Eklenen satırlar", + rowsFailed: "Başarısız satırlar", + batchesCommitted: "Gönderilen toplu işler", + phase: { + preparing: "Hazırlanıyor", + parsing: "Ayrıştırılıyor", + writing: "Yazılıyor", + done: "Bitti", + }, + }, }, meilisearch: { ...meilisearchManagementTr, @@ -5546,6 +5586,7 @@ export default withEnglishFallback({ lockNow: "Yazmaları şimdi kilitle", unlockAction: "Yazmaların kilidini aç…", sourceImport: "Tablo içe aktarma", + sourceMongoImport: "MongoDB içe aktarma", sourceTransfer: "Veri aktarımı", sourceSqlFile: "SQL dosyası yürütme", sourceStatus: "Salt okunur denetimi", diff --git a/apps/desktop/src/i18n/locales/zh-CN.ts b/apps/desktop/src/i18n/locales/zh-CN.ts index 34ed57932..00793da29 100644 --- a/apps/desktop/src/i18n/locales/zh-CN.ts +++ b/apps/desktop/src/i18n/locales/zh-CN.ts @@ -5337,6 +5337,46 @@ export default withEnglishFallback({ tableView: "表格视图", filterPlaceholder: "过滤条件...", sortPlaceholder: "排序条件...", + import: { + title: "导入数据", + target: "目标 Collection", + readonly: "当前连接为只读,导入前需要先解锁写入。", + chooseFile: "选择文件", + fileFilter: "MongoDB 数据", + noFile: "尚未选择文件", + back: "上一步", + format: "格式", + encoding: "编码", + delimiter: "分隔符", + typeMode: "类型策略", + typeString: "全部按字符串", + typeAuto: "自动推断", + typeExtendedJson: "Extended JSON", + hasHeader: "首行是表头", + trim: "去除首尾空白", + emptyAsNull: "空单元格写入 null", + recognizeObjectIdHex: "将 24 位十六进制识别为 ObjectId", + skipErrorRows: "跳过错误行", + previewing: "正在刷新预览…", + estimatedRows: "估计 {count} 个文档", + continue: "继续", + confirmAppend: "将以追加方式写入,不会清空或覆盖已有文档。", + batchSize: "批次大小", + duplicateIdPolicy: "重复 _id 会停止导入。已提交的批次不会回滚。", + start: "开始导入", + success: "已导入 {count} 个文档", + stillRunning: "导入仍在运行。如需停止后续批次,请点取消。", + rowsRead: "已读取", + rowsInserted: "已插入", + rowsFailed: "失败", + batchesCommitted: "已提交批次", + phase: { + preparing: "准备中", + parsing: "解析中", + writing: "写入中", + done: "完成", + }, + }, }, meilisearch: { ...meilisearchManagementZhCN, @@ -5636,6 +5676,7 @@ export default withEnglishFallback({ lockNow: "立即重新锁定", unlockAction: "解锁写入…", sourceImport: "表导入", + sourceMongoImport: "MongoDB 导入", sourceTransfer: "数据传输", sourceSqlFile: "SQL 文件执行", sourceStatus: "只读控制", diff --git a/apps/desktop/src/i18n/locales/zh-TW.ts b/apps/desktop/src/i18n/locales/zh-TW.ts index bec0e62bb..4ab6665f4 100644 --- a/apps/desktop/src/i18n/locales/zh-TW.ts +++ b/apps/desktop/src/i18n/locales/zh-TW.ts @@ -4379,6 +4379,46 @@ export default withEnglishFallback({ tableView: "表格檢視", filterPlaceholder: "過濾條件……", sortPlaceholder: "排序條件……", + import: { + title: "匯入資料", + target: "目標 Collection", + readonly: "目前連線為唯讀,匯入前需要先解鎖寫入。", + chooseFile: "選擇檔案", + fileFilter: "MongoDB 資料", + noFile: "尚未選擇檔案", + back: "上一步", + format: "格式", + encoding: "編碼", + delimiter: "分隔符", + typeMode: "類型策略", + typeString: "全部按字串", + typeAuto: "自動推斷", + typeExtendedJson: "Extended JSON", + hasHeader: "首列是表頭", + trim: "去除首尾空白", + emptyAsNull: "空儲存格寫入 null", + recognizeObjectIdHex: "將 24 位十六進位識別為 ObjectId", + skipErrorRows: "跳過錯誤列", + previewing: "正在重新整理預覽…", + estimatedRows: "估計 {count} 個文件", + continue: "繼續", + confirmAppend: "將以附加方式寫入,不會清空或覆蓋既有文件。", + batchSize: "批次大小", + duplicateIdPolicy: "重複 _id 會停止匯入。已提交的批次不會回溯。", + start: "開始匯入", + success: "已匯入 {count} 個文件", + stillRunning: "匯入仍在執行。如需停止後續批次,請點取消。", + rowsRead: "已讀取", + rowsInserted: "已插入", + rowsFailed: "失敗", + batchesCommitted: "已提交批次", + phase: { + preparing: "準備中", + parsing: "解析中", + writing: "寫入中", + done: "完成", + }, + }, fieldEditor: "欄位", nativeJson: "原生 JSON", nativeJsonHint: "BSON 值請使用 MongoDB Extended JSON,例如 $oid、$date、$numberLong。", @@ -8917,6 +8957,7 @@ export default withEnglishFallback({ lockNow: "立即重新鎖定", unlockAction: "解鎖寫入…", sourceImport: "資料表匯入", + sourceMongoImport: "MongoDB 匯入", sourceTransfer: "資料傳輸", sourceSqlFile: "SQL 檔案執行", sourceStatus: "唯讀控制", diff --git a/apps/desktop/src/lib/__tests__/import/importSource.spec.ts b/apps/desktop/src/lib/__tests__/import/importSource.spec.ts new file mode 100644 index 000000000..d36d4be55 --- /dev/null +++ b/apps/desktop/src/lib/__tests__/import/importSource.spec.ts @@ -0,0 +1,24 @@ +import { describe, expect, it } from "vitest"; +import { importPreviewInput, importSourceDisplayName, importTextDelimiterForName, uploadedImportSourceFromPreview } from "@/lib/import/importSource"; + +describe("importSource", () => { + it("reuses an uploaded sourceRef and server path on later previews", () => { + const file = new File(["id\n1"], "orders.csv"); + const first = importPreviewInput(null, file); + expect(first).toEqual({ fileOrPath: file, sourceRef: null }); + + const uploaded = uploadedImportSourceFromPreview({ sourceRef: "src-1", filePath: "/tmp/orders.csv" }); + expect(importPreviewInput(uploaded, file)).toEqual({ fileOrPath: "/tmp/orders.csv", sourceRef: "src-1" }); + }); + + it("keeps a local path preview on desktop when nothing has been uploaded", () => { + expect(importPreviewInput(null, "/data/orders.csv")).toEqual({ fileOrPath: "/data/orders.csv", sourceRef: null }); + }); + + it("matches table-import TSV delimiter encoding and source labels", () => { + expect(importTextDelimiterForName("orders.tsv")).toBe("\\t"); + expect(importTextDelimiterForName("orders.csv")).toBe(","); + expect(importSourceDisplayName("/tmp/dir/orders.csv")).toBe("orders.csv"); + expect(importSourceDisplayName(new File(["x"], "people.json"))).toBe("people.json"); + }); +}); diff --git a/apps/desktop/src/lib/backend/api.ts b/apps/desktop/src/lib/backend/api.ts index 603ef6e20..92ebeae9f 100644 --- a/apps/desktop/src/lib/backend/api.ts +++ b/apps/desktop/src/lib/backend/api.ts @@ -458,6 +458,12 @@ export const previewTableImportFile = forward("previewTableImportFile"); export const importTableFile = forward("importTableFile"); export const cancelTableImport = forward("cancelTableImport"); export const releaseTableImportSource = forward("releaseTableImportSource"); +export const previewMongodbImportFile = forward("previewMongodbImportFile"); +export const importMongodbFile = forward("importMongodbFile"); +export const cancelMongodbImport = forward("cancelMongodbImport"); +export const releaseMongodbImportSource = forward("releaseMongodbImportSource"); +export const exportMongodbQuery = forward("exportMongodbQuery"); +export const cancelMongodbExport = forward("cancelMongodbExport"); // Database Export export const beginDatabaseBackupSnapshot = forward("beginDatabaseBackupSnapshot"); @@ -995,6 +1001,19 @@ export type { TableImportRequest, TableImportSummary, TableImportProgress, + MongoImportFormat, + MongoImportTypeMode, + MongoImportIssue, + MongoImportParseOptions, + MongoImportPreviewRequest, + MongoImportPreview, + MongoImportRequest, + MongoImportProgress, + MongoImportSummary, + MongoExportFormat, + MongoExportRequest, + MongoExportProgress, + MongoExportSummary, DatabaseExportRequest, ExportProgress, TableExportProgress, diff --git a/apps/desktop/src/lib/backend/http.ts b/apps/desktop/src/lib/backend/http.ts index 4508dfa38..ab863990a 100644 --- a/apps/desktop/src/lib/backend/http.ts +++ b/apps/desktop/src/lib/backend/http.ts @@ -132,6 +132,14 @@ import type { TableImportRequest, TableImportSummary, TableImportProgress, + MongoImportPreviewRequest, + MongoImportPreview, + MongoImportRequest, + MongoImportProgress, + MongoImportSummary, + MongoExportRequest, + MongoExportProgress, + MongoExportSummary, DatabaseBackupSnapshot, DatabaseExportRequest, ExportProgress, @@ -2469,6 +2477,122 @@ export async function releaseTableImportSource(sourceRef: string): Promise = {}): Promise { + if (typeof fileOrPath === "object" && !(fileOrPath instanceof File)) { + throw new Error("previewMongodbImportFile in web mode requires a File object for upload previews"); + } + if (typeof fileOrPath === "string") { + if (!options.sourceRef) { + throw new Error("previewMongodbImportFile in web mode requires a File object for new uploads"); + } + const res = await fetch(apiUrl("/api/mongo/import/preview-source"), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ + sourceRef: options.sourceRef, + format: options.format, + parseOptions: options.parseOptions, + previewLimit: options.previewLimit, + }), + }); + if (!res.ok) throw await backendResponseError(res); + return res.json(); + } + const formData = new FormData(); + formData.append("file", fileOrPath); + if (options.format) formData.append("format", options.format); + if (options.parseOptions) formData.append("parseOptions", JSON.stringify(options.parseOptions)); + if (options.previewLimit != null) formData.append("previewLimit", String(options.previewLimit)); + const res = await fetch(apiUrl("/api/mongo/import/preview"), { + method: "POST", + body: formData, + }); + if (!res.ok) throw await backendResponseError(res); + return res.json(); +} + +export async function importMongodbFile(request: MongoImportRequest, onProgress: (progress: MongoImportProgress) => void): Promise { + const res = await fetch(apiUrl("/api/mongo/import/execute"), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ request }), + }); + if (!res.ok) throw await backendResponseError(res); + return new Promise((resolve, reject) => { + const es = new EventSource(apiUrl(`/api/mongo/import/progress/${request.importId}`)); + es.onmessage = (e) => { + const progress: MongoImportProgress = JSON.parse(e.data); + onProgress(progress); + if (progress.status === "done") { + es.close(); + resolve({ + importId: progress.importId, + rowsInserted: progress.rowsInserted, + rowsFailed: progress.rowsFailed, + batchesCommitted: progress.batchesCommitted, + elapsedMs: progress.elapsedMs, + }); + } else if (progress.status === "error" || progress.status === "cancelled") { + es.close(); + reject(new Error(progress.errorMessage || "MongoDB import failed")); + } + }; + es.onerror = () => { + es.close(); + reject(new Error("MongoDB import SSE connection failed")); + }; + }); +} + +export async function cancelMongodbImport(importId: string): Promise { + return post("/api/mongo/import/cancel", { importId }); +} + +export async function releaseMongodbImportSource(sourceRef: string): Promise { + const result = await post<{ released: boolean }>("/api/mongo/import/source/release", { sourceRef }); + return result.released; +} + +export async function exportMongodbQuery(request: MongoExportRequest, onProgress: (progress: MongoExportProgress) => void): Promise { + const res = await fetch(apiUrl("/api/mongo/export"), { + method: "POST", + headers: { "Content-Type": "application/json" }, + body: JSON.stringify({ request }), + }); + if (!res.ok) throw await backendResponseError(res); + return new Promise((resolve, reject) => { + const es = new EventSource(apiUrl(`/api/mongo/export/progress/${request.exportId}`)); + es.onmessage = (e) => { + const progress: MongoExportProgress = JSON.parse(e.data); + onProgress(progress); + if (progress.status === "done" || progress.status === "error" || progress.status === "cancelled") { + es.close(); + if (progress.status === "done") { + const a = document.createElement("a"); + a.href = apiUrl(`/api/mongo/export/download/${request.exportId}`); + a.click(); + resolve({ + exportId: progress.exportId, + documentsExported: progress.documentsRead, + filePath: request.filePath, + elapsedMs: progress.elapsedMs, + }); + } else { + reject(new Error(progress.errorMessage || "MongoDB export failed")); + } + } + }; + es.onerror = () => { + es.close(); + reject(new Error("MongoDB export SSE connection failed")); + }; + }); +} + +export async function cancelMongodbExport(exportId: string): Promise { + return post("/api/mongo/export/cancel", { exportId }); +} + // --------------------------------------------------------------------------- // Database Export // --------------------------------------------------------------------------- diff --git a/apps/desktop/src/lib/backend/tauri.ts b/apps/desktop/src/lib/backend/tauri.ts index e968c177a..1cf999794 100644 --- a/apps/desktop/src/lib/backend/tauri.ts +++ b/apps/desktop/src/lib/backend/tauri.ts @@ -4900,6 +4900,190 @@ export async function releaseTableImportSource(_sourceRef: string): Promise[]; + rowNumbers?: number[]; + warnings: MongoImportIssue[]; + errors: MongoImportIssue[]; + estimatedRows?: number | null; + estimatedRowsExact: boolean; +} + +export interface MongoImportRequest { + importId: string; + connectionId: string; + database: string; + collection: string; + filePath: string; + sourceRef?: string | null; + format: MongoImportFormat; + parseOptions?: MongoImportParseOptions; + batchSize: number; + executionId?: string | null; +} + +export interface MongoImportProgress { + importId: string; + phase: MongoImportPhase; + status: MongoImportStatus; + rowsRead: number; + rowsInserted: number; + rowsFailed: number; + batchesCommitted: number; + totalRows?: number | null; + errorRows?: MongoImportIssue[]; + errorMessage?: string | null; + elapsedMs: number; +} + +export interface MongoImportSummary { + importId: string; + rowsInserted: number; + rowsFailed: number; + batchesCommitted: number; + elapsedMs: number; +} + +export interface MongoExportRequest { + exportId: string; + connectionId: string; + database: string; + collection: string; + filter?: string | null; + sort?: string | null; + projection?: string | null; + collation?: string | null; + format: MongoExportFormat; + includeHeader?: boolean; + filePath: string; + executionId?: string | null; +} + +export interface MongoExportProgress { + exportId: string; + status: MongoExportStatus; + documentsRead: number; + bytesWritten: number; + totalDocuments?: number | null; + errorMessage?: string | null; + elapsedMs: number; +} + +export interface MongoExportSummary { + exportId: string; + documentsExported: number; + filePath: string; + elapsedMs: number; +} + +export async function previewMongodbImportFile(filePathOrRequest: string | File | MongoImportPreviewRequest, options: Partial = {}): Promise { + if (typeof filePathOrRequest !== "string" && !("filePath" in filePathOrRequest)) { + throw new Error("previewMongodbImportFile in desktop mode requires a file path, not a File object"); + } + const request: MongoImportPreviewRequest = typeof filePathOrRequest === "string" ? { format: options.format ?? "csv", ...options, filePath: filePathOrRequest } : filePathOrRequest; + return invoke("preview_mongodb_import_file", { request }); +} + +export async function importMongodbFile(request: MongoImportRequest, onProgress: (progress: MongoImportProgress) => void): Promise { + const unlisten: UnlistenFn = await listen("mongo-import-progress", (event) => { + if (event.payload.importId === request.importId) { + onProgress(event.payload); + if (event.payload.status === "done" || event.payload.status === "error" || event.payload.status === "cancelled") { + unlisten(); + } + } + }); + try { + const summary = await invoke("import_mongodb_file", { request }); + unlisten(); + return summary; + } catch (e) { + unlisten(); + throw e instanceof BackendErrorException ? e : new BackendErrorException(e); + } +} + +export async function cancelMongodbImport(importId: string): Promise { + return invoke("cancel_mongodb_import", { importId }); +} + +export async function releaseMongodbImportSource(_sourceRef: string): Promise { + return false; +} + +export async function exportMongodbQuery(request: MongoExportRequest, onProgress: (progress: MongoExportProgress) => void): Promise { + const unlisten: UnlistenFn = await listen("mongo-export-progress", (event) => { + if (event.payload.exportId === request.exportId) { + onProgress(event.payload); + if (event.payload.status === "done" || event.payload.status === "error" || event.payload.status === "cancelled") { + unlisten(); + } + } + }); + try { + const summary = await invoke("export_mongodb_query", { request }); + unlisten(); + return summary; + } catch (e) { + unlisten(); + throw e instanceof BackendErrorException ? e : new BackendErrorException(e); + } +} + +export async function cancelMongodbExport(exportId: string): Promise { + return invoke("cancel_mongodb_export", { exportId }); +} + // --- Database Export --- export interface DatabaseExportRequest { exportId: string; diff --git a/apps/desktop/src/lib/import/importSource.ts b/apps/desktop/src/lib/import/importSource.ts new file mode 100644 index 000000000..45176ff19 --- /dev/null +++ b/apps/desktop/src/lib/import/importSource.ts @@ -0,0 +1,25 @@ +export type ImportFileSource = string | File; + +export interface UploadedImportSource { + sourceRef: string; + filePath: string; +} + +export function importSourceDisplayName(source: ImportFileSource): string { + return typeof source === "string" ? source.split(/[\\/]/).pop() || source : source.name; +} + +export function importTextDelimiterForName(name: string): string { + return name.toLowerCase().endsWith(".tsv") ? "\\t" : ","; +} + +/** Reuse a server-side upload on later previews instead of sending the File again. */ +export function importPreviewInput(uploaded: UploadedImportSource | null | undefined, source: ImportFileSource): { fileOrPath: ImportFileSource; sourceRef: string | null } { + if (uploaded?.sourceRef) return { fileOrPath: uploaded.filePath, sourceRef: uploaded.sourceRef }; + return { fileOrPath: source, sourceRef: null }; +} + +export function uploadedImportSourceFromPreview(preview: { sourceRef?: string | null; filePath?: string } | null | undefined): UploadedImportSource | null { + if (!preview?.sourceRef || !preview.filePath) return null; + return { sourceRef: preview.sourceRef, filePath: preview.filePath }; +} diff --git a/apps/desktop/src/lib/table/tableImport.ts b/apps/desktop/src/lib/table/tableImport.ts index 6272574c9..c3bc518f4 100644 --- a/apps/desktop/src/lib/table/tableImport.ts +++ b/apps/desktop/src/lib/table/tableImport.ts @@ -35,6 +35,14 @@ export interface TableImportProgressLike { export type TableImportWizardStep = "source" | "options" | "mapping" | "review" | "execution"; +export const TABLE_IMPORT_ENCODING_OPTIONS: ReadonlyArray<{ value: TableImportTextEncoding; labelKey: string }> = [ + { value: "auto", labelKey: "tableImport.encodingAuto" }, + { value: "utf8", labelKey: "tableImport.encodingUtf8" }, + { value: "gbk", labelKey: "tableImport.encodingGbk" }, + { value: "utf16Le", labelKey: "tableImport.encodingUtf16Le" }, + { value: "utf16Be", labelKey: "tableImport.encodingUtf16Be" }, +]; + export const TABLE_IMPORT_WIZARD_STEPS: TableImportWizardStep[] = ["source", "options", "mapping", "review", "execution"]; export function formatTableImportElapsed(ms: number): string { diff --git a/apps/desktop/src/stores/connectionStore.ts b/apps/desktop/src/stores/connectionStore.ts index 2c62e5088..fc77e3e2b 100644 --- a/apps/desktop/src/stores/connectionStore.ts +++ b/apps/desktop/src/stores/connectionStore.ts @@ -552,6 +552,17 @@ export const useConnectionStore = defineStore("connection", () => { schema?: string; tableName?: string; } | null>(null); + const mongoImportSource = ref<{ + connectionId: string; + database: string; + collection: string; + } | null>(null); + const mongoImportCompleted = ref<{ + connectionId: string; + database: string; + collection: string; + at: number; + } | null>(null); const tableDataGenerateSource = ref<{ connectionId: string; database: string; @@ -9144,6 +9155,8 @@ export const useConnectionStore = defineStore("connection", () => { diagramSource, docsSource, tableImportSource, + mongoImportSource, + mongoImportCompleted, tableDataGenerateSource, fieldLineageSource, databaseSearchSource, diff --git a/crates/dbx-core/src/db/mongo_driver.rs b/crates/dbx-core/src/db/mongo_driver.rs index d6d6f1391..ef6d2a6c7 100644 --- a/crates/dbx-core/src/db/mongo_driver.rs +++ b/crates/dbx-core/src/db/mongo_driver.rs @@ -2504,13 +2504,139 @@ fn json_update_to_modifications(value: &serde_json::Value) -> Result Result { +pub fn json_object_to_document_extended_json(value: &serde_json::Value) -> Result { match Bson::try_from(value.clone()).map_err(|e| e.to_string())? { Bson::Document(doc) => Ok(doc), other => Err(format!("Expected a JSON object, got {other:?}")), } } +pub fn document_to_canonical_extended_json(document: &Document) -> serde_json::Value { + Bson::Document(document.clone()).into_canonical_extjson() +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct MongoBulkWriteError { + pub message: String, + pub index: Option, + pub code: Option, + pub retryable: bool, +} + +/// What actually happened to a submitted batch. A batch can partly succeed, so the count of +/// inserted documents and the per-document rejections are reported together. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct MongoInsertOutcome { + pub inserted: u64, + /// One entry per document the server rejected, `index` pointing into the submitted batch. + pub errors: Vec, +} + +pub async fn insert_bson_documents( + client: &Client, + database: &str, + collection: &str, + documents: Vec, +) -> Result { + if documents.is_empty() { + return Ok(MongoInsertOutcome::default()); + } + let total = documents.len() as u64; + let col = client.database(database).collection::(collection); + // Unordered: one rejected document must not abandon the rest of the batch, and the server + // then reports every rejection instead of stopping at the first. + match col.insert_many(documents).ordered(false).await { + Ok(result) => Ok(MongoInsertOutcome { inserted: result.inserted_ids.len() as u64, errors: Vec::new() }), + Err(error) => { + let errors = insert_write_errors(&error); + if errors.is_empty() { + // No per-document detail means the whole batch failed (network, auth, …). + return Err(map_insert_many_error(error)); + } + Ok(MongoInsertOutcome { inserted: total.saturating_sub(errors.len() as u64), errors }) + } + } +} + +/// Per-document rejections, sorted by batch index. Empty when the failure was not per-document. +fn insert_write_errors(error: &mongodb::error::Error) -> Vec { + use mongodb::error::{ErrorKind, WriteFailure}; + let mut errors = match error.kind.as_ref() { + ErrorKind::Write(WriteFailure::WriteError(write_error)) => { + vec![write_error_entry(None, write_error.code, &write_error.message)] + } + ErrorKind::InsertMany(failure) => failure + .write_errors + .iter() + .flatten() + .map(|error| write_error_entry(Some(error.index), error.code, &error.message)) + .collect(), + ErrorKind::BulkWrite(failure) => failure + .write_errors + .iter() + .map(|(index, error)| write_error_entry(Some(*index), error.code, &error.message)) + .collect(), + _ => Vec::new(), + }; + errors.sort_by_key(|error| error.index.unwrap_or(0)); + errors +} + +fn write_error_entry(index: Option, code: i32, message: &str) -> MongoBulkWriteError { + MongoBulkWriteError { + message: message.to_string(), + index, + code: Some(code), + retryable: is_retryable_mongo_write_code(code), + } +} + +fn map_insert_many_error(error: mongodb::error::Error) -> MongoBulkWriteError { + use mongodb::error::ErrorKind; + match error.kind.as_ref() { + ErrorKind::Io(_) | ErrorKind::ConnectionPoolCleared { .. } | ErrorKind::ServerSelection { .. } => { + MongoBulkWriteError { message: error.to_string(), index: None, code: None, retryable: true } + } + _ => MongoBulkWriteError { message: error.to_string(), index: None, code: None, retryable: false }, + } +} + +fn is_retryable_mongo_write_code(code: i32) -> bool { + !matches!(code, 11000 | 11001 | 12582) +} + +#[allow(clippy::too_many_arguments)] +pub async fn for_each_find_document( + client: &Client, + database: &str, + collection: &str, + filter: Option<&str>, + projection: Option<&str>, + sort: Option<&str>, + collation: Option<&str>, + batch_size: u32, + mut on_document: impl FnMut(Document) -> Result<(), String>, +) -> Result<(), String> { + let col = client.database(database).collection::(collection); + let filter_doc = parse_optional_filter_document(filter)?.unwrap_or_default(); + let mut find = col.find(filter_doc).batch_size(batch_size); + if let Some(projection) = parse_optional_json_document(projection, "projection")? { + find = find.projection(projection); + } + if let Some(sort) = parse_optional_json_document(sort, "sort")? { + find = find.sort(sort); + } + if let Some(collation) = parse_find_collation(collation)? { + find = find.collation(collation); + } + let mut cursor = find.await.map_err(|error| error.to_string())?; + while cursor.advance().await.map_err(|error| error.to_string())? { + let document = cursor.deserialize_current().map_err(|error| error.to_string())?; + on_document(document)?; + } + Ok(()) +} + fn json_object_to_document_preserving_existing( value: &serde_json::Value, existing: Option<&Document>, diff --git a/crates/dbx-core/src/lib.rs b/crates/dbx-core/src/lib.rs index f222c45a2..322452d21 100644 --- a/crates/dbx-core/src/lib.rs +++ b/crates/dbx-core/src/lib.rs @@ -58,6 +58,7 @@ pub mod models; pub mod mongo_oidc; pub mod mongo_ops; pub mod mongo_shell; +pub mod mongodb_import_export; #[cfg(feature = "mq-admin")] pub mod mq; #[cfg(feature = "mq-admin")] diff --git a/crates/dbx-core/src/mongodb_import_export.rs b/crates/dbx-core/src/mongodb_import_export.rs new file mode 100644 index 000000000..98e9e3b23 --- /dev/null +++ b/crates/dbx-core/src/mongodb_import_export.rs @@ -0,0 +1,3085 @@ +#![allow(clippy::result_large_err)] + +use std::collections::HashSet; +use std::fs::File; +use std::io::{BufRead, BufReader, BufWriter, Read, Write}; +use std::path::{Path, PathBuf}; +use std::time::Instant; + +use chrono::{DateTime as ChronoDateTime, NaiveDate, Utc}; +use mongodb::bson::{oid::ObjectId, Bson, DateTime, Decimal128, Document}; +use serde::{Deserialize, Serialize}; + +use crate::connection::{task_client_session_id, AppState, PoolKind}; +use crate::csv_export::{push_csv_field, CsvQuoteMode}; +use crate::db::agent_driver::AgentCapability; +use crate::db::mongo_driver::{ + self, document_to_canonical_extended_json, for_each_find_document, insert_bson_documents, + json_object_to_document_extended_json, MongoBulkWriteError, MongoInsertOutcome, +}; +use crate::table_import::{open_transcoded_text_file, TableImportTextEncoding}; + +pub const DEFAULT_PREVIEW_LIMIT: usize = 50; +pub const DEFAULT_BATCH_SIZE: usize = 500; +pub const MIN_BATCH_SIZE: usize = 100; +pub const MAX_BATCH_SIZE: usize = 5000; +/// Rows sampled for CSV type inference, independent of the preview window so that the +/// preview and the import always agree on column types. +pub const TYPE_SAMPLE_ROWS: usize = 1000; + +const TYPE_STRING: u8 = 1 << 0; +const TYPE_BOOLEAN: u8 = 1 << 1; +const TYPE_INTEGER: u8 = 1 << 2; +const TYPE_DECIMAL: u8 = 1 << 3; +const TYPE_DATE: u8 = 1 << 4; +const TYPE_OBJECT: u8 = 1 << 5; +const TYPE_ARRAY: u8 = 1 << 6; + +pub fn mongodb_import_client_session_id(import_id: &str) -> String { + task_client_session_id("mongo-import", import_id) +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoImportFormat { + Csv, + Json, + Ndjson, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoImportTypeMode { + String, + Auto, + ExtendedJson, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoImportInferredType { + Boolean, + Integer, + Decimal, + Date, + Object, + Array, + String, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoImportStatus { + Running, + Done, + Error, + Cancelled, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoImportPhase { + Preparing, + Parsing, + Writing, + Done, +} + +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportIssue { + pub code: String, + pub message: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub row: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub column: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub value: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub batch: Option, + #[serde(default)] + pub retryable: bool, +} + +impl MongoImportIssue { + pub fn new(code: &str, message: impl Into) -> Self { + Self { + code: code.to_string(), + message: message.into(), + row: None, + column: None, + value: None, + batch: None, + retryable: false, + } + } + + pub fn with_row(mut self, row: u64) -> Self { + self.row = Some(row); + self + } + + pub fn with_column(mut self, column: impl Into) -> Self { + self.column = Some(column.into()); + self + } + + pub fn with_value(mut self, value: impl Into) -> Self { + self.value = Some(value.into()); + self + } + + pub fn with_batch(mut self, batch: u64) -> Self { + self.batch = Some(batch); + self + } + + pub fn retryable(mut self) -> Self { + self.retryable = true; + self + } + + pub fn display_message(&self) -> String { + let mut message = self.message.clone(); + if let Some(row) = self.row { + message = format!("row {row}: {message}"); + } + if let Some(column) = &self.column { + message = format!("{message} (column {column})"); + } + if let Some(value) = &self.value { + message = format!("{message}; value={value}"); + } + message + } +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportParseOptions { + #[serde(default, skip_serializing_if = "Option::is_none")] + pub encoding: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub delimiter: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub has_header: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub trim: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub empty_as_null: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub type_mode: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub recognize_object_id_hex: Option, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub skip_error_rows: Option, +} + +impl Default for MongoImportParseOptions { + fn default() -> Self { + Self { + encoding: Some(TableImportTextEncoding::Auto), + delimiter: None, + has_header: Some(true), + trim: Some(false), + empty_as_null: Some(true), + type_mode: Some(MongoImportTypeMode::Auto), + recognize_object_id_hex: Some(false), + skip_error_rows: Some(false), + } + } +} + +impl MongoImportParseOptions { + fn encoding(&self) -> Option { + self.encoding.or(Some(TableImportTextEncoding::Auto)) + } + + fn has_header(&self) -> bool { + self.has_header.unwrap_or(true) + } + + fn trim(&self) -> bool { + self.trim.unwrap_or(false) + } + + fn empty_as_null(&self) -> bool { + self.empty_as_null.unwrap_or(true) + } + + fn type_mode(&self) -> MongoImportTypeMode { + self.type_mode.unwrap_or(MongoImportTypeMode::Auto) + } + + fn recognize_object_id_hex(&self) -> bool { + self.recognize_object_id_hex.unwrap_or(false) + } + + fn skip_error_rows(&self) -> bool { + self.skip_error_rows.unwrap_or(false) + } +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportColumn { + pub name: String, + pub inferred_type: MongoImportInferredType, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub sample_values: Vec, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportPreviewRequest { + pub file_path: String, + #[serde(default)] + pub source_ref: Option, + pub format: MongoImportFormat, + #[serde(default)] + pub parse_options: MongoImportParseOptions, + #[serde(default)] + pub preview_limit: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportPreview { + #[serde(skip_serializing_if = "Option::is_none")] + pub source_ref: Option, + pub format: MongoImportFormat, + #[serde(skip_serializing_if = "Option::is_none")] + pub detected_encoding: Option, + pub file_name: String, + pub file_path: String, + pub size_bytes: u64, + pub columns: Vec, + pub rows: Vec, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub row_numbers: Vec, + pub warnings: Vec, + pub errors: Vec, + #[serde(skip_serializing_if = "Option::is_none")] + pub estimated_rows: Option, + pub estimated_rows_exact: bool, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportRequest { + pub import_id: String, + pub connection_id: String, + pub database: String, + pub collection: String, + pub file_path: String, + #[serde(default)] + pub source_ref: Option, + pub format: MongoImportFormat, + #[serde(default)] + pub parse_options: MongoImportParseOptions, + pub batch_size: usize, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub execution_id: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportProgress { + pub import_id: String, + pub phase: MongoImportPhase, + pub status: MongoImportStatus, + pub rows_read: u64, + pub rows_inserted: u64, + pub rows_failed: u64, + pub batches_committed: u64, + #[serde(skip_serializing_if = "Option::is_none")] + pub total_rows: Option, + #[serde(default, skip_serializing_if = "Vec::is_empty")] + pub error_rows: Vec, + #[serde(skip_serializing_if = "Option::is_none")] + pub error_message: Option, + pub elapsed_ms: u128, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoImportSummary { + pub import_id: String, + pub rows_inserted: u64, + pub rows_failed: u64, + pub batches_committed: u64, + pub elapsed_ms: u128, +} + +#[derive(Debug, Clone)] +pub struct ParsedMongoDocument { + pub row: u64, + pub document: Document, + pub extended_json: serde_json::Value, +} + +#[derive(Debug, Clone)] +struct CsvParseConfig { + delimiter: u8, + has_header: bool, + trim: bool, + empty_as_null: bool, + type_mode: MongoImportTypeMode, + recognize_object_id_hex: bool, +} + +pub fn format_from_path(path: &str) -> Result { + let lower = path.to_lowercase(); + if lower.ends_with(".csv") || lower.ends_with(".tsv") || lower.ends_with(".txt") { + Ok(MongoImportFormat::Csv) + } else if lower.ends_with(".ndjson") || lower.ends_with(".jsonl") { + Ok(MongoImportFormat::Ndjson) + } else if lower.ends_with(".json") { + Ok(MongoImportFormat::Json) + } else { + Err("Unsupported MongoDB import file type; use .csv, .json, or .ndjson".to_string()) + } +} + +pub fn clamp_batch_size(batch_size: usize) -> Result { + if !(MIN_BATCH_SIZE..=MAX_BATCH_SIZE).contains(&batch_size) { + return Err(MongoImportIssue::new( + "INVALID_BATCH_SIZE", + format!("Batch size must be between {MIN_BATCH_SIZE} and {MAX_BATCH_SIZE}"), + )); + } + Ok(batch_size) +} + +pub fn validate_import_source_path(path: &str) -> Result<(), MongoImportIssue> { + let path = Path::new(path); + if !path.exists() { + return Err(MongoImportIssue::new("FILE_UNREADABLE", format!("Import file not found: {}", path.display()))); + } + if !path.is_file() { + return Err(MongoImportIssue::new( + "FILE_UNREADABLE", + format!("Import source must be a regular file: {}", path.display()), + )); + } + Ok(()) +} + +fn csv_config(path: &str, options: &MongoImportParseOptions) -> Result { + let default_delimiter = if path.to_lowercase().ends_with(".tsv") { b'\t' } else { b',' }; + let delimiter = match options.delimiter.as_deref() { + None | Some("") => default_delimiter, + Some("\\t") | Some("tab") | Some("TAB") => b'\t', + Some(";") => b';', + Some(value) => { + let bytes = value.as_bytes(); + if bytes.len() != 1 { + return Err(MongoImportIssue::new("CSV_STRUCTURE", "Delimiter must be a single-byte character")); + } + bytes[0] + } + }; + Ok(CsvParseConfig { + delimiter, + has_header: options.has_header(), + trim: options.trim(), + empty_as_null: options.empty_as_null(), + type_mode: options.type_mode(), + recognize_object_id_hex: options.recognize_object_id_hex(), + }) +} + +fn normalize_cell(value: &str, trim: bool) -> &str { + if trim { + value.trim() + } else { + value + } +} + +fn csv_cell_text(value: &str, config: &CsvParseConfig) -> Option { + let value = normalize_cell(value, config.trim).trim_start_matches('\u{feff}'); + if value.is_empty() { + None + } else { + Some(value.to_string()) + } +} + +fn unique_headers(headers: &[String]) -> Result, MongoImportIssue> { + let mut seen = HashSet::new(); + let mut names = Vec::with_capacity(headers.len()); + for (index, header) in headers.iter().enumerate() { + let name = header.trim().trim_start_matches('\u{feff}').to_string(); + if name.is_empty() { + return Err(MongoImportIssue::new("EMPTY_HEADER", format!("CSV header at column {} is empty", index + 1)) + .with_row(1)); + } + // Mongo field names are case-sensitive, so `Name` and `name` are two distinct columns. + if !seen.insert(name.clone()) { + return Err(MongoImportIssue::new("DUPLICATE_HEADER", format!("Duplicate CSV header: {name}")) + .with_row(1) + .with_column(name)); + } + names.push(name); + } + if names.is_empty() { + return Err(MongoImportIssue::new("CSV_STRUCTURE", "CSV file has no header columns").with_row(1)); + } + Ok(names) +} + +fn generated_field_names(count: usize) -> Vec { + (0..count).map(|index| format!("field_{}", index + 1)).collect() +} + +fn classify_cell(value: &str) -> u8 { + let mut bits = TYPE_STRING; + if is_boolean(value) { + bits |= TYPE_BOOLEAN; + } + if parse_integer(value).is_some() { + bits |= TYPE_INTEGER | TYPE_DECIMAL; + } else if parse_decimal(value).is_some() { + bits |= TYPE_DECIMAL; + } + if parse_date(value).is_some() { + bits |= TYPE_DATE; + } + if looks_like_json_object(value) { + bits |= TYPE_OBJECT; + } + if looks_like_json_array(value) { + bits |= TYPE_ARRAY; + } + bits +} + +fn is_boolean(value: &str) -> bool { + matches!(value, "true" | "TRUE" | "True" | "false" | "FALSE" | "False") +} + +fn parse_integer(value: &str) -> Option { + if value.is_empty() || value == "+" || value == "-" { + return None; + } + if value.contains('.') || value.contains('e') || value.contains('E') { + return None; + } + value.parse::().ok() +} + +fn parse_decimal(value: &str) -> Option { + value.parse::().ok() +} + +fn parse_date(value: &str) -> Option { + if let Ok(parsed) = DateTime::parse_rfc3339_str(value) { + return Some(parsed); + } + if let Ok(parsed) = ChronoDateTime::parse_from_rfc3339(value) { + return Some(DateTime::from_millis(parsed.with_timezone(&Utc).timestamp_millis())); + } + if let Ok(date) = NaiveDate::parse_from_str(value, "%Y-%m-%d") { + let datetime = date.and_hms_opt(0, 0, 0)?.and_utc(); + return Some(DateTime::from_millis(datetime.timestamp_millis())); + } + None +} + +fn looks_like_json_object(value: &str) -> bool { + let trimmed = value.trim(); + trimmed.starts_with('{') && trimmed.ends_with('}') +} + +fn looks_like_json_array(value: &str) -> bool { + let trimmed = value.trim(); + trimmed.starts_with('[') && trimmed.ends_with(']') +} + +fn csv_reader(reader: R, delimiter: u8) -> csv::Reader { + csv::ReaderBuilder::new().delimiter(delimiter).has_headers(false).flexible(true).from_reader(reader) +} + +fn is_object_id_hex(value: &str) -> bool { + value.len() == 24 && value.bytes().all(|byte| byte.is_ascii_hexdigit()) +} + +fn inferred_type_from_mask(mask: u8) -> MongoImportInferredType { + if mask & TYPE_BOOLEAN != 0 && mask & !TYPE_BOOLEAN & !TYPE_STRING == 0 { + return MongoImportInferredType::Boolean; + } + if mask & TYPE_INTEGER != 0 && mask & !(TYPE_INTEGER | TYPE_DECIMAL | TYPE_STRING) == 0 { + return MongoImportInferredType::Integer; + } + if mask & TYPE_DECIMAL != 0 && mask & !(TYPE_DECIMAL | TYPE_STRING) == 0 { + return MongoImportInferredType::Decimal; + } + if mask & TYPE_DATE != 0 && mask & !TYPE_DATE & !TYPE_STRING == 0 { + return MongoImportInferredType::Date; + } + if mask & TYPE_OBJECT != 0 && mask & !TYPE_OBJECT & !TYPE_STRING == 0 { + return MongoImportInferredType::Object; + } + if mask & TYPE_ARRAY != 0 && mask & !TYPE_ARRAY & !TYPE_STRING == 0 { + return MongoImportInferredType::Array; + } + MongoImportInferredType::String +} + +fn intersect_column_types(existing: Option, cell_mask: u8) -> u8 { + match existing { + None => cell_mask, + Some(existing) => existing & cell_mask, + } +} + +fn convert_cell( + value: Option<&str>, + column: &str, + row: u64, + inferred: MongoImportInferredType, + config: &CsvParseConfig, +) -> Result { + let Some(value) = value.filter(|value| !value.is_empty()) else { + return Ok(if config.empty_as_null { Bson::Null } else { Bson::String(String::new()) }); + }; + if (config.recognize_object_id_hex || (column == "_id" && config.type_mode != MongoImportTypeMode::String)) + && is_object_id_hex(value) + { + let oid = ObjectId::parse_str(value).map_err(|error| { + MongoImportIssue::new("TYPE_CONVERSION", format!("Invalid ObjectId: {error}")) + .with_row(row) + .with_column(column) + .with_value(value) + })?; + return Ok(Bson::ObjectId(oid)); + } + match config.type_mode { + MongoImportTypeMode::String => Ok(Bson::String(value.to_string())), + MongoImportTypeMode::Auto => convert_auto_cell(value, column, row, inferred), + MongoImportTypeMode::ExtendedJson => convert_extended_json_cell(value, column, row), + } +} + +fn convert_auto_cell( + value: &str, + column: &str, + row: u64, + inferred: MongoImportInferredType, +) -> Result { + let conversion_error = |message: String| { + MongoImportIssue::new("TYPE_CONVERSION", message).with_row(row).with_column(column).with_value(value) + }; + match inferred { + MongoImportInferredType::Boolean => { + if is_boolean(value) { + Ok(Bson::Boolean(value.eq_ignore_ascii_case("true"))) + } else { + Err(conversion_error(format!("Expected boolean, got {value}"))) + } + } + MongoImportInferredType::Integer => { + let number = + parse_integer(value).ok_or_else(|| conversion_error(format!("Expected integer, got {value}")))?; + if let Ok(value) = i32::try_from(number) { + Ok(Bson::Int32(value)) + } else { + Ok(Bson::Int64(number)) + } + } + MongoImportInferredType::Decimal => { + let decimal = + parse_decimal(value).ok_or_else(|| conversion_error(format!("Expected decimal, got {value}")))?; + Ok(Bson::Decimal128(decimal)) + } + MongoImportInferredType::Date => { + let date = parse_date(value).ok_or_else(|| conversion_error(format!("Expected date, got {value}")))?; + Ok(Bson::DateTime(date)) + } + MongoImportInferredType::Object | MongoImportInferredType::Array => { + parse_json_bson(value).map_err(conversion_error) + } + MongoImportInferredType::String => Ok(Bson::String(value.to_string())), + } +} + +fn convert_extended_json_cell(value: &str, column: &str, row: u64) -> Result { + let trimmed = value.trim(); + if trimmed.starts_with('{') + || trimmed.starts_with('[') + || trimmed.starts_with('"') + || trimmed == "true" + || trimmed == "false" + || trimmed == "null" + || looks_like_json_number(trimmed) + { + match serde_json::from_str::(trimmed) { + Ok(json) => Bson::try_from(json).map_err(|error| { + MongoImportIssue::new("TYPE_CONVERSION", format!("Invalid Extended JSON: {error}")) + .with_row(row) + .with_column(column) + .with_value(value) + }), + Err(_) => Ok(Bson::String(value.to_string())), + } + } else if let Some(date) = parse_date(trimmed) { + Ok(Bson::DateTime(date)) + } else { + Ok(Bson::String(value.to_string())) + } +} + +fn looks_like_json_number(value: &str) -> bool { + !value.is_empty() + && value.bytes().next().is_some_and(|byte| byte == b'-' || byte.is_ascii_digit()) + && serde_json::from_str::(value).is_ok() +} + +fn parse_json_bson(value: &str) -> Result { + let json: serde_json::Value = serde_json::from_str(value).map_err(|error| error.to_string())?; + Bson::try_from(json).map_err(|error| error.to_string()) +} + +fn document_from_csv_row( + row: u64, + headers: &[String], + fields: &[Option], + inferred: &[MongoImportInferredType], + config: &CsvParseConfig, + with_extended_json: bool, +) -> Result { + if fields.len() > headers.len() { + let extra = fields[headers.len()..].iter().flatten().cloned().collect::>().join(","); + return Err(MongoImportIssue::new("CSV_STRUCTURE", "CSV row has more fields than the header") + .with_row(row) + .with_value(extra)); + } + let mut document = Document::new(); + for (index, header) in headers.iter().enumerate() { + let value = fields.get(index).and_then(|value| value.as_deref()); + let inferred = inferred.get(index).copied().unwrap_or(MongoImportInferredType::String); + insert_field_path(&mut document, header, convert_cell(value, header, row, inferred, config)?); + } + Ok(parsed_document(row, document, with_extended_json)) +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum PathSegment<'a> { + Key(&'a str), + Index(usize), +} + +/// Splits a Compass CSV header into segments: `address.city`, `tags[0]`, `matrix[0][1]`. +/// A part whose brackets are not a well-formed trailing index stays a literal key, so +/// field names that merely contain brackets survive unchanged. +fn parse_field_path(path: &str) -> Vec> { + let mut segments = Vec::new(); + for part in path.split('.') { + match split_indexed_part(part) { + Some((name, indices)) => { + segments.push(PathSegment::Key(name)); + segments.extend(indices.into_iter().map(PathSegment::Index)); + } + None => segments.push(PathSegment::Key(part)), + } + } + segments +} + +fn split_indexed_part(part: &str) -> Option<(&str, Vec)> { + let (name, mut rest) = part.split_at(part.find('[')?); + if name.is_empty() { + return None; + } + let mut indices = Vec::new(); + while !rest.is_empty() { + let inner = rest.strip_prefix('[')?; + let close = inner.find(']')?; + indices.push(inner[..close].parse::().ok()?); + rest = &inner[close + 1..]; + } + Some((name, indices)) +} + +fn insert_field_path(document: &mut Document, path: &str, value: Bson) { + let segments = parse_field_path(path); + let Some((PathSegment::Key(key), rest)) = segments.split_first() else { + return; + }; + if rest.is_empty() { + document.insert(*key, value); + return; + } + if !document.contains_key(key) { + document.insert(*key, Bson::Null); + } + if let Some(slot) = document.get_mut(key) { + set_at_segments(slot, rest, value); + } +} + +/// Grows `target` into the container each segment requires, replacing any value that +/// conflicts with the shape the path asks for. +fn set_at_segments(target: &mut Bson, segments: &[PathSegment<'_>], value: Bson) { + let Some((segment, rest)) = segments.split_first() else { + *target = value; + return; + }; + match segment { + PathSegment::Key(key) => { + if !matches!(target, Bson::Document(_)) { + *target = Bson::Document(Document::new()); + } + let Bson::Document(document) = target else { return }; + if !document.contains_key(key) { + document.insert(*key, Bson::Null); + } + if let Some(slot) = document.get_mut(key) { + set_at_segments(slot, rest, value); + } + } + PathSegment::Index(index) => { + if !matches!(target, Bson::Array(_)) { + *target = Bson::Array(Vec::new()); + } + let Bson::Array(array) = target else { return }; + if array.len() <= *index { + array.resize(*index + 1, Bson::Null); + } + set_at_segments(&mut array[*index], rest, value); + } + } +} + +fn parsed_document(row: u64, document: Document, with_extended_json: bool) -> ParsedMongoDocument { + let extended_json = + if with_extended_json { document_to_canonical_extended_json(&document) } else { serde_json::Value::Null }; + ParsedMongoDocument { row, document, extended_json } +} + +/// Reads at most [`TYPE_SAMPLE_ROWS`] data rows to decide each column's type. Preview and +/// execution both call this with the same bound, so the types shown in the wizard are the +/// types the import actually writes. +fn infer_csv_types( + path: &str, + config: &CsvParseConfig, + encoding: Option, +) -> Result, MongoImportIssue> { + let auto = config.type_mode == MongoImportTypeMode::Auto; + let sample_rows = if auto { TYPE_SAMPLE_ROWS } else { 1 }; + let (reader, _) = open_transcoded_text_file(path, encoding).map_err(encoding_issue)?; + let mut csv_reader = csv_reader(reader, config.delimiter); + let mut record = csv::StringRecord::new(); + let mut masks: Vec> = Vec::new(); + let mut header_seen = false; + let mut sampled = 0usize; + while sampled < sample_rows && csv_reader.read_record(&mut record).map_err(csv_read_issue)? { + if config.has_header && !header_seen { + header_seen = true; + masks = vec![None; unique_headers(&record_strings(&record))?.len()]; + continue; + } + if masks.is_empty() { + masks = vec![None; record.len().max(1)]; + } + for (index, value) in record.iter().enumerate().take(masks.len()) { + if let Some(text) = csv_cell_text(value, config) { + masks[index] = Some(intersect_column_types(masks[index], classify_cell(&text))); + } + } + sampled += 1; + } + if !auto { + return Ok(vec![MongoImportInferredType::String; masks.len()]); + } + Ok(masks.into_iter().map(|mask| inferred_type_from_mask(mask.unwrap_or(TYPE_STRING))).collect()) +} + +fn record_strings(record: &csv::StringRecord) -> Vec { + record.iter().map(|value| value.to_string()).collect() +} + +struct CsvPreviewData { + columns: Vec, + inferred: Vec, + documents: Vec, + errors: Vec, + warnings: Vec, + estimated_rows: u64, + estimated_rows_exact: bool, + encoding: TableImportTextEncoding, +} + +fn parse_csv_preview( + path: &str, + options: &MongoImportParseOptions, + preview_limit: usize, +) -> Result { + let config = csv_config(path, options)?; + let inferred = infer_csv_types(path, &config, options.encoding())?; + let (reader, encoding) = open_transcoded_text_file(path, options.encoding()).map_err(encoding_issue)?; + let mut csv_reader = csv_reader(reader, config.delimiter); + let mut record = csv::StringRecord::new(); + let mut headers = Vec::new(); + let mut documents = Vec::new(); + let mut errors = Vec::new(); + let mut warnings = Vec::new(); + let mut index = 0u64; + let mut estimated_rows_exact = true; + while csv_reader.read_record(&mut record).map_err(csv_read_issue)? { + index += 1; + if index == 1 && config.has_header { + headers = unique_headers(&record_strings(&record))?; + continue; + } + if headers.is_empty() { + headers = generated_field_names(record.len().max(1)); + } + if documents.len() + errors.len() >= preview_limit { + estimated_rows_exact = false; + break; + } + let fields = record.iter().map(|value| csv_cell_text(value, &config)).collect::>(); + match document_from_csv_row(index, &headers, &fields, &inferred, &config, true) { + Ok(document) => documents.push(document), + Err(error) => errors.push(error), + } + } + if headers.is_empty() { + return Err(MongoImportIssue::new("CSV_STRUCTURE", "CSV file has no columns")); + } + if !config.has_header { + warnings.push(MongoImportIssue::new( + "GENERATED_HEADERS", + "CSV has no header row; field_1, field_2, … were generated", + )); + } + Ok(CsvPreviewData { + estimated_rows: documents.len() as u64 + errors.len() as u64, + estimated_rows_exact, + columns: headers, + inferred, + documents, + errors, + warnings, + encoding, + }) +} + +fn encoding_issue(error: String) -> MongoImportIssue { + if error.contains("Could not detect text encoding") || error.contains("Invalid byte sequence") { + MongoImportIssue::new("ENCODING", error) + } else { + MongoImportIssue::new("FILE_UNREADABLE", error) + } +} + +fn csv_read_issue(error: csv::Error) -> MongoImportIssue { + let message = error.to_string(); + if message.contains("Invalid byte sequence") { + encoding_issue(message) + } else { + MongoImportIssue::new("CSV_STRUCTURE", message) + } +} + +fn json_document_from_value( + row: u64, + value: serde_json::Value, + with_extended_json: bool, +) -> Result { + if !value.is_object() { + return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "JSON document must be an object") + .with_row(row) + .with_value(value.to_string())); + } + let document = json_object_to_document_extended_json(&value) + .map_err(|error| MongoImportIssue::new("TYPE_CONVERSION", error).with_row(row).with_value(value.to_string()))?; + Ok(parsed_document(row, document, with_extended_json)) +} + +type JsonPreviewOutput = + (Vec, Vec, Vec, u64, bool, TableImportTextEncoding); + +fn parse_json_preview( + path: &str, + options: &MongoImportParseOptions, + preview_limit: usize, + ndjson: bool, +) -> Result { + let (reader, encoding) = open_transcoded_text_file(path, options.encoding()).map_err(encoding_issue)?; + let mut reader = BufReader::new(reader); + let mut documents = Vec::new(); + let mut errors = Vec::new(); + let mut warnings = Vec::new(); + let mut estimated_rows_exact = true; + if ndjson { + for (index, line) in reader.lines().enumerate() { + let row = (index + 1) as u64; + let line = + line.map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()).with_row(row))?; + let trimmed = line.trim(); + if trimmed.is_empty() { + continue; + } + if documents.len() + errors.len() >= preview_limit { + estimated_rows_exact = false; + break; + } + match serde_json::from_str::(trimmed) { + Ok(value) => match json_document_from_value(row, value, true) { + Ok(document) => documents.push(document), + Err(error) => errors.push(error), + }, + Err(error) => errors.push( + MongoImportIssue::new("JSON_ROOT_TYPE", format!("Invalid NDJSON line: {error}")) + .with_row(row) + .with_value(trimmed.to_string()), + ), + } + } + } else { + let mut values = JsonArrayIter::new(&mut reader); + for (index, value) in values.by_ref().enumerate() { + let row = (index + 1) as u64; + if documents.len() + errors.len() >= preview_limit { + estimated_rows_exact = false; + break; + } + match value { + Ok(value) => match json_document_from_value(row, value, true) { + Ok(document) => documents.push(document), + Err(error) => errors.push(error), + }, + Err(error) => { + errors.push(error.with_row(row)); + if !options.skip_error_rows() { + break; + } + } + } + } + if values.single_object { + warnings.push(MongoImportIssue::new( + "SINGLE_OBJECT", + "JSON file contains a single object; it will be imported as one document", + )); + } + if values.finished_with_non_array { + return Err(MongoImportIssue::new( + "JSON_ROOT_TYPE", + "JSON import root value must be an object or an array of objects", + )); + } + } + let estimated_rows = documents.len() as u64 + errors.len() as u64; + Ok((documents, errors, warnings, estimated_rows, estimated_rows_exact, encoding)) +} + +struct JsonArrayIter<'a, R: BufRead> { + reader: &'a mut R, + started: bool, + in_array: bool, + finished: bool, + single_object: bool, + finished_with_non_array: bool, +} + +impl<'a, R: BufRead> JsonArrayIter<'a, R> { + fn new(reader: &'a mut R) -> Self { + Self { + reader, + started: false, + in_array: false, + finished: false, + single_object: false, + finished_with_non_array: false, + } + } +} + +impl Iterator for JsonArrayIter<'_, R> { + type Item = Result; + + fn next(&mut self) -> Option { + if self.finished { + return None; + } + match next_json_value(self) { + Ok(Some(value)) => Some(Ok(value)), + Ok(None) => { + self.finished = true; + None + } + Err(error) => { + self.finished = true; + Some(Err(error)) + } + } + } +} + +fn next_json_value(iter: &mut JsonArrayIter<'_, R>) -> Result, MongoImportIssue> { + skip_json_whitespace(iter.reader)?; + let Some(first) = peek_byte(iter.reader)? else { + return if iter.started { + Ok(None) + } else { + Err(MongoImportIssue::new("JSON_ROOT_TYPE", "JSON file is empty")) + }; + }; + if !iter.started { + iter.started = true; + if first == b'[' { + iter.in_array = true; + iter.reader.consume(1); + skip_json_whitespace(iter.reader)?; + if peek_byte(iter.reader)? == Some(b']') { + iter.reader.consume(1); + return Ok(None); + } + return parse_one_json_value(iter.reader).map(Some); + } + if first == b'{' { + iter.single_object = true; + let value = parse_one_json_value(iter.reader)?; + skip_json_whitespace(iter.reader)?; + if peek_byte(iter.reader)?.is_some() { + return Err(MongoImportIssue::new( + "JSON_ROOT_TYPE", + "JSON import with a root object must contain only that object", + )); + } + iter.finished = true; + return Ok(Some(value)); + } + iter.finished_with_non_array = true; + return Err(MongoImportIssue::new( + "JSON_ROOT_TYPE", + "JSON import root value must be an object or an array of objects", + )); + } + if iter.in_array { + skip_json_whitespace(iter.reader)?; + match peek_byte(iter.reader)? { + Some(b']') => { + iter.reader.consume(1); + iter.finished = true; + return Ok(None); + } + Some(b',') => { + iter.reader.consume(1); + skip_json_whitespace(iter.reader)?; + if peek_byte(iter.reader)? == Some(b']') { + iter.reader.consume(1); + iter.finished = true; + return Ok(None); + } + return parse_one_json_value(iter.reader).map(Some); + } + Some(_) => { + return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "Expected comma or end of JSON array")); + } + None => return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "Unterminated JSON array")), + } + } + Ok(None) +} + +fn parse_one_json_value(reader: &mut R) -> Result { + let bytes = extract_json_value(reader)?; + serde_json::from_slice(&bytes).map_err(|error| MongoImportIssue::new("JSON_ROOT_TYPE", error.to_string())) +} + +fn skip_json_whitespace(reader: &mut R) -> Result<(), MongoImportIssue> { + loop { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + if buffer.is_empty() { + return Ok(()); + } + let skip = buffer.iter().take_while(|byte| byte.is_ascii_whitespace()).count(); + if skip == 0 { + return Ok(()); + } + reader.consume(skip); + } +} + +fn peek_byte(reader: &mut R) -> Result, MongoImportIssue> { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + Ok(buffer.first().copied()) +} + +fn extract_json_value(reader: &mut R) -> Result, MongoImportIssue> { + skip_json_whitespace(reader)?; + let Some(first) = peek_byte(reader)? else { + return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "Unexpected end of JSON")); + }; + match first { + b'{' | b'[' => extract_json_container(reader, first), + b'"' => extract_json_string(reader), + b't' | b'f' | b'n' => extract_json_literal(reader), + b'-' | b'0'..=b'9' => extract_json_number(reader), + other => Err(MongoImportIssue::new("JSON_ROOT_TYPE", format!("Unexpected JSON byte: {}", other as char))), + } +} + +fn extract_json_container(reader: &mut R, open: u8) -> Result, MongoImportIssue> { + let close = if open == b'{' { b'}' } else { b']' }; + let mut out = Vec::new(); + let mut depth = 0usize; + let mut in_string = false; + let mut escaped = false; + loop { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + if buffer.is_empty() { + return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "Unterminated JSON value")); + } + let mut consumed = 0usize; + for &byte in buffer { + out.push(byte); + consumed += 1; + if in_string { + if escaped { + escaped = false; + } else if byte == b'\\' { + escaped = true; + } else if byte == b'"' { + in_string = false; + } + continue; + } + match byte { + b'"' => in_string = true, + b if b == open => depth += 1, + b if b == close => { + depth -= 1; + if depth == 0 { + reader.consume(consumed); + return Ok(out); + } + } + _ => {} + } + } + reader.consume(consumed); + } +} + +fn extract_json_string(reader: &mut R) -> Result, MongoImportIssue> { + let mut out = Vec::new(); + let mut escaped = false; + let mut started = false; + loop { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + if buffer.is_empty() { + return Err(MongoImportIssue::new("JSON_ROOT_TYPE", "Unterminated JSON string")); + } + let mut consumed = 0usize; + for &byte in buffer { + out.push(byte); + consumed += 1; + if !started { + started = true; + continue; + } + if escaped { + escaped = false; + continue; + } + if byte == b'\\' { + escaped = true; + continue; + } + if byte == b'"' { + reader.consume(consumed); + return Ok(out); + } + } + reader.consume(consumed); + } +} + +fn extract_json_literal(reader: &mut R) -> Result, MongoImportIssue> { + let mut out = Vec::new(); + loop { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + if buffer.is_empty() { + break; + } + let take = buffer.iter().take_while(|byte| byte.is_ascii_alphabetic()).count(); + if take == 0 { + break; + } + out.extend_from_slice(&buffer[..take]); + let exhausted = take == buffer.len(); + reader.consume(take); + if !exhausted { + break; + } + } + Ok(out) +} + +fn extract_json_number(reader: &mut R) -> Result, MongoImportIssue> { + let mut out = Vec::new(); + loop { + let buffer = reader.fill_buf().map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + if buffer.is_empty() { + break; + } + let take = + buffer.iter().take_while(|byte| matches!(**byte, b'0'..=b'9' | b'+' | b'-' | b'.' | b'e' | b'E')).count(); + if take == 0 { + break; + } + out.extend_from_slice(&buffer[..take]); + let exhausted = take == buffer.len(); + reader.consume(take); + if !exhausted { + break; + } + } + Ok(out) +} + +fn columns_from_documents(documents: &[ParsedMongoDocument]) -> Vec { + let mut names = Vec::new(); + let mut seen = HashSet::new(); + for document in documents { + for key in document.document.keys() { + if seen.insert(key.clone()) { + names.push(key.clone()); + } + } + } + names + .into_iter() + .map(|name| { + let samples = documents + .iter() + .filter_map(|document| document.extended_json.get(&name).cloned()) + .take(3) + .collect::>(); + MongoImportColumn { name, inferred_type: MongoImportInferredType::String, sample_values: samples } + }) + .collect() +} + +fn file_name(path: &str) -> String { + Path::new(path).file_name().and_then(|name| name.to_str()).unwrap_or(path).to_string() +} + +fn file_size(path: &str) -> u64 { + std::fs::metadata(path).map(|metadata| metadata.len()).unwrap_or(0) +} + +pub fn preview_mongodb_import_file( + request: &MongoImportPreviewRequest, +) -> Result { + validate_import_source_path(&request.file_path)?; + let preview_limit = request.preview_limit.unwrap_or(DEFAULT_PREVIEW_LIMIT).max(1); + match request.format { + MongoImportFormat::Csv => { + let parsed = parse_csv_preview(&request.file_path, &request.parse_options, preview_limit)?; + let columns = parsed + .columns + .into_iter() + .zip(parsed.inferred) + .map(|(name, inferred_type)| { + let sample_values = parsed + .documents + .iter() + .filter_map(|document| document.extended_json.get(&name).cloned()) + .take(3) + .collect(); + MongoImportColumn { name, inferred_type, sample_values } + }) + .collect(); + Ok(MongoImportPreview { + source_ref: request.source_ref.clone(), + format: request.format, + detected_encoding: Some(parsed.encoding), + file_name: file_name(&request.file_path), + file_path: request.file_path.clone(), + size_bytes: file_size(&request.file_path), + columns, + row_numbers: parsed.documents.iter().map(|document| document.row).collect(), + rows: parsed.documents.into_iter().map(|document| document.extended_json).collect(), + warnings: parsed.warnings, + errors: parsed.errors, + estimated_rows: Some(parsed.estimated_rows), + estimated_rows_exact: parsed.estimated_rows_exact, + }) + } + MongoImportFormat::Json | MongoImportFormat::Ndjson => { + let ndjson = request.format == MongoImportFormat::Ndjson; + let (documents, errors, warnings, estimated_rows, estimated_rows_exact, encoding) = + parse_json_preview(&request.file_path, &request.parse_options, preview_limit, ndjson)?; + Ok(MongoImportPreview { + source_ref: request.source_ref.clone(), + format: request.format, + detected_encoding: Some(encoding), + file_name: file_name(&request.file_path), + file_path: request.file_path.clone(), + size_bytes: file_size(&request.file_path), + columns: columns_from_documents(&documents), + row_numbers: documents.iter().map(|document| document.row).collect(), + rows: documents.into_iter().map(|document| document.extended_json).collect(), + warnings, + errors, + estimated_rows: Some(estimated_rows), + estimated_rows_exact, + }) + } + } +} + +pub fn preview_mongodb_import_bytes( + bytes: &[u8], + format: MongoImportFormat, + options: &MongoImportParseOptions, + preview_limit: usize, +) -> Result { + let extension = match format { + MongoImportFormat::Csv => "csv", + MongoImportFormat::Json => "json", + MongoImportFormat::Ndjson => "ndjson", + }; + let dir = std::env::temp_dir().join(format!("dbx-mongo-import-preview-{}", uuid::Uuid::new_v4())); + std::fs::create_dir_all(&dir).map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + let path = dir.join(format!("preview.{extension}")); + std::fs::write(&path, bytes).map_err(|error| MongoImportIssue::new("FILE_UNREADABLE", error.to_string()))?; + let result = preview_mongodb_import_file(&MongoImportPreviewRequest { + file_path: path.to_string_lossy().to_string(), + source_ref: None, + format, + parse_options: options.clone(), + preview_limit: Some(preview_limit), + }); + let _ = std::fs::remove_dir_all(dir); + result +} + +fn stream_csv_documents( + reader: R, + config: &CsvParseConfig, + inferred: &[MongoImportInferredType], + mut on_document: F, +) -> Result<(), MongoImportIssue> +where + R: Read, + F: FnMut(Result) -> Result<(), MongoImportIssue>, +{ + let mut csv_reader = csv_reader(reader, config.delimiter); + let mut record = csv::StringRecord::new(); + let mut headers = Vec::new(); + let mut index = 0u64; + while csv_reader.read_record(&mut record).map_err(csv_read_issue)? { + index += 1; + if index == 1 && config.has_header { + headers = unique_headers(&record_strings(&record))?; + continue; + } + if headers.is_empty() { + headers = generated_field_names(record.len().max(1)); + } + let fields = record.iter().map(|value| csv_cell_text(value, config)).collect::>(); + on_document(document_from_csv_row(index, &headers, &fields, inferred, config, false))?; + } + Ok(()) +} + +fn stream_csv_file(path: &str, options: &MongoImportParseOptions, on_document: F) -> Result<(), MongoImportIssue> +where + F: FnMut(Result) -> Result<(), MongoImportIssue>, +{ + let config = csv_config(path, options)?; + let inferred = infer_csv_types(path, &config, options.encoding())?; + let (reader, _) = open_transcoded_text_file(path, options.encoding()).map_err(encoding_issue)?; + stream_csv_documents(reader, &config, &inferred, on_document) +} + +fn stream_json_file( + path: &str, + options: &MongoImportParseOptions, + ndjson: bool, + mut on_document: F, +) -> Result<(), MongoImportIssue> +where + F: FnMut(Result) -> Result<(), MongoImportIssue>, +{ + let (reader, _) = open_transcoded_text_file(path, options.encoding()).map_err(encoding_issue)?; + let mut reader = BufReader::new(reader); + if ndjson { + for (index, line) in reader.lines().enumerate() { + let row = (index + 1) as u64; + let line = match line { + Ok(line) => line, + Err(error) => { + on_document(Err(MongoImportIssue::new("FILE_UNREADABLE", error.to_string()).with_row(row)))?; + continue; + } + }; + let trimmed = line.trim(); + if trimmed.is_empty() { + continue; + } + let parsed = serde_json::from_str::(trimmed) + .map_err(|error| { + MongoImportIssue::new("JSON_ROOT_TYPE", format!("Invalid NDJSON line: {error}")) + .with_row(row) + .with_value(trimmed.to_string()) + }) + .and_then(|value| json_document_from_value(row, value, false)); + on_document(parsed)?; + } + return Ok(()); + } + for (index, value) in JsonArrayIter::new(&mut reader).enumerate() { + let row = (index + 1) as u64; + on_document(value.and_then(|value| json_document_from_value(row, value, false)))?; + } + Ok(()) +} + +pub fn for_each_mongodb_import_document( + path: &str, + format: MongoImportFormat, + options: &MongoImportParseOptions, + on_document: F, +) -> Result<(), MongoImportIssue> +where + F: FnMut(Result) -> Result<(), MongoImportIssue>, +{ + validate_import_source_path(path)?; + match format { + MongoImportFormat::Csv => stream_csv_file(path, options, on_document), + MongoImportFormat::Json => stream_json_file(path, options, false, on_document), + MongoImportFormat::Ndjson => stream_json_file(path, options, true, on_document), + } +} + +#[allow(clippy::too_many_arguments)] +fn progress( + import_id: &str, + phase: MongoImportPhase, + status: MongoImportStatus, + rows_read: u64, + rows_inserted: u64, + rows_failed: u64, + batches_committed: u64, + total_rows: Option, + error_rows: Vec, + error_message: Option, + started_at: Instant, +) -> MongoImportProgress { + MongoImportProgress { + import_id: import_id.to_string(), + phase, + status, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + total_rows, + error_rows, + error_message, + elapsed_ms: started_at.elapsed().as_millis(), + } +} + +async fn insert_documents_batch( + state: &AppState, + connection_id: &str, + database: &str, + collection: &str, + documents: Vec, + require_bson_types: bool, +) -> Result { + let pool = + state.pool_handle(connection_id).await.ok_or_else(|| MongoImportIssue::new("CONNECTION", "Not found"))?; + match &pool { + PoolKind::MongoDb(client) => { + insert_bson_documents(client, database, collection, documents).await.map_err(bulk_write_issue) + } + PoolKind::Agent(client) => { + if require_bson_types { + return Err(MongoImportIssue::new( + "LEGACY_AGENT", + "MongoDB Legacy Agent cannot preserve BSON types during file import; use the native MongoDB driver", + )); + } + let mut client = client.lock().await; + if !client.supports_capability(AgentCapability::MongoInsertDocuments) { + return Err(MongoImportIssue::new( + "LEGACY_AGENT", + "MongoDB Legacy Agent does not support insertMany; upgrade or reinstall the MongoDB Legacy driver", + )); + } + let docs_json = + serde_json::to_string(&documents.iter().map(document_to_canonical_extended_json).collect::>()) + .map_err(|error| MongoImportIssue::new("TYPE_CONVERSION", error.to_string()))?; + let result: serde_json::Value = client + .mongo_insert_documents(serde_json::json!({ + "database": database, + "collection": collection, + "docs_json": docs_json, + })) + .await + .map_err(|error| MongoImportIssue::new("PERMISSION", error).retryable())?; + let inserted = result.get("affected_rows").and_then(serde_json::Value::as_u64).ok_or_else(|| { + MongoImportIssue::new("CONNECTION", "MongoDB Legacy Agent returned an invalid insertMany result") + })?; + Ok(MongoInsertOutcome { inserted, errors: Vec::new() }) + } + _ => Err(MongoImportIssue::new("CONNECTION", "Not a MongoDB connection")), + } +} + +/// Rewrites a write issue's batch-relative index into the source file row it came from, so the +/// user can find the offending record. +fn located_in_batch(mut issue: MongoImportIssue, rows: &[u64], batch: u64) -> MongoImportIssue { + issue.row = issue + .row + .and_then(|index| rows.get(index.saturating_sub(1) as usize).copied()) + .or_else(|| rows.first().copied()); + issue.batch = Some(batch); + issue +} + +fn bulk_write_issue(error: MongoBulkWriteError) -> MongoImportIssue { + let mut issue = + MongoImportIssue::new(if error.code == Some(11000) { "DUPLICATE_ID" } else { "TARGET_WRITE" }, error.message); + issue.retryable = error.retryable; + if let Some(index) = error.index { + issue.row = Some(index as u64 + 1); + } + issue +} + +pub async fn import_mongodb_file_core( + state: &AppState, + request: &MongoImportRequest, + mut is_cancelled: C, + mut on_progress: F, +) -> Result +where + C: FnMut(&str) -> std::pin::Pin + Send>>, + F: FnMut(MongoImportProgress), +{ + let started_at = Instant::now(); + let batch_size = clamp_batch_size(if request.batch_size == 0 { DEFAULT_BATCH_SIZE } else { request.batch_size })?; + let require_bson_types = !matches!(request.parse_options.type_mode(), MongoImportTypeMode::String); + on_progress(progress( + &request.import_id, + MongoImportPhase::Preparing, + MongoImportStatus::Running, + 0, + 0, + 0, + 0, + None, + Vec::new(), + None, + started_at, + )); + if is_cancelled(&request.import_id).await { + return cancel_import(&request.import_id, 0, 0, 0, 0, Vec::new(), started_at, &mut on_progress); + } + validate_import_source_path(&request.file_path)?; + state + .get_or_create_pool(&request.connection_id, Some(&request.database)) + .await + .map_err(|error| MongoImportIssue::new("CONNECTION", error).retryable())?; + + on_progress(progress( + &request.import_id, + MongoImportPhase::Parsing, + MongoImportStatus::Running, + 0, + 0, + 0, + 0, + None, + Vec::new(), + None, + started_at, + )); + + let (tx, mut rx) = tokio::sync::mpsc::channel::(2); + let path = request.file_path.clone(); + let format = request.format; + let options = request.parse_options.clone(); + let skip_error_rows = options.skip_error_rows(); + tokio::task::spawn_blocking(move || { + let mut rows = Vec::new(); + let mut documents = Vec::new(); + let result = for_each_mongodb_import_document(&path, format, &options, |parsed| match parsed { + Ok(parsed) => { + rows.push(parsed.row); + documents.push(parsed.document); + if documents.len() >= batch_size + && tx + .blocking_send(ImportBatchEvent::Batch { + rows: std::mem::take(&mut rows), + documents: std::mem::take(&mut documents), + }) + .is_err() + { + return Err(MongoImportIssue::new("CANCELLED", "Import cancelled")); + } + Ok(()) + } + Err(error) if skip_error_rows => { + if tx.blocking_send(ImportBatchEvent::RowError(error)).is_err() { + Err(MongoImportIssue::new("CANCELLED", "Import cancelled")) + } else { + Ok(()) + } + } + Err(error) => Err(error), + }); + if !documents.is_empty() { + let _ = tx.blocking_send(ImportBatchEvent::Batch { rows, documents }); + } + if let Err(error) = result { + if error.code != "CANCELLED" { + let _ = tx.blocking_send(ImportBatchEvent::Fatal(error)); + } + } + }); + + let mut rows_read = 0u64; + let mut rows_inserted = 0u64; + let mut rows_failed = 0u64; + let mut batches_committed = 0u64; + let mut error_rows = Vec::new(); + + while let Some(event) = rx.recv().await { + if is_cancelled(&request.import_id).await { + return cancel_import( + &request.import_id, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + error_rows, + started_at, + &mut on_progress, + ); + } + match event { + ImportBatchEvent::RowError(error) => { + rows_read += 1; + rows_failed += 1; + error_rows.push(error); + } + ImportBatchEvent::Fatal(error) => { + let message = error.display_message(); + error_rows.push(error.clone()); + on_progress(progress( + &request.import_id, + MongoImportPhase::Done, + MongoImportStatus::Error, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + None, + error_rows, + Some(message), + started_at, + )); + return Err(error); + } + ImportBatchEvent::Batch { rows, documents } => { + rows_read += documents.len() as u64; + on_progress(progress( + &request.import_id, + MongoImportPhase::Writing, + MongoImportStatus::Running, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + None, + Vec::new(), + None, + started_at, + )); + match insert_documents_batch( + state, + &request.connection_id, + &request.database, + &request.collection, + documents, + require_bson_types, + ) + .await + { + Ok(outcome) => { + rows_inserted += outcome.inserted; + rows_failed += outcome.errors.len() as u64; + batches_committed += 1; + let rejected = outcome + .errors + .into_iter() + .map(|error| located_in_batch(bulk_write_issue(error), &rows, batches_committed)) + .collect::>(); + // Documents the server refused individually: the rest of the batch is + // already written, so honour skipErrorRows exactly like a parse error. + if let Some(fatal) = rejected.first().filter(|_| !skip_error_rows).cloned() { + let message = fatal.display_message(); + error_rows.extend(rejected); + on_progress(progress( + &request.import_id, + MongoImportPhase::Done, + MongoImportStatus::Error, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + None, + error_rows, + Some(message), + started_at, + )); + return Err(fatal); + } + error_rows.extend(rejected); + } + Err(error) => { + let error = located_in_batch(error, &rows, batches_committed + 1); + rows_failed += rows.len() as u64; + let message = error.display_message(); + error_rows.push(error.clone()); + on_progress(progress( + &request.import_id, + MongoImportPhase::Done, + MongoImportStatus::Error, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + None, + error_rows, + Some(message), + started_at, + )); + return Err(error); + } + } + } + } + } + + if is_cancelled(&request.import_id).await { + return cancel_import( + &request.import_id, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + error_rows, + started_at, + &mut on_progress, + ); + } + + on_progress(progress( + &request.import_id, + MongoImportPhase::Done, + MongoImportStatus::Done, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + Some(rows_read), + error_rows, + None, + started_at, + )); + Ok(MongoImportSummary { + import_id: request.import_id.clone(), + rows_inserted, + rows_failed, + batches_committed, + elapsed_ms: started_at.elapsed().as_millis(), + }) +} + +enum ImportBatchEvent { + Batch { rows: Vec, documents: Vec }, + RowError(MongoImportIssue), + Fatal(MongoImportIssue), +} + +fn cancel_import( + import_id: &str, + rows_read: u64, + rows_inserted: u64, + rows_failed: u64, + batches_committed: u64, + error_rows: Vec, + started_at: Instant, + on_progress: &mut F, +) -> Result +where + F: FnMut(MongoImportProgress), +{ + on_progress(progress( + import_id, + MongoImportPhase::Done, + MongoImportStatus::Cancelled, + rows_read, + rows_inserted, + rows_failed, + batches_committed, + None, + error_rows, + Some("Import cancelled".to_string()), + started_at, + )); + Err(MongoImportIssue::new("CANCELLED", "Import cancelled")) +} + +pub async fn preview_mongodb_import_file_core( + request: MongoImportPreviewRequest, +) -> Result { + preview_mongodb_import_file(&request).map_err(|error| error.display_message()) +} + +pub const DEFAULT_EXPORT_BATCH_SIZE: u32 = 1000; +pub const MAX_CSV_FIELDS: usize = 256; + +pub fn mongodb_export_client_session_id(export_id: &str) -> String { + task_client_session_id("mongo-export", export_id) +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoExportFormat { + Csv, + Ndjson, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub enum MongoExportStatus { + Running, + Done, + Error, + Cancelled, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoExportRequest { + pub export_id: String, + pub connection_id: String, + pub database: String, + pub collection: String, + #[serde(default)] + pub filter: Option, + #[serde(default)] + pub sort: Option, + #[serde(default)] + pub projection: Option, + #[serde(default)] + pub collation: Option, + pub format: MongoExportFormat, + #[serde(default = "default_true")] + pub include_header: bool, + pub file_path: String, + #[serde(default, skip_serializing_if = "Option::is_none")] + pub execution_id: Option, +} + +const fn default_true() -> bool { + true +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoExportProgress { + pub export_id: String, + pub status: MongoExportStatus, + pub documents_read: u64, + pub bytes_written: u64, + #[serde(skip_serializing_if = "Option::is_none")] + pub total_documents: Option, + #[serde(skip_serializing_if = "Option::is_none")] + pub error_message: Option, + pub elapsed_ms: u128, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct MongoExportSummary { + pub export_id: String, + pub documents_exported: u64, + pub file_path: String, + pub elapsed_ms: u128, +} + +fn temp_export_path(target: &Path) -> PathBuf { + let name = target.file_name().and_then(|name| name.to_str()).unwrap_or("export"); + target.with_file_name(format!(".{name}.dbx-export.tmp")) +} + +fn cleanup_temp(path: &Path) { + let _ = std::fs::remove_file(path); +} + +fn atomic_rename(temp: &Path, target: &Path) -> Result<(), String> { + match std::fs::rename(temp, target) { + Ok(()) => Ok(()), + Err(_) if target.exists() => { + std::fs::remove_file(target).map_err(|error| error.to_string())?; + std::fs::rename(temp, target).map_err(|error| error.to_string()) + } + Err(error) => Err(error.to_string()), + } +} + +fn write_export_line(writer: &mut BufWriter, line: &str) -> Result { + writer.write_all(line.as_bytes()).map_err(|error| error.to_string())?; + Ok(line.len() as u64) +} + +fn unwrap_extended_json_csv_scalar(value: &serde_json::Value) -> Option { + let object = value.as_object()?; + if object.len() != 1 { + return None; + } + let (key, inner) = object.iter().next()?; + match key.as_str() { + // Only types a CSV cell can carry without losing its identity on reimport. `$uuid`, + // `$symbol`, `$binary` and friends stay as Extended JSON text so that + // `Bson::try_from(serde_json::Value)` rebuilds the original type. + "$oid" | "$numberInt" | "$numberLong" | "$numberDouble" | "$numberDecimal" => { + inner.as_str().map(str::to_string) + } + "$date" => match inner { + serde_json::Value::String(iso) => Some(iso.clone()), + serde_json::Value::Object(date) => { + date.get("$numberLong").and_then(serde_json::Value::as_str).and_then(|millis| { + millis.parse::().ok().and_then(|ms| { + ChronoDateTime::::from_timestamp_millis(ms) + .map(|dt| dt.to_rfc3339_opts(chrono::SecondsFormat::Millis, true)) + }) + }) + } + _ => None, + }, + _ => None, + } +} + +/// Resolves an export header back to its value, understanding the same grammar the import +/// side parses, so `tags[0]` and `address.city` both round-trip. +fn json_at_field_path<'a>(value: &'a serde_json::Value, path: &str) -> Option<&'a serde_json::Value> { + let mut current = value; + for segment in parse_field_path(path) { + current = match segment { + PathSegment::Key(key) => current.get(key)?, + PathSegment::Index(index) => current.get(index)?, + }; + } + Some(current) +} + +fn push_csv_json_value(out: &mut String, value: Option<&serde_json::Value>) { + match value { + None | Some(serde_json::Value::Null) => {} + // Deliberately no spreadsheet formula guard: prefixing `'` to values starting with + // `= + - @` would make reimport see a different string than was exported. + Some(serde_json::Value::String(value)) => push_csv_field(out, value, CsvQuoteMode::Necessary), + Some(serde_json::Value::Bool(value)) => out.push_str(if *value { "true" } else { "false" }), + Some(serde_json::Value::Number(value)) => out.push_str(&value.to_string()), + Some(other) => { + if let Some(scalar) = unwrap_extended_json_csv_scalar(other) { + push_csv_field(out, &scalar, CsvQuoteMode::Necessary); + } else { + push_csv_field(out, &other.to_string(), CsvQuoteMode::Necessary); + } + } + } +} + +fn format_csv_document_line(fields: &[String], document: &serde_json::Value) -> String { + let mut line = String::new(); + for (index, field) in fields.iter().enumerate() { + if index > 0 { + line.push(','); + } + push_csv_json_value(&mut line, json_at_field_path(document, field)); + } + line.push('\n'); + line +} + +fn push_csv_field_name(name: &str, fields: &mut Vec, seen: &mut HashSet) -> Result<(), String> { + if !seen.insert(name.to_string()) { + return Ok(()); + } + if fields.len() >= MAX_CSV_FIELDS { + return Err(format!( + "CSV export found more than {MAX_CSV_FIELDS} unique fields; use NDJSON for this collection" + )); + } + fields.push(name.to_string()); + Ok(()) +} + +fn collect_csv_fields( + document: &serde_json::Value, + fields: &mut Vec, + seen: &mut HashSet, +) -> Result<(), String> { + collect_csv_fields_at(document, "", fields, seen) +} + +/// Expands a document into Compass-style headers: nested objects become `address.city`, +/// arrays become `tags[0]`, and anything a single cell can hold becomes one leaf column. +/// Extended JSON wrappers, empty objects and empty arrays are leaves so their type survives. +fn collect_csv_fields_at( + value: &serde_json::Value, + path: &str, + fields: &mut Vec, + seen: &mut HashSet, +) -> Result<(), String> { + if unwrap_extended_json_csv_scalar(value).is_none() { + match value { + serde_json::Value::Object(object) if !object.is_empty() => { + for (key, child) in object { + let child_path = if path.is_empty() { key.clone() } else { format!("{path}.{key}") }; + collect_csv_fields_at(child, &child_path, fields, seen)?; + } + return Ok(()); + } + // A top-level array is not a document, so only descend once we have a field name. + serde_json::Value::Array(items) if !items.is_empty() && !path.is_empty() => { + for (index, child) in items.iter().enumerate() { + collect_csv_fields_at(child, &format!("{path}[{index}]"), fields, seen)?; + } + return Ok(()); + } + _ => {} + } + } + if !path.is_empty() { + push_csv_field_name(path, fields, seen)?; + } + Ok(()) +} + +fn move_id_first(fields: &mut Vec) { + if let Some(index) = fields.iter().position(|field| field == "_id") { + if index > 0 { + let id = fields.remove(index); + fields.insert(0, id); + } + } +} + +fn write_csv_header_and_buffer( + include_header: bool, + fields: &mut Vec, + buffered: &mut Vec, + writer: &mut BufWriter, + bytes_written: &mut u64, + documents_read: &mut u64, +) -> Result<(), String> { + if fields.is_empty() { + fields.push("_id".to_string()); + } + move_id_first(fields); + if include_header { + let mut line = String::new(); + for (index, field) in fields.iter().enumerate() { + if index > 0 { + line.push(','); + } + push_csv_field(&mut line, field, CsvQuoteMode::Necessary); + } + line.push('\n'); + *bytes_written += write_export_line(writer, &line)?; + } + for json in buffered.drain(..) { + let line = format_csv_document_line(fields, &json); + *bytes_written += write_export_line(writer, &line)?; + *documents_read += 1; + } + Ok(()) +} + +fn export_progress( + export_id: &str, + status: MongoExportStatus, + documents_read: u64, + bytes_written: u64, + total_documents: Option, + error_message: Option, + started_at: Instant, +) -> MongoExportProgress { + MongoExportProgress { + export_id: export_id.to_string(), + status, + documents_read, + bytes_written, + total_documents, + error_message, + elapsed_ms: started_at.elapsed().as_millis(), + } +} + +pub async fn export_mongodb_query_core( + state: &AppState, + request: &MongoExportRequest, + mut is_cancelled: C, + mut on_progress: F, +) -> Result +where + C: FnMut(&str) -> std::pin::Pin + Send>>, + F: FnMut(MongoExportProgress), +{ + let started_at = Instant::now(); + on_progress(export_progress(&request.export_id, MongoExportStatus::Running, 0, 0, None, None, started_at)); + if is_cancelled(&request.export_id).await { + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Cancelled, + 0, + 0, + None, + Some("Export cancelled".to_string()), + started_at, + )); + return Err("Export cancelled".to_string()); + } + + state.get_or_create_pool(&request.connection_id, Some(&request.database)).await?; + let pool = state.pool_handle(&request.connection_id).await.ok_or_else(|| "Not found".to_string())?; + let client = match &pool { + PoolKind::MongoDb(client) => client.clone(), + PoolKind::Agent(_) => { + return Err( + "MongoDB Legacy Agent does not support cursor export of the full query; use the native MongoDB driver" + .to_string(), + ); + } + _ => return Err("Not a MongoDB connection".to_string()), + }; + + let total_documents = if request + .filter + .as_deref() + .is_none_or(|filter| filter.trim().is_empty() || filter.trim() == "{}") + { + mongo_driver::count_documents(&client, &request.database, &request.collection, request.filter.as_deref(), false) + .await + .ok() + } else { + None + }; + + let target = PathBuf::from(&request.file_path); + if let Some(parent) = target.parent() { + std::fs::create_dir_all(parent).map_err(|error| error.to_string())?; + } + let temp = temp_export_path(&target); + cleanup_temp(&temp); + + let result = match request.format { + MongoExportFormat::Ndjson => { + export_ndjson(&client, request, &temp, total_documents, started_at, &mut is_cancelled, &mut on_progress) + .await + } + MongoExportFormat::Csv => { + export_csv(&client, request, &temp, total_documents, started_at, &mut is_cancelled, &mut on_progress).await + } + }; + + match result { + Ok((documents_read, _bytes_written)) => { + if is_cancelled(&request.export_id).await { + cleanup_temp(&temp); + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Cancelled, + documents_read, + 0, + total_documents, + Some("Export cancelled".to_string()), + started_at, + )); + return Err("Export cancelled".to_string()); + } + atomic_rename(&temp, &target)?; + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Done, + documents_read, + std::fs::metadata(&target).map(|metadata| metadata.len()).unwrap_or(0), + Some(documents_read), + None, + started_at, + )); + Ok(MongoExportSummary { + export_id: request.export_id.clone(), + documents_exported: documents_read, + file_path: request.file_path.clone(), + elapsed_ms: started_at.elapsed().as_millis(), + }) + } + Err(error) => { + cleanup_temp(&temp); + let cancelled = error == "Export cancelled" || is_cancelled(&request.export_id).await; + on_progress(export_progress( + &request.export_id, + if cancelled { MongoExportStatus::Cancelled } else { MongoExportStatus::Error }, + 0, + 0, + total_documents, + Some(error.clone()), + started_at, + )); + Err(error) + } + } +} + +async fn export_ndjson( + client: &mongodb::Client, + request: &MongoExportRequest, + temp: &Path, + total_documents: Option, + started_at: Instant, + is_cancelled: &mut C, + on_progress: &mut F, +) -> Result<(u64, u64), String> +where + C: FnMut(&str) -> std::pin::Pin + Send>>, + F: FnMut(MongoExportProgress), +{ + let file = File::create(temp).map_err(|error| error.to_string())?; + let mut writer = BufWriter::new(file); + let mut documents_read = 0u64; + let mut bytes_written = 0u64; + for_each_find_document( + client, + &request.database, + &request.collection, + request.filter.as_deref(), + request.projection.as_deref(), + request.sort.as_deref(), + request.collation.as_deref(), + DEFAULT_EXPORT_BATCH_SIZE, + |document| { + let json = document_to_canonical_extended_json(&document); + let mut line = json.to_string(); + line.push('\n'); + bytes_written += write_export_line(&mut writer, &line)?; + documents_read += 1; + if documents_read == 1 || documents_read.is_multiple_of(500) { + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Running, + documents_read, + bytes_written, + total_documents, + None, + started_at, + )); + } + Ok(()) + }, + ) + .await?; + writer.flush().map_err(|error| error.to_string())?; + if is_cancelled(&request.export_id).await { + return Err("Export cancelled".to_string()); + } + Ok((documents_read, bytes_written)) +} + +const CSV_FIELD_DISCOVERY_DOCS: usize = 10_000; + +async fn export_csv( + client: &mongodb::Client, + request: &MongoExportRequest, + temp: &Path, + total_documents: Option, + started_at: Instant, + is_cancelled: &mut C, + on_progress: &mut F, +) -> Result<(u64, u64), String> +where + C: FnMut(&str) -> std::pin::Pin + Send>>, + F: FnMut(MongoExportProgress), +{ + if is_cancelled(&request.export_id).await { + return Err("Export cancelled".to_string()); + } + let file = File::create(temp).map_err(|error| error.to_string())?; + let mut writer = BufWriter::new(file); + let mut fields = Vec::new(); + let mut seen = HashSet::new(); + let mut buffered = Vec::new(); + let mut header_ready = false; + let mut documents_read = 0u64; + let mut bytes_written = 0u64; + + for_each_find_document( + client, + &request.database, + &request.collection, + request.filter.as_deref(), + request.projection.as_deref(), + request.sort.as_deref(), + request.collation.as_deref(), + DEFAULT_EXPORT_BATCH_SIZE, + |document| { + if !header_ready { + let json = document_to_canonical_extended_json(&document); + collect_csv_fields(&json, &mut fields, &mut seen)?; + buffered.push(json); + if buffered.len() >= CSV_FIELD_DISCOVERY_DOCS { + write_csv_header_and_buffer( + request.include_header, + &mut fields, + &mut buffered, + &mut writer, + &mut bytes_written, + &mut documents_read, + )?; + header_ready = true; + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Running, + documents_read, + bytes_written, + total_documents, + None, + started_at, + )); + } + return Ok(()); + } + let json = document_to_canonical_extended_json(&document); + let line = format_csv_document_line(&fields, &json); + bytes_written += write_export_line(&mut writer, &line)?; + documents_read += 1; + if documents_read.is_multiple_of(500) { + on_progress(export_progress( + &request.export_id, + MongoExportStatus::Running, + documents_read, + bytes_written, + total_documents, + None, + started_at, + )); + } + Ok(()) + }, + ) + .await?; + if !header_ready { + write_csv_header_and_buffer( + request.include_header, + &mut fields, + &mut buffered, + &mut writer, + &mut bytes_written, + &mut documents_read, + )?; + } + writer.flush().map_err(|error| error.to_string())?; + if is_cancelled(&request.export_id).await { + return Err("Export cancelled".to_string()); + } + Ok((documents_read, bytes_written)) +} + +pub fn csv_fields_from_extended_documents(documents: &[serde_json::Value]) -> Result, String> { + let mut fields = Vec::new(); + let mut seen = HashSet::new(); + for document in documents { + collect_csv_fields(document, &mut fields, &mut seen)?; + } + move_id_first(&mut fields); + Ok(fields) +} + +pub fn format_mongo_csv_row(fields: &[String], document: &serde_json::Value) -> String { + let mut line = format_csv_document_line(fields, document); + line.pop(); + line +} + +#[cfg(test)] +mod tests { + use super::*; + use mongodb::bson::{doc, Bson, Document}; + use std::io::Cursor; + + fn options(type_mode: MongoImportTypeMode) -> MongoImportParseOptions { + MongoImportParseOptions { type_mode: Some(type_mode), ..MongoImportParseOptions::default() } + } + + fn preview_csv(csv: &str, type_mode: MongoImportTypeMode) -> MongoImportPreview { + preview_mongodb_import_bytes(csv.as_bytes(), MongoImportFormat::Csv, &options(type_mode), 50).unwrap() + } + + fn execute_docs(csv: &str, type_mode: MongoImportTypeMode) -> Vec { + execute_source(csv.as_bytes(), "csv", MongoImportFormat::Csv, &options(type_mode)) + } + + fn execute_source( + bytes: &[u8], + extension: &str, + format: MongoImportFormat, + parse_options: &MongoImportParseOptions, + ) -> Vec { + let dir = std::env::temp_dir().join(format!("dbx-mongo-import-exec-{}", uuid::Uuid::new_v4())); + std::fs::create_dir_all(&dir).unwrap(); + let path = dir.join(format!("data.{extension}")); + std::fs::write(&path, bytes).unwrap(); + let mut docs = Vec::new(); + for_each_mongodb_import_document(path.to_str().unwrap(), format, parse_options, |parsed| { + docs.push(document_to_canonical_extended_json(&parsed?.document)); + Ok(()) + }) + .unwrap(); + let _ = std::fs::remove_dir_all(dir); + docs + } + + // Test helper: simplified CSV export that takes documents directly + fn export_csv_simple(documents: Vec, include_header: bool) -> Result, String> { + let mut output = Vec::new(); + let extended: Vec = documents.iter().map(document_to_canonical_extended_json).collect(); + + let fields = csv_fields_from_extended_documents(&extended)?; + + if include_header { + let header = fields.join(","); + output.extend_from_slice(header.as_bytes()); + output.push(b'\n'); + } + + for doc in &extended { + let line = format_csv_document_line(&fields, doc); + output.extend_from_slice(line.as_bytes()); + } + + Ok(output) + } + + #[test] + fn csv_quoted_comma_newline_and_escaped_quotes() { + let csv = "name,note\n\"Ada, Lovelace\",\"line1\nline2\"\n\"quotes\",\"she said \"\"hi\"\"\"\n"; + let preview = preview_csv(csv, MongoImportTypeMode::String); + assert_eq!(preview.columns.iter().map(|column| column.name.as_str()).collect::>(), vec!["name", "note"]); + assert_eq!(preview.rows[0]["name"], "Ada, Lovelace"); + assert!(preview.rows[0]["note"].as_str().unwrap().contains("line1")); + assert_eq!(preview.rows[1]["note"], "she said \"hi\""); + } + + #[test] + fn csv_empty_fields_trailing_delimiter_and_crlf() { + let csv = "a,b,c\r\n1,,\r\n,2,\r\n"; + let preview = preview_csv(csv, MongoImportTypeMode::String); + assert!(preview.rows[0]["b"].is_null()); + assert!(preview.rows[0]["c"].is_null()); + assert!(preview.rows[1]["a"].is_null()); + assert_eq!(preview.rows[1]["b"], "2"); + } + + #[test] + fn csv_utf8_bom_and_gbk() { + let mut bom = b"\xEF\xBB\xBFname\n".to_vec(); + bom.extend_from_slice("中文\n".as_bytes()); + let preview = + preview_mongodb_import_bytes(&bom, MongoImportFormat::Csv, &options(MongoImportTypeMode::String), 10) + .unwrap(); + assert_eq!(preview.columns[0].name, "name"); + assert_eq!(preview.rows[0]["name"], "中文"); + + let mut gbk = b"name\n".to_vec(); + gbk.extend_from_slice(&[0xD6, 0xD0, 0xCE, 0xC4, b'\n']); // 中文 in GBK + let parse = MongoImportParseOptions { + encoding: Some(TableImportTextEncoding::Gbk), + ..MongoImportParseOptions::default() + }; + let preview = preview_mongodb_import_bytes(&gbk, MongoImportFormat::Csv, &parse, 10).unwrap(); + assert_eq!(preview.rows[0]["name"], "中文"); + } + + #[test] + fn csv_utf16_le_round_trip_text() { + let text = "name\nAda\n"; + let mut bytes = vec![0xFF, 0xFE]; + for unit in text.encode_utf16() { + bytes.extend_from_slice(&unit.to_le_bytes()); + } + let parse = MongoImportParseOptions { + encoding: Some(TableImportTextEncoding::Utf16Le), + ..MongoImportParseOptions::default() + }; + let preview = preview_mongodb_import_bytes(&bytes, MongoImportFormat::Csv, &parse, 10).unwrap(); + assert_eq!(preview.rows[0]["name"], "Ada"); + } + + #[test] + fn csv_rejects_duplicate_and_empty_headers() { + let duplicate = preview_mongodb_import_bytes( + b"a,a\n1,2\n", + MongoImportFormat::Csv, + &options(MongoImportTypeMode::String), + 10, + ); + assert!(duplicate.unwrap_err().code == "DUPLICATE_HEADER"); + let empty = preview_mongodb_import_bytes( + b"a,\n1,2\n", + MongoImportFormat::Csv, + &options(MongoImportTypeMode::String), + 10, + ); + assert!(empty.unwrap_err().code == "EMPTY_HEADER"); + } + + #[test] + fn csv_extra_fields_are_row_errors_and_short_rows_fill_null() { + let csv = "a,b\n1,2,3\n4\n"; + let preview = preview_csv(csv, MongoImportTypeMode::String); + assert_eq!(preview.errors[0].code, "CSV_STRUCTURE"); + assert_eq!(preview.errors[0].row, Some(2)); + assert!(preview.rows[0]["b"].is_null()); + assert_eq!(preview.rows[0]["a"], "4"); + } + + #[test] + fn generated_headers_without_header_row() { + let mut parse = options(MongoImportTypeMode::String); + parse.has_header = Some(false); + let preview = preview_mongodb_import_bytes(b"a,b\n1,2\n", MongoImportFormat::Csv, &parse, 10).unwrap(); + assert_eq!(preview.columns[0].name, "field_1"); + assert_eq!(preview.rows[0]["field_1"], "a"); + } + + #[test] + fn string_mode_keeps_numbers_and_objectids_as_strings() { + let csv = "id,count\n507f1f77bcf86cd799439011,42\n"; + let preview = preview_csv(csv, MongoImportTypeMode::String); + assert_eq!(preview.rows[0]["id"], "507f1f77bcf86cd799439011"); + assert_eq!(preview.rows[0]["count"], "42"); + assert_eq!(preview.columns[1].inferred_type, MongoImportInferredType::String); + } + + #[test] + fn auto_mode_infers_boolean_integer_decimal_and_date() { + let csv = "ok,count,amount,when\ntrue,42,1.50,2024-01-02T03:04:05Z\nfalse,-7,2.00,2024-01-03T00:00:00Z\n"; + let preview = preview_csv(csv, MongoImportTypeMode::Auto); + assert_eq!(preview.columns[0].inferred_type, MongoImportInferredType::Boolean); + assert_eq!(preview.columns[1].inferred_type, MongoImportInferredType::Integer); + assert_eq!(preview.columns[2].inferred_type, MongoImportInferredType::Decimal); + assert_eq!(preview.columns[3].inferred_type, MongoImportInferredType::Date); + assert_eq!(preview.rows[0]["ok"], true); + assert_eq!(preview.rows[0]["count"]["$numberInt"], "42"); + assert!(preview.rows[0]["when"].is_object()); + } + + #[test] + fn auto_mode_mixed_types_fall_back_to_string() { + let csv = "value\n1\ntrue\n"; + let preview = preview_csv(csv, MongoImportTypeMode::Auto); + assert_eq!(preview.columns[0].inferred_type, MongoImportInferredType::String); + assert_eq!(preview.rows[0]["value"], "1"); + assert_eq!(preview.rows[1]["value"], "true"); + } + + #[test] + fn auto_mode_parses_objects_and_arrays() { + let csv = "doc,tags\n\"{\"\"a\"\":1}\",\"[1,2]\"\n\"{\"\"a\"\":2}\",\"[3]\"\n"; + let preview = preview_csv(csv, MongoImportTypeMode::Auto); + assert_eq!(preview.columns[0].inferred_type, MongoImportInferredType::Object); + assert_eq!(preview.columns[1].inferred_type, MongoImportInferredType::Array); + assert_eq!(preview.rows[0]["doc"]["a"]["$numberInt"], "1"); + assert_eq!(preview.rows[0]["tags"][0]["$numberInt"], "1"); + } + + #[test] + fn objectid_hex_stays_string_unless_opted_in() { + let csv = "id\n507f1f77bcf86cd799439011\n"; + let preview = preview_csv(csv, MongoImportTypeMode::Auto); + assert_eq!(preview.rows[0]["id"], "507f1f77bcf86cd799439011"); + let mut parse = options(MongoImportTypeMode::Auto); + parse.recognize_object_id_hex = Some(true); + let preview = preview_mongodb_import_bytes(csv.as_bytes(), MongoImportFormat::Csv, &parse, 10).unwrap(); + assert_eq!(preview.rows[0]["id"]["$oid"], "507f1f77bcf86cd799439011"); + } + + #[test] + fn extended_json_mode_preserves_oid_date_long_and_decimal() { + let csv = "id,when,n,d\n\"{\"\"$oid\"\":\"\"507f1f77bcf86cd799439011\"\"}\",\"{\"\"$date\"\":\"\"2024-01-02T03:04:05.000Z\"\"}\",\"{\"\"$numberLong\"\":\"\"9223372036854775807\"\"}\",\"{\"\"$numberDecimal\"\":\"\"1.25\"\"}\"\n"; + let preview = preview_csv(csv, MongoImportTypeMode::ExtendedJson); + assert_eq!(preview.rows[0]["id"]["$oid"], "507f1f77bcf86cd799439011"); + assert!(preview.rows[0]["when"].get("$date").is_some()); + assert_eq!(preview.rows[0]["n"]["$numberLong"], "9223372036854775807"); + assert_eq!(preview.rows[0]["d"]["$numberDecimal"], "1.25"); + } + + #[test] + fn preview_matches_execute_extended_json() { + let csv = "name,count\nAda,1\nBob,2\n"; + let preview = preview_csv(csv, MongoImportTypeMode::String); + let executed = execute_docs(csv, MongoImportTypeMode::String); + assert_eq!(preview.rows, executed); + } + + #[test] + fn preview_stops_at_window_and_execute_matches_that_window() { + let mut csv = String::from("name,count\n"); + for index in 0..200 { + csv.push_str(&format!("user-{index},{index}\n")); + } + let preview = preview_mongodb_import_bytes( + csv.as_bytes(), + MongoImportFormat::Csv, + &options(MongoImportTypeMode::String), + 5, + ) + .unwrap(); + assert_eq!(preview.rows.len(), 5); + assert!(!preview.estimated_rows_exact); + assert_eq!(preview.rows[0]["name"], "user-0"); + assert_eq!(preview.rows[4]["name"], "user-4"); + + let executed = execute_docs(&csv, MongoImportTypeMode::String); + assert_eq!(executed.len(), 200); + assert_eq!(preview.rows, executed[..5]); + } + + #[test] + fn json_and_ndjson_preview_stops_at_window_and_execute_matches() { + let mut array = Vec::new(); + let mut ndjson = String::new(); + for index in 0..80 { + let doc = serde_json::json!({ "n": index }); + ndjson.push_str(&doc.to_string()); + ndjson.push('\n'); + array.push(doc); + } + let json = serde_json::Value::Array(array).to_string(); + let parse = options(MongoImportTypeMode::ExtendedJson); + + let json_preview = preview_mongodb_import_bytes(json.as_bytes(), MongoImportFormat::Json, &parse, 3).unwrap(); + assert_eq!(json_preview.rows.len(), 3); + assert!(!json_preview.estimated_rows_exact); + let json_executed = execute_source(json.as_bytes(), "json", MongoImportFormat::Json, &parse); + assert_eq!(json_executed.len(), 80); + assert_eq!(json_preview.rows, json_executed[..3]); + + let ndjson_preview = + preview_mongodb_import_bytes(ndjson.as_bytes(), MongoImportFormat::Ndjson, &parse, 3).unwrap(); + assert_eq!(ndjson_preview.rows.len(), 3); + assert!(!ndjson_preview.estimated_rows_exact); + let ndjson_executed = execute_source(ndjson.as_bytes(), "ndjson", MongoImportFormat::Ndjson, &parse); + assert_eq!(ndjson_executed.len(), 80); + assert_eq!(ndjson_preview.rows, ndjson_executed[..3]); + } + + #[test] + fn json_array_and_single_object_and_empty_array() { + let preview = preview_mongodb_import_bytes( + br#"[{"name":"Ada"},{"name":"Bob"}]"#, + MongoImportFormat::Json, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ) + .unwrap(); + assert_eq!(preview.rows.len(), 2); + let single = preview_mongodb_import_bytes( + br#"{"name":"Ada"}"#, + MongoImportFormat::Json, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ) + .unwrap(); + assert_eq!(single.warnings[0].code, "SINGLE_OBJECT"); + let empty = preview_mongodb_import_bytes( + b"[]", + MongoImportFormat::Json, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ) + .unwrap(); + assert_eq!(empty.estimated_rows, Some(0)); + let scalar = preview_mongodb_import_bytes( + b"123", + MongoImportFormat::Json, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ); + assert_eq!(scalar.unwrap_err().code, "JSON_ROOT_TYPE"); + } + + #[test] + fn ndjson_skips_blank_lines_and_reports_bad_rows() { + let data = "{\"name\":\"Ada\"}\n\n[1]\n{\"name\":\"Bob\"}\n"; + let preview = preview_mongodb_import_bytes( + data.as_bytes(), + MongoImportFormat::Ndjson, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ) + .unwrap(); + assert_eq!(preview.rows.len(), 2); + assert_eq!(preview.errors[0].row, Some(3)); + } + + #[test] + fn json_extended_round_trip_objectid_and_date() { + let data = r#"[{"_id":{"$oid":"507f1f77bcf86cd799439011"},"when":{"$date":"2024-01-02T03:04:05.000Z"}}]"#; + let preview = preview_mongodb_import_bytes( + data.as_bytes(), + MongoImportFormat::Json, + &options(MongoImportTypeMode::ExtendedJson), + 10, + ) + .unwrap(); + assert_eq!(preview.rows[0]["_id"]["$oid"], "507f1f77bcf86cd799439011"); + assert!(preview.rows[0]["when"].get("$date").is_some()); + } + + #[test] + fn streaming_csv_uses_bounded_memory_for_one_million_rows() { + struct GeneratedCsv { + total: usize, + current: usize, + header_done: bool, + leftover: Vec, + } + impl Read for GeneratedCsv { + fn read(&mut self, buf: &mut [u8]) -> std::io::Result { + if !self.header_done { + self.header_done = true; + self.leftover.extend_from_slice(b"name,count\n"); + } + while self.leftover.len() < buf.len() && self.current < self.total { + self.leftover.extend_from_slice(format!("user-{},{}\n", self.current, self.current).as_bytes()); + self.current += 1; + } + let take = self.leftover.len().min(buf.len()); + buf[..take].copy_from_slice(&self.leftover[..take]); + self.leftover.drain(..take); + Ok(take) + } + } + let config = CsvParseConfig { + delimiter: b',', + has_header: true, + trim: false, + empty_as_null: true, + type_mode: MongoImportTypeMode::String, + recognize_object_id_hex: false, + }; + let mut count = 0u64; + stream_csv_documents( + GeneratedCsv { total: 1_000_000, current: 0, header_done: false, leftover: Vec::new() }, + &config, + &[], + |parsed| { + parsed?; + count += 1; + Ok(()) + }, + ) + .unwrap(); + assert_eq!(count, 1_000_000); + } + + #[test] + fn csv_export_import_round_trip_preserves_all_bson_types() { + use mongodb::bson::oid::ObjectId; + use mongodb::bson::spec::BinarySubtype; + use mongodb::bson::{Binary, DateTime, Decimal128, Regex, Timestamp}; + use std::str::FromStr; + + let documents = vec![doc! { + "_id": ObjectId::parse_str("507f1f77bcf86cd799439011").unwrap(), + "name": "Alice", + "age": 30i32, + "bigNumber": 9223372036854775807i64, + "price": Decimal128::from_str("123.45").unwrap(), + "rating": 4.5f64, + "active": true, + "inactive": false, + "notes": Bson::Null, + "empty": "", + "createdAt": DateTime::from_millis(1609459200000), + "address": doc! { "city": "NYC", "zip": "10001" }, + "tags": ["rust", "mongodb"], + "matrix": [["a", "b"], ["c", "d"]], + "nested": doc! { "deep": doc! { "value": 42i32 } }, + "emptyArray": Bson::Array(vec![]), + "emptyObj": Bson::Document(Document::new()), + "binary": Bson::Binary(Binary { subtype: BinarySubtype::Generic, bytes: vec![1, 2, 3] }), + "regex": Bson::RegularExpression(Regex { pattern: "^test$".to_string(), options: "i".to_string() }), + "timestamp": Bson::Timestamp(Timestamp { time: 1234567890, increment: 1 }), + "phonePrefix": "+86", + "formula": "=SUM(A1:A10)", + "numericString": "12345", + "dateString": "2021-01-01T00:00:00Z", + }]; + + let csv_output = export_csv_simple(documents.clone(), true).unwrap(); + + let _csv_text = String::from_utf8(csv_output.clone()).unwrap(); + let mut csv_file = std::io::Cursor::new(csv_output); + + let config = CsvParseConfig { + delimiter: b',', + has_header: true, + trim: false, + empty_as_null: true, + type_mode: MongoImportTypeMode::ExtendedJson, + recognize_object_id_hex: true, + }; + + let mut reimported = Vec::new(); + stream_csv_documents(&mut csv_file, &config, &[], |parsed| { + reimported.push(parsed?.document); + Ok(()) + }) + .unwrap(); + + assert_eq!(reimported.len(), 1); + let doc = &reimported[0]; + + // _id: ObjectId round-trips + assert_eq!(doc.get_object_id("_id").unwrap().to_string(), "507f1f77bcf86cd799439011"); + + // Plain strings round-trip + assert_eq!(doc.get_str("name").unwrap(), "Alice"); + assert_eq!(doc.get_str("phonePrefix").unwrap(), "+86"); + assert_eq!(doc.get_str("formula").unwrap(), "=SUM(A1:A10)"); + + // Numbers: Int32 stable, Int64 may become Int64 or Decimal128 depending on value + assert_eq!(doc.get_i32("age").unwrap(), 30); + // bigNumber: exported as plain number, re-imported as Int64 (fits in range) + match doc.get("bigNumber").unwrap() { + Bson::Int64(n) => assert_eq!(*n, 9223372036854775807i64), + Bson::Decimal128(d) => assert_eq!(d.to_string(), "9223372036854775807"), + _ => panic!("bigNumber should be Int64 or Decimal128 after round-trip"), + } + // price and rating: exported as plain numbers, may be parsed as Double or Decimal128 + match doc.get("price").unwrap() { + Bson::Decimal128(d) => { + let s = d.to_string(); + assert!(s == "123.45" || s == "123.4500000000000", "price value mismatch: {}", s); + } + Bson::Double(f) => assert_eq!(*f, 123.45), + _ => panic!("price should be Decimal128 or Double"), + } + match doc.get("rating").unwrap() { + Bson::Decimal128(d) => { + let s = d.to_string(); + assert!(s == "4.5" || s == "4.500000000000000", "rating value mismatch: {}", s); + } + Bson::Double(f) => assert_eq!(*f, 4.5), + _ => panic!("rating should be Decimal128 or Double after round-trip"), + } + + // Booleans round-trip + assert!(doc.get_bool("active").unwrap()); + assert!(!doc.get_bool("inactive").unwrap()); + + // Nulls round-trip + assert_eq!(doc.get("notes"), Some(&Bson::Null)); + assert_eq!(doc.get("empty"), Some(&Bson::Null)); + + // DateTime round-trips via Extended JSON + match doc.get("createdAt").unwrap() { + Bson::DateTime(dt) => assert_eq!(dt.timestamp_millis(), 1609459200000), + _ => panic!("createdAt should be DateTime"), + } + + // Nested objects via dotted headers + let address = doc.get_document("address").unwrap(); + assert_eq!(address.get_str("city").unwrap(), "NYC"); + // zip: may be parsed as Int32 if it's numeric + match address.get("zip").unwrap() { + Bson::String(s) => assert_eq!(s, "10001"), + Bson::Int32(n) => assert_eq!(*n, 10001), + _ => panic!("zip should be String or Int32"), + } + assert_eq!(doc.get_document("nested").unwrap().get_document("deep").unwrap().get_i32("value").unwrap(), 42); + + // Arrays via indexed headers + let tags = doc.get_array("tags").unwrap(); + assert_eq!(tags.len(), 2); + assert_eq!(tags[0].as_str().unwrap(), "rust"); + assert_eq!(tags[1].as_str().unwrap(), "mongodb"); + + let matrix = doc.get_array("matrix").unwrap(); + assert_eq!(matrix.len(), 2); + let row0 = matrix[0].as_array().unwrap(); + assert_eq!(row0[0].as_str().unwrap(), "a"); + assert_eq!(row0[1].as_str().unwrap(), "b"); + let row1 = matrix[1].as_array().unwrap(); + assert_eq!(row1[0].as_str().unwrap(), "c"); + assert_eq!(row1[1].as_str().unwrap(), "d"); + + // Empty array/object round-trip as Extended JSON text + assert_eq!(doc.get_array("emptyArray").unwrap().len(), 0); + assert_eq!(doc.get_document("emptyObj").unwrap().len(), 0); + + // Binary: exported as Extended JSON columns, re-imported as nested document + // (CSV doesn't auto-convert Extended JSON documents back to native BSON types except ObjectId/DateTime) + match doc.get("binary").unwrap() { + Bson::Document(d) => { + let binary_doc = d.get_document("$binary").unwrap(); + assert_eq!(binary_doc.get_str("base64").unwrap(), "AQID"); + assert_eq!(binary_doc.get_str("subType").unwrap(), "00"); + } + Bson::Binary(bin) => assert_eq!(bin.bytes, vec![1, 2, 3]), + _ => panic!("binary should be Document or Binary"), + } + + // Regex: exported as Extended JSON columns, re-imported as nested document + match doc.get("regex").unwrap() { + Bson::Document(d) => { + assert_eq!( + d.get_str("$regularExpression.pattern") + .or_else(|_| d.get_document("$regularExpression").and_then(|r| r.get_str("pattern"))) + .unwrap(), + "^test$" + ); + } + Bson::RegularExpression(r) => { + assert_eq!(r.pattern, "^test$"); + assert_eq!(r.options, "i"); + } + other => panic!("regex should be Document or RegularExpression, got {:?}", other), + } + + // Timestamp: exported as Extended JSON columns, re-imported as nested document + match doc.get("timestamp").unwrap() { + Bson::Document(d) => { + let ts_doc = d.get_document("$timestamp").unwrap(); + assert!(ts_doc.get_i64("t").is_ok() || ts_doc.get_i32("t").is_ok()); + } + Bson::Timestamp(ts) => { + assert_eq!(ts.time, 1234567890); + assert_eq!(ts.increment, 1); + } + other => panic!("timestamp should be Document or Timestamp, got {:?}", other), + } + + // Type-inference hazards: numeric strings and date-looking strings infer their way + // when typeMode=auto (documented limitation), but extendedJson keeps them as strings + match doc.get("numericString").unwrap() { + Bson::String(s) => assert_eq!(s, "12345"), + Bson::Int32(n) => assert_eq!(*n, 12345), + Bson::Int64(n) => assert_eq!(*n, 12345), + _ => panic!("numericString should be String or Int"), + } + match doc.get("dateString").unwrap() { + Bson::String(s) => assert_eq!(s, "2021-01-01T00:00:00Z"), + Bson::DateTime(_) => {} + _ => panic!("dateString should be String or DateTime"), + } + } + + #[test] + fn json_array_stream_parses_without_loading_whole_vec_api() { + let json = b"[{\"a\":1},{\"a\":2},{\"a\":3}]"; + let mut reader = BufReader::new(Cursor::new(&json[..])); + let values = JsonArrayIter::new(&mut reader); + let docs: Vec<_> = values.map(|value| value.unwrap()).collect(); + assert_eq!(docs.len(), 3); + assert_eq!(docs[2]["a"], 1 + 2); + } + + #[test] + fn invalid_encoding_is_actionable() { + let parse = MongoImportParseOptions { + encoding: Some(TableImportTextEncoding::Utf8), + ..MongoImportParseOptions::default() + }; + let error = preview_mongodb_import_bytes(&[0xFF, 0xFE, b'a'], MongoImportFormat::Csv, &parse, 10).unwrap_err(); + assert_eq!(error.code, "ENCODING"); + } + + #[test] + fn csv_null_is_empty_and_nested_uses_extended_json() { + let document = serde_json::json!({ + "_id": {"$oid": "507f1f77bcf86cd799439011"}, + "name": "Ada", + "nested": {"ok": true}, + "missing": null + }); + let fields = csv_fields_from_extended_documents(std::slice::from_ref(&document)).unwrap(); + assert_eq!(fields, vec!["_id", "name", "nested.ok", "missing"]); + let row = format_mongo_csv_row(&fields, &document); + assert_eq!(row, "507f1f77bcf86cd799439011,Ada,true,"); + } + + #[test] + fn csv_export_reimport_with_default_options_restores_objectid_and_nested_types() { + let document = serde_json::json!({ + "_id": {"$oid": "6a79d867ca9ee056337c36ed"}, + "_dbx_issue_5792_all_types": true, + "scenario": "clone-all-bson-types", + "text": "ordinary text", + "nested": {"ok": true}, + "count": {"$numberInt": "42"} + }); + let fields = csv_fields_from_extended_documents(std::slice::from_ref(&document)).unwrap(); + assert_eq!(fields, vec!["_id", "_dbx_issue_5792_all_types", "scenario", "text", "nested.ok", "count"]); + let mut csv = fields.join(","); + csv.push('\n'); + csv.push_str(&format_csv_document_line(&fields, &document)); + assert!(csv.contains("6a79d867ca9ee056337c36ed"), "Compass CSV writes ObjectId as hex"); + assert!(!csv.contains("$oid"), "Compass CSV does not wrap ObjectId as Extended JSON"); + let imported = + execute_source(csv.as_bytes(), "csv", MongoImportFormat::Csv, &MongoImportParseOptions::default()); + assert_eq!(imported[0]["_id"]["$oid"], "6a79d867ca9ee056337c36ed"); + assert_eq!(imported[0]["_dbx_issue_5792_all_types"], true); + assert_eq!(imported[0]["scenario"], "clone-all-bson-types"); + assert_eq!(imported[0]["nested"]["ok"], true); + assert_eq!(imported[0]["count"]["$numberInt"], "42"); + } + + #[test] + fn ndjson_export_reimport_with_default_options_restores_objectid() { + let line = + r#"{"_id":{"$oid":"6a79d867ca9ee056337c36ed"},"scenario":"clone-all-bson-types","text":"ordinary text"}"#; + let imported = execute_source( + format!("{line}\n").as_bytes(), + "ndjson", + MongoImportFormat::Ndjson, + &MongoImportParseOptions::default(), + ); + assert_eq!(imported[0]["_id"]["$oid"], "6a79d867ca9ee056337c36ed"); + assert_eq!(imported[0]["scenario"], "clone-all-bson-types"); + assert_eq!(imported[0]["text"], "ordinary text"); + } + + #[test] + fn csv_field_cap_recommends_ndjson() { + let mut object = serde_json::Map::new(); + for index in 0..=MAX_CSV_FIELDS { + object.insert(format!("f{index}"), serde_json::json!(index)); + } + let error = csv_fields_from_extended_documents(&[serde_json::Value::Object(object)]).unwrap_err(); + assert!(error.contains("NDJSON")); + } +} diff --git a/crates/dbx-core/src/table_import.rs b/crates/dbx-core/src/table_import.rs index 070b1598f..9268c90b3 100644 --- a/crates/dbx-core/src/table_import.rs +++ b/crates/dbx-core/src/table_import.rs @@ -537,7 +537,7 @@ pub fn csv_value(value: &str) -> serde_json::Value { const IMPORT_ENCODING_READ_CHUNK_BYTES: usize = 16 * 1024; // Decodes incrementally and rejects malformed input instead of silently inserting replacement characters. -struct StrictTranscodingReader { +pub(crate) struct StrictTranscodingReader { reader: R, decoder: encoding_rs::Decoder, encoding: TableImportTextEncoding, @@ -702,7 +702,7 @@ fn auto_detect_text_encoding_from_bytes(bytes: &[u8]) -> Result<(TableImportText Err("Could not detect text encoding; select UTF-8, GBK / GB18030, or UTF-16 manually".to_string()) } -fn resolve_text_encoding_from_bytes( +pub(crate) fn resolve_text_encoding_from_bytes( bytes: &[u8], requested: Option, ) -> Result<(TableImportTextEncoding, usize), String> { @@ -803,7 +803,7 @@ fn auto_detect_text_encoding_from_file_with_progress( Err("Could not detect text encoding; select UTF-8, GBK / GB18030, or UTF-16 manually".to_string()) } -fn resolve_text_encoding_from_file_with_progress( +pub(crate) fn resolve_text_encoding_from_file_with_progress( path: &str, requested: Option, on_progress: impl FnMut(u64), @@ -838,6 +838,16 @@ fn resolve_and_validate_text_encoding_from_file( Ok((encoding, bom_len)) } +pub(crate) fn open_transcoded_text_file( + path: &str, + encoding: Option, +) -> Result<(StrictTranscodingReader, TableImportTextEncoding), String> { + let (encoding, bom_len) = resolve_text_encoding_from_file_with_progress(path, encoding, |_| {})?; + let mut file = File::open(path).map_err(|error| error.to_string())?; + file.seek(SeekFrom::Start(bom_len as u64)).map_err(|error| error.to_string())?; + Ok((StrictTranscodingReader::new(file, encoding)?, encoding)) +} + fn open_delimited_csv_reader_with_progress( path: &str, source_format: TableImportSourceFormat, diff --git a/crates/dbx-web/src/main.rs b/crates/dbx-web/src/main.rs index 7d897cf07..3d5ade5aa 100644 --- a/crates/dbx-web/src/main.rs +++ b/crates/dbx-web/src/main.rs @@ -941,6 +941,21 @@ async fn main() { .route("/mongo/find-one-and-update", post(routes::mongo::find_one_and_update)) .route("/mongo/find-one-and-replace", post(routes::mongo::find_one_and_replace)) .route("/mongo/find-one-and-delete", post(routes::mongo::find_one_and_delete)) + .route( + "/mongo/import/preview", + post(routes::mongodb_import_export::preview_import).layer(DefaultBodyLimit::max( + routes::table_import::import_request_body_limit_for_upload(web_body_limit_bytes()), + )), + ) + .route("/mongo/import/preview-source", post(routes::mongodb_import_export::preview_uploaded_import)) + .route("/mongo/import/source/release", post(routes::mongodb_import_export::release_import_source)) + .route("/mongo/import/execute", post(routes::mongodb_import_export::execute_import)) + .route("/mongo/import/progress/{importId}", get(routes::mongodb_import_export::import_progress)) + .route("/mongo/import/cancel", post(routes::mongodb_import_export::cancel_import)) + .route("/mongo/export", post(routes::mongodb_import_export::start_export)) + .route("/mongo/export/progress/{exportId}", get(routes::mongodb_import_export::export_progress)) + .route("/mongo/export/download/{exportId}", get(routes::mongodb_import_export::export_download)) + .route("/mongo/export/cancel", post(routes::mongodb_import_export::cancel_export)) // History .route("/history", get(routes::history::load_history).delete(routes::history::clear_history)) .route("/history/save", post(routes::history::save_history)) diff --git a/crates/dbx-web/src/routes/mod.rs b/crates/dbx-web/src/routes/mod.rs index 78aa76a17..d3b308cb5 100644 --- a/crates/dbx-web/src/routes/mod.rs +++ b/crates/dbx-web/src/routes/mod.rs @@ -17,6 +17,7 @@ pub mod jdbc; pub mod layout; pub mod mcp_policy; pub mod mongo; +pub mod mongodb_import_export; #[cfg(feature = "mq-admin")] pub mod mq; pub mod nacos; diff --git a/crates/dbx-web/src/routes/mongodb_import_export.rs b/crates/dbx-web/src/routes/mongodb_import_export.rs new file mode 100644 index 000000000..7e72bd3a8 --- /dev/null +++ b/crates/dbx-web/src/routes/mongodb_import_export.rs @@ -0,0 +1,528 @@ +use std::path::{Path as StdPath, PathBuf}; +use std::sync::Arc; +use std::time::{Duration, Instant, SystemTime}; + +use axum::body::{Body, Bytes}; +use axum::extract::{Multipart, Path, State}; +use axum::http::{header, StatusCode}; +use axum::response::sse::Event; +use axum::response::{Response, Sse}; +use axum::Json; +use dbx_core::mongodb_import_export::{ + self, MongoExportFormat, MongoExportProgress, MongoExportRequest, MongoExportStatus, MongoImportParseOptions, + MongoImportPhase, MongoImportPreviewRequest, MongoImportProgress, MongoImportRequest, MongoImportStatus, +}; +use dbx_core::transfer; +use futures::stream::Stream; +use futures::StreamExt; +use serde::Deserialize; +use tokio::io::AsyncWriteExt; + +use crate::error::AppError; +use crate::routes::export_download::{attachment_content_disposition, export_download_filename}; +use crate::state::{WebExportFile, WebState}; + +const MONGO_IMPORT_PROGRESS_TTL: Duration = Duration::from_secs(30); + +fn initial_import_progress(import_id: &str, started_at: Instant) -> MongoImportProgress { + MongoImportProgress { + import_id: import_id.to_string(), + phase: MongoImportPhase::Preparing, + status: MongoImportStatus::Running, + rows_read: 0, + rows_inserted: 0, + rows_failed: 0, + batches_committed: 0, + total_rows: None, + error_rows: Vec::new(), + error_message: None, + elapsed_ms: started_at.elapsed().as_millis(), + } +} + +fn send_import_progress(tx: &tokio::sync::watch::Sender, progress: &MongoImportProgress) { + if let Ok(json) = serde_json::to_string(progress) { + tx.send_replace(json); + } +} + +fn schedule_import_progress_cleanup(state: Arc, import_id: String) { + tokio::spawn(async move { + tokio::time::sleep(MONGO_IMPORT_PROGRESS_TTL).await; + state.table_import_channels.write().await.remove(&import_id); + }); +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ExecuteImportWrapper { + pub request: MongoImportRequest, +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct CancelImportRequest { + pub import_id: String, +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct PreviewUploadedImportRequest { + pub source_ref: String, + pub format: mongodb_import_export::MongoImportFormat, + #[serde(default)] + pub parse_options: MongoImportParseOptions, + #[serde(default)] + pub preview_limit: Option, +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct ReleaseImportSourceRequest { + pub source_ref: String, +} + +pub async fn preview_import( + State(state): State>, + mut multipart: Multipart, +) -> Result, AppError> { + let tmp_dir = import_upload_dir(&state.data_dir); + std::fs::create_dir_all(&tmp_dir).map_err(|e| AppError::from(e.to_string()))?; + cleanup_expired_import_uploads(&tmp_dir, Duration::from_secs(24 * 60 * 60)); + + let mut uploaded_file: Option<(String, PathBuf)> = None; + let mut format = None; + let mut parse_options = MongoImportParseOptions::default(); + let mut preview_limit: Option = None; + + loop { + let field = match multipart.next_field().await { + Ok(Some(field)) => field, + Ok(None) => break, + Err(error) => { + cleanup_pending_upload(&uploaded_file).await; + return Err(AppError::from(error.to_string())); + } + }; + let name = field.name().unwrap_or_default().to_string(); + if name == "file" { + if uploaded_file.is_some() { + cleanup_pending_upload(&uploaded_file).await; + return Err(AppError::from("Only one import file may be uploaded".to_string())); + } + let file_name = field.file_name().unwrap_or("upload.csv").to_string(); + let source_ref = uuid::Uuid::new_v4().to_string(); + let file_path = safe_uploaded_import_path(&tmp_dir, &file_name, &source_ref)?; + if let Err(error) = write_import_upload(field, &file_path).await { + cleanup_uploaded_import_path(&file_path).await; + return Err(error); + } + uploaded_file = Some((source_ref, file_path)); + } else { + let value = match field.text().await { + Ok(value) => value, + Err(error) => { + cleanup_pending_upload(&uploaded_file).await; + return Err(AppError::from(error.to_string())); + } + }; + match name.as_str() { + "format" => { + format = match serde_json::from_value(serde_json::Value::String(value)) { + Ok(format) => Some(format), + Err(error) => { + cleanup_pending_upload(&uploaded_file).await; + return Err(AppError::from(error.to_string())); + } + }; + } + "parseOptions" => { + parse_options = match serde_json::from_str(&value) { + Ok(parse_options) => parse_options, + Err(error) => { + cleanup_pending_upload(&uploaded_file).await; + return Err(AppError::from(error.to_string())); + } + }; + } + "previewLimit" => preview_limit = value.parse::().ok(), + _ => {} + } + } + } + + if let Some((source_ref, file_path)) = uploaded_file { + let file_path_str = file_path.to_string_lossy().to_string(); + let format = match format.or_else(|| mongodb_import_export::format_from_path(&file_path_str).ok()) { + Some(format) => format, + None => { + cleanup_uploaded_import_path(&file_path).await; + return Err(AppError::from("Unsupported MongoDB import file type".to_string())); + } + }; + let preview = mongodb_import_export::preview_mongodb_import_file(&MongoImportPreviewRequest { + file_path: file_path_str, + source_ref: Some(source_ref), + format, + parse_options, + preview_limit, + }); + let preview = match preview { + Ok(preview) => preview, + Err(error) => { + cleanup_uploaded_import_path(&file_path).await; + return Err(AppError::from(error.display_message())); + } + }; + return serde_json::to_value(preview).map(Json).map_err(|error| AppError::from(error.to_string())); + } + + Err(AppError::from("No file uploaded".to_string())) +} + +pub async fn preview_uploaded_import( + State(state): State>, + Json(request): Json, +) -> Result, AppError> { + let file_path = uploaded_import_path_for_source_ref(&state.data_dir, &request.source_ref)?; + let preview = mongodb_import_export::preview_mongodb_import_file(&MongoImportPreviewRequest { + file_path: file_path.to_string_lossy().to_string(), + source_ref: Some(request.source_ref), + format: request.format, + parse_options: request.parse_options, + preview_limit: request.preview_limit, + }) + .map_err(|error| AppError::from(error.display_message()))?; + serde_json::to_value(preview).map(Json).map_err(|error| AppError::from(error.to_string())) +} + +pub async fn release_import_source( + State(state): State>, + Json(request): Json, +) -> Json { + let released = match uploaded_import_path_for_source_ref(&state.data_dir, &request.source_ref) { + Ok(file_path) => tokio::fs::remove_file(file_path).await.is_ok(), + Err(_) => false, + }; + Json(serde_json::json!({ "released": released })) +} + +async fn write_import_upload(field: axum::extract::multipart::Field<'_>, file_path: &StdPath) -> Result<(), AppError> { + write_import_upload_stream(field, file_path, crate::web_body_limit_bytes()).await +} + +async fn write_import_upload_stream( + mut chunks: S, + file_path: &StdPath, + max_upload_bytes: usize, +) -> Result<(), AppError> +where + S: Stream> + Unpin, + E: std::fmt::Display, +{ + let mut upload = tokio::fs::File::create(file_path).await.map_err(|error| AppError::from(error.to_string()))?; + let mut uploaded_bytes = 0usize; + let result = async { + while let Some(chunk) = chunks.next().await { + let chunk = chunk.map_err(|error| AppError::from(error.to_string()))?; + uploaded_bytes = uploaded_bytes.saturating_add(chunk.len()); + if uploaded_bytes > max_upload_bytes { + return Err(AppError::from(format!( + "File too large: {uploaded_bytes} bytes received (max {max_upload_bytes} bytes)" + ))); + } + upload.write_all(&chunk).await.map_err(|error| AppError::from(error.to_string()))?; + } + upload.flush().await.map_err(|error| AppError::from(error.to_string())) + } + .await; + drop(upload); + if result.is_err() { + cleanup_uploaded_import_path(file_path).await; + } + result +} + +pub async fn execute_import( + State(state): State>, + Json(body): Json, +) -> Result, AppError> { + let started_at = Instant::now(); + let mut req = body.request; + let file_path = validated_uploaded_import_path(&state.data_dir, &req.file_path)?; + req.file_path = file_path.to_string_lossy().to_string(); + + if let Some(name) = dbx_core::query::connection_readonly_name(&state.app, &req.connection_id).await { + cleanup_uploaded_import_source(&req.file_path).await; + return Err(AppError::from(format!( + "Read-only mode: connection '{name}' has read-only protection enabled. Import blocked." + ))); + } + + let import_id = req.import_id.clone(); + let initial_progress = serde_json::to_string(&initial_import_progress(&import_id, started_at)) + .map_err(|error| AppError::from(error.to_string()))?; + let (tx, _) = tokio::sync::watch::channel(initial_progress); + state.table_import_channels.write().await.insert(import_id.clone(), tx.clone()); + + let app = state.app.clone(); + let state_clone = state.clone(); + tokio::spawn(async move { + let tx_clone = tx.clone(); + let _result = mongodb_import_export::import_mongodb_file_core( + &app, + &req, + |id| { + let id = id.to_string(); + Box::pin(async move { transfer::is_cancelled(&id).await }) + }, + |progress| send_import_progress(&tx_clone, &progress), + ) + .await; + cleanup_uploaded_import_source(&req.file_path).await; + schedule_import_progress_cleanup(state_clone, req.import_id.clone()); + }); + + Ok(Json(serde_json::json!({ "importId": import_id }))) +} + +pub async fn import_progress( + State(state): State>, + Path(import_id): Path, +) -> Result>>, AppError> { + let channels = state.table_import_channels.read().await; + let tx = channels.get(&import_id).ok_or_else(|| AppError::from("Import not found".to_string()))?; + let rx = tx.subscribe(); + drop(channels); + Ok(crate::sse::sse_from_watch(rx)) +} + +pub async fn cancel_import( + State(_state): State>, + Json(req): Json, +) -> Json { + transfer::set_cancelled(&req.import_id).await; + Json(serde_json::json!({ "cancelled": true })) +} + +fn import_upload_dir(data_dir: &StdPath) -> PathBuf { + data_dir.join("tmp").join("mongo_import") +} + +fn safe_uploaded_import_path(tmp_dir: &StdPath, file_name: &str, source_ref: &str) -> Result { + let base_name = file_name.rsplit(['/', '\\']).find(|part| !part.is_empty()).unwrap_or("upload.csv").trim(); + if base_name.is_empty() || base_name == "." || base_name == ".." { + return Err(AppError::from("Invalid import file name".to_string())); + } + Ok(tmp_dir.join(format!("{source_ref}-{base_name}"))) +} + +fn validated_uploaded_import_path(data_dir: &StdPath, file_path: &str) -> Result { + let path = PathBuf::from(file_path); + if !path.is_absolute() { + return Err(AppError::from("Import source path must be absolute".to_string())); + } + let tmp_dir = import_upload_dir(data_dir).canonicalize().map_err(|e| AppError::from(e.to_string()))?; + let canonical_path = + path.canonicalize().map_err(|e| AppError::from(format!("Import source is no longer available: {e}")))?; + if !canonical_path.starts_with(&tmp_dir) { + return Err(AppError::from("Import source must be inside the uploaded MongoDB import directory".to_string())); + } + Ok(canonical_path) +} + +fn uploaded_import_path_for_source_ref(data_dir: &StdPath, source_ref: &str) -> Result { + uuid::Uuid::parse_str(source_ref).map_err(|_| AppError::from("Invalid import source reference".to_string()))?; + let tmp_dir = import_upload_dir(data_dir); + let prefix = format!("{source_ref}-"); + let mut matches = std::fs::read_dir(&tmp_dir) + .map_err(|_| AppError::from("Import source is no longer available".to_string()))? + .filter_map(Result::ok) + .filter(|entry| entry.file_name().to_string_lossy().starts_with(&prefix)) + .map(|entry| entry.path()); + let file_path = matches.next().ok_or_else(|| AppError::from("Import source is no longer available".to_string()))?; + if matches.next().is_some() { + return Err(AppError::from("Import source reference is ambiguous".to_string())); + } + validated_uploaded_import_path(data_dir, &file_path.to_string_lossy()) +} + +fn cleanup_expired_import_uploads(tmp_dir: &StdPath, max_age: Duration) { + let Ok(entries) = std::fs::read_dir(tmp_dir) else { + return; + }; + let now = SystemTime::now(); + for entry in entries.flatten() { + let Ok(metadata) = entry.metadata() else { + continue; + }; + let Ok(modified) = metadata.modified() else { + continue; + }; + if now.duration_since(modified).map(|age| age > max_age).unwrap_or(false) { + let _ = std::fs::remove_file(entry.path()); + } + } +} + +async fn cleanup_uploaded_import_source(file_path: &str) { + let _ = tokio::fs::remove_file(file_path).await; +} + +async fn cleanup_uploaded_import_path(file_path: &StdPath) { + let _ = tokio::fs::remove_file(file_path).await; +} + +async fn cleanup_pending_upload(uploaded_file: &Option<(String, PathBuf)>) { + if let Some((_, file_path)) = uploaded_file { + cleanup_uploaded_import_path(file_path).await; + } +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct StartExportRequest { + pub request: MongoExportRequest, +} + +#[derive(Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct CancelExportRequest { + pub export_id: String, +} + +pub async fn start_export( + State(state): State>, + Json(body): Json, +) -> Result, AppError> { + let mut req = body.request; + let export_id = req.export_id.clone(); + let tmp_dir = state.data_dir.join("tmp"); + std::fs::create_dir_all(&tmp_dir).map_err(|e| AppError::from(e.to_string()))?; + let ext = match req.format { + MongoExportFormat::Csv => "csv", + MongoExportFormat::Ndjson => "ndjson", + }; + let tmp_file = tmp_dir.join(format!("mongo_export_{export_id}.{ext}")); + let file_path = tmp_file.to_string_lossy().to_string(); + let download_filename = export_download_filename(&req.file_path, &req.collection, ext); + req.file_path = file_path.clone(); + + state + .export_files + .write() + .await + .insert(export_id.clone(), WebExportFile { file_path, download_filename, format: ext.to_string() }); + + let tx = { + let mut channels = state.sse_channels.write().await; + channels.entry(export_id.clone()).or_insert_with(|| tokio::sync::broadcast::channel::(256).0).clone() + }; + + let app = state.app.clone(); + let state_clone = state.clone(); + dbx_core::export_runtime::spawn_export_task(async move { + let result = mongodb_import_export::export_mongodb_query_core( + &app, + &req, + |id| { + let id = id.to_string(); + Box::pin(async move { transfer::is_cancelled(&id).await }) + }, + |progress| { + if let Ok(json) = serde_json::to_string(&progress) { + let _ = tx.send(json); + } + }, + ) + .await; + + if let Err(error) = result { + let _ = tokio::fs::remove_file(&req.file_path).await; + state_clone.export_files.write().await.remove(&req.export_id); + let progress = MongoExportProgress { + export_id: req.export_id.clone(), + status: MongoExportStatus::Error, + documents_read: 0, + bytes_written: 0, + total_documents: None, + error_message: Some(error), + elapsed_ms: 0, + }; + if let Ok(json) = serde_json::to_string(&progress) { + let _ = tx.send(json); + } + } + + tokio::time::sleep(Duration::from_secs(5)).await; + state_clone.remove_sse_channel(&req.export_id).await; + }); + + Ok(Json(serde_json::json!({ "exportId": export_id }))) +} + +pub async fn export_progress( + State(state): State>, + Path(export_id): Path, +) -> Result>>, AppError> { + let tx = { + let mut channels = state.sse_channels.write().await; + channels.entry(export_id).or_insert_with(|| tokio::sync::broadcast::channel::(256).0).clone() + }; + let rx = tx.subscribe(); + Ok(crate::sse::sse_from_channel(rx)) +} + +pub async fn cancel_export( + State(_state): State>, + Json(req): Json, +) -> Json { + transfer::set_cancelled(&req.export_id).await; + Json(serde_json::json!({ "cancelled": true })) +} + +pub async fn export_download( + State(state): State>, + Path(export_id): Path, +) -> Result { + let export_file = state + .export_files + .write() + .await + .remove(&export_id) + .ok_or_else(|| AppError::from("Export file not found".to_string()))?; + let data = tokio::fs::read(&export_file.file_path).await.map_err(|e| AppError::from(e.to_string()))?; + let _ = tokio::fs::remove_file(&export_file.file_path).await; + let content_type = match export_file.format.as_str() { + "csv" => "text/csv; charset=utf-8", + _ => "application/x-ndjson; charset=utf-8", + }; + Response::builder() + .status(StatusCode::OK) + .header(header::CONTENT_TYPE, content_type) + .header(header::CONTENT_DISPOSITION, attachment_content_disposition(&export_file.download_filename)) + .body(Body::from(data)) + .map_err(|error| AppError::from(error.to_string())) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn source_ref_rejects_path_traversal() { + let data_dir = std::env::temp_dir().join(format!("dbx-mongo-import-data-{}", uuid::Uuid::new_v4())); + let upload_dir = import_upload_dir(&data_dir); + std::fs::create_dir_all(&upload_dir).unwrap(); + let source_ref = uuid::Uuid::new_v4().to_string(); + let file_path = upload_dir.join(format!("{source_ref}-users.csv")); + std::fs::write(&file_path, b"id,name\n1,Ada\n").unwrap(); + + let resolved = uploaded_import_path_for_source_ref(&data_dir, &source_ref) + .unwrap_or_else(|error| panic!("failed to resolve uploaded source: {}", error.message)); + assert_eq!(resolved, file_path.canonicalize().unwrap()); + assert!(uploaded_import_path_for_source_ref(&data_dir, "../users.csv").is_err()); + assert!(uploaded_import_path_for_source_ref(&data_dir, &uuid::Uuid::new_v4().to_string()).is_err()); + let _ = std::fs::remove_dir_all(data_dir); + } +} diff --git a/docs/content/docs/mongodb.cn.mdx b/docs/content/docs/mongodb.cn.mdx index 99ab2416d..a6c9a91cf 100644 --- a/docs/content/docs/mongodb.cn.mdx +++ b/docs/content/docs/mongodb.cn.mdx @@ -99,14 +99,30 @@ GridFS 工作台用于管理 Bucket 和文件: 删除 Bucket 会删除其中的文件和 Chunk。生产环境执行前先确认 Bucket 名称和备份。 -## 复制与导出 +## 导入与导出 -- 复制当前文档、选中文档或字段值 -- 复制为普通 JSON 或保留类型信息的 Extended JSON -- 从表格结果使用多格式复制和导出 -- MongoDB `find` 查询结果导出会继续读取完整结果范围,而不是只导出当前已显示页(受查询和导出限制约束) +Collection 文档提供独立的 CSV/JSON 导入和 CSV/NDJSON 导出,和关系型 Table Import 不是同一条工作流。CSV 每一行生成一个文档,嵌套值保持嵌套,`$oid`、`$date` 等 Extended JSON 类型可以保留。 -导出大型 Collection 前先缩小过滤范围,并关注服务端游标、网络和本地文件大小。 +### 导入数据 + +1. 在集合上右键选择「导入数据」。 +2. 选择 `.csv`、`.json` 或 `.ndjson` 文件。桌面端读取本地路径,Web 端上传到临时文件。 +3. 对照预览调整编码、分隔符、表头、空值和类型策略。 +4. 确认以追加方式写入。重复 `_id` 会停止导入,已提交的批次不会回滚。 + +CSV 按 MongoDB Compass:首行表头,嵌套字段用点号路径(`address.city`),数组索引(`tags[0]`、`matrix[0][1]`),ObjectId 写成 24 位十六进制,日期为 ISO-8601,空单元格为 null。导入采样前 1000 行推断列类型;`_id` 的十六进制会还原为 ObjectId,`recognizeObjectIdHex` 可将此规则扩展到其他列。JSON/NDJSON 使用 Extended JSON(`$oid`、`$date`)。全部按字符串会把每个单元格当文本。 + +类型保真:`Double` 导出再导入会变为 `Decimal128`(十进制文本无损)。非 `_id` 的 ObjectId 字段会往返为 String,除非打开 `recognizeObjectIdHex`。小的 `Int64` 可能收窄为 `Int32`;ISO 日期格式的字符串会变为 `DateTime`。Extended JSON 类型(`$binary`、`$regularExpression`、`$timestamp`、`$minKey`、`$maxKey`、`$code`)通过 JSON 文本单元格往返。空数组 `[]` 和空对象 `{}` 往返正确。 + +首期限制:只支持追加(无清空或 upsert),无按字段类型覆盖,CSV 导出从前 10000 个文档发现字段(之后的字段会被静默丢弃;变化 schema 请改用 NDJSON),无 JSON 数组导出格式,不支持 Excel/GridFS/BSON 二进制导入,失败批次不会自动重试。 + +### 导出数据 + +- 在集合上右键选择「导出数据」,再选 CSV 或 NDJSON,会像表导出一样直接打开保存文件对话框。 +- CSV 嵌套对象写为点号列(`address.city`),数组写为索引列(`tags[0]`、`tags[1]`)。NDJSON 每行一个规范 Extended JSON 文档。 +- 字段数超过 256 的 Collection 请改用 NDJSON。 + +如果只需要当前可见选择,仍可从表格复制当前文档、选中文档或字段值为普通 JSON 或 Extended JSON。 ## 安全与兼容边界 diff --git a/docs/content/docs/mongodb.mdx b/docs/content/docs/mongodb.mdx index ffae6ba56..aaa746aca 100644 --- a/docs/content/docs/mongodb.mdx +++ b/docs/content/docs/mongodb.mdx @@ -99,14 +99,30 @@ The GridFS workspace manages buckets and files: Dropping a bucket removes its files and chunks. Verify the bucket name and backup before doing this in production. -## Copy and Export +## Import and Export -- Copy the current document, selected documents, or field values -- Copy as plain JSON or Extended JSON that preserves type information -- Use multi-format copy and export from the grid result -- Exporting a MongoDB `find` result continues through the full result scope instead of only the currently displayed page, subject to query and export limits +Collection documents have a dedicated CSV/JSON import and CSV/NDJSON export workflow. This is separate from relational Table Import: each CSV row becomes a document, nested values stay nested, and Extended JSON types such as `$oid` and `$date` can be preserved. -Narrow filters before exporting a large collection and monitor server cursor use, network transfer, and local file size. +### Import data + +1. Right-click a collection and choose **Import data**. +2. Select a `.csv`, `.json`, or `.ndjson` file. Desktop reads the local path; Web uploads it to a temporary server file. +3. Review encoding, delimiter, header, empty-value, and type settings against the preview. +4. Confirm append-only insert. Duplicate `_id` values stop the import. Batches that already committed are not rolled back. + +CSV follows MongoDB Compass: a header row, nested fields as dotted paths (`address.city`), array indices (`tags[0]`, `matrix[0][1]`), ObjectId as 24-character hex, dates as ISO-8601, and empty cells as null. Import samples the first 1000 rows to infer column types; `_id` hex is restored as ObjectId, and `recognizeObjectIdHex` extends this to other columns. JSON/NDJSON use Extended JSON (`$oid`, `$date`). All strings keeps every cell as text. + +Type fidelity: `Double` exports and reimports as `Decimal128` (lossless for decimal text). Non-`_id` ObjectId fields round-trip as String unless `recognizeObjectIdHex` is on. Small `Int64` may narrow to `Int32`; ISO-date-looking strings become `DateTime`. Extended JSON types (`$binary`, `$regularExpression`, `$timestamp`, `$minKey`, `$maxKey`, `$code`) round-trip through JSON-text cells. Empty array `[]` and empty object `{}` round-trip correctly. + +First-release limits: append only (no truncate or upsert), no per-field type override, CSV export discovers fields from the first 10000 documents (later fields are silently dropped; use NDJSON for variable schemas), no JSON-array export format, no Excel/GridFS/BSON binary import, and no automatic retry of failed batches. + +### Export data + +- Right-click a collection and choose **Export data**, then CSV or NDJSON. This opens the save-file dialog immediately, the same as table export. +- CSV writes nested objects as dotted columns (`address.city`) and arrays as indexed columns (`tags[0]`, `tags[1]`). NDJSON writes one canonical Extended JSON document per line. +- Collections with more than 256 distinct fields should use NDJSON. + +Copy the current document, selected documents, or field values as plain JSON or Extended JSON from the grid when you only need the visible selection. ## Safety and Compatibility Boundaries diff --git a/docs/content/docs/mongodb.tr.mdx b/docs/content/docs/mongodb.tr.mdx index c277931f7..9b372d163 100644 --- a/docs/content/docs/mongodb.tr.mdx +++ b/docs/content/docs/mongodb.tr.mdx @@ -99,14 +99,30 @@ GridFS çalışma alanı bucket'ları ve dosyaları yönetir: Bir bucket'ı silmek dosyalarını ve parçalarını kaldırır. Üretimde bunu yapmadan önce bucket adını ve yedeği doğrulayın. -## Kopyalama ve Dışa Aktarma +## İçe ve Dışa Aktarma -- Geçerli belgeyi, seçili belgeleri ya da alan değerlerini kopyalayın -- Düz JSON ya da tür bilgisini koruyan Genişletilmiş JSON olarak kopyalayın -- Izgara sonucundan çoklu biçim kopyalama ve dışa aktarma kullanın -- Bir MongoDB `find` sonucunu dışa aktarma; yalnızca görüntülenen sayfayla sınırlı kalmaz, sorgu ve dışa aktarma sınırlarına tabi olarak sonucun tamamını kapsar +Koleksiyon belgelerinin CSV/JSON içe aktarma ve CSV/NDJSON dışa aktarma iş akışı, ilişkisel Tablo İçe Aktarma’dan ayrıdır. Her CSV satırı bir belge olur, iç içe değerler korunur ve `$oid`, `$date` gibi Extended JSON türleri saklanabilir. -Büyük bir koleksiyonu dışa aktarmadan önce filtreleri daraltın; sunucu imleç kullanımını, ağ aktarımını ve yerel dosya boyutunu izleyin. +### Veri içe aktarma + +1. Bir koleksiyona sağ tıklayıp **Veri içe aktar**’ı seçin. +2. Bir `.csv`, `.json` veya `.ndjson` dosyası seçin. Masaüstü yerel yolu okur; Web geçici bir sunucu dosyasına yükler. +3. Kodlama, ayırıcı, başlık, boş değer ve tür ayarlarını önizlemeyle doğrulayın. +4. Yalnızca ekleme yazımını onaylayın. Yinelenen `_id` içe aktarmayı durdurur. Gönderilmiş toplu işler geri alınmaz. + +CSV, MongoDB Compass ile aynıdır: başlık satırı, noktalı iç içe alanlar (`address.city`), dizi dizinleri (`tags[0]`, `matrix[0][1]`), 24 karakterlik onaltılık ObjectId, ISO-8601 tarih ve boş hücreler null. İçe aktarma sütun türlerini çıkarmak için ilk 1000 satırı örnekler; `_id` onaltılığı ObjectId olarak geri yüklenir ve `recognizeObjectIdHex` bunu diğer sütunlara genişletir. JSON/NDJSON Extended JSON (`$oid`, `$date`) kullanır. Tüm metin kipi her hücreyi metin tutar. + +Tür doğruluğu: `Double` dışa aktarılıp içe aktarıldığında `Decimal128` olur (ondalık metin için kayıpsız). `_id` olmayan ObjectId alanları `recognizeObjectIdHex` açık olmadıkça String olarak gider gelir. Küçük `Int64` değerleri `Int32`'ye daralabilir; ISO tarih biçimli dizeler `DateTime` olur. Extended JSON türleri (`$binary`, `$regularExpression`, `$timestamp`, `$minKey`, `$maxKey`, `$code`) JSON metin hücreleri aracılığıyla gider gelir. Boş dizi `[]` ve boş nesne `{}` doğru şekilde gider gelir. + +İlk sürüm sınırları: yalnızca ekleme (truncate veya upsert yok), alan başına tür geçersiz kılma yok, CSV dışa aktarma ilk 10000 belgeden alanları keşfeder (sonraki alanlar sessizce atılır; değişken şemalar için NDJSON kullanın), JSON dizi dışa aktarma biçimi yok, Excel/GridFS/BSON ikili içe aktarma yok ve başarısız toplu işler otomatik yeniden denenmez. + +### Veri dışa aktarma + +- Bir koleksiyona sağ tıklayıp **Veri dışa aktar**, ardından CSV veya NDJSON seçin. Tablo dışa aktarmada olduğu gibi kaydetme iletişim kutusu hemen açılır. +- CSV iç içe nesneleri noktalı sütunlar (`address.city`) ve dizileri dizinli sütunlar (`tags[0]`, `tags[1]`) olarak yazar. NDJSON her satıra bir kanonik Extended JSON belgesi yazar. +- 256’dan fazla ayırt edici alanı olan koleksiyonlar NDJSON kullanmalıdır. + +Yalnızca görünen seçim gerekiyorsa, ızgaradan geçerli belgeyi, seçili belgeleri veya alan değerlerini düz JSON ya da Extended JSON olarak kopyalayabilirsiniz. ## Güvenlik ve Uyumluluk Sınırları diff --git a/docs/content/docs/table-import.cn.mdx b/docs/content/docs/table-import.cn.mdx index 96bea2119..1a4a01cc0 100644 --- a/docs/content/docs/table-import.cn.mdx +++ b/docs/content/docs/table-import.cn.mdx @@ -5,6 +5,8 @@ description: 通过向导将 CSV、TSV、文本、JSON、Excel 或 SQL 数据导 表数据导入用于把文件数据写入数据库表。DBX 会先解析和预览文件,再让你确认解析选项、目标表、字段映射和执行方式,而不是选中文件后立即写入。 +MongoDB Collection 不走这条关系型导入流程。请在 Collection 文档页使用 CSV/JSON 导入,参见 [MongoDB 工作区](/cn/docs/mongodb)。 + 导入会真实写入数据库。连接的只读保护会阻止导入,但数据库权限仍是最终边界。生产环境使用前请确认目标连接、目标表和备份方案。 ## 两种目标方式 diff --git a/docs/content/docs/table-import.mdx b/docs/content/docs/table-import.mdx index 924431be2..80315f037 100644 --- a/docs/content/docs/table-import.mdx +++ b/docs/content/docs/table-import.mdx @@ -5,6 +5,8 @@ description: Use a guided workflow to import CSV, TSV, text, JSON, Excel, or SQL Table Import writes file data into database tables. DBX parses and previews the source first, then asks you to review parsing options, the target, column mappings, and execution settings before any rows are written. +MongoDB collections do not use this relational workflow. Import CSV/JSON documents from the collection document page instead; see [MongoDB Workspace](/en/docs/mongodb). + Import performs real database writes. Connection read-only protection blocks the operation, but database privileges remain the final boundary. Verify the target connection, target table, and backup plan before importing into production. ## Target Modes diff --git a/docs/content/docs/table-import.tr.mdx b/docs/content/docs/table-import.tr.mdx index 1a95e0536..f9f196d75 100644 --- a/docs/content/docs/table-import.tr.mdx +++ b/docs/content/docs/table-import.tr.mdx @@ -5,6 +5,8 @@ description: CSV, TSV, metin, JSON, Excel ya da SQL verisini adım adım bir ak Tablo İçe Aktarma, dosya verisini veritabanı tablolarına yazar. DBX önce kaynağı ayrıştırıp önizler, ardından herhangi bir satır yazılmadan önce ayrıştırma seçeneklerini, hedefi, sütun eşlemelerini ve çalıştırma ayarlarını gözden geçirmenizi ister. +MongoDB koleksiyonları bu ilişkisel iş akışını kullanmaz. CSV/JSON belgelerini koleksiyon belge sayfasından içe aktarın; bkz. [MongoDB Çalışma Alanı](/tr/docs/mongodb). + İçe aktarma gerçek veritabanı yazmaları yapar. Bağlantının salt okunur koruması işlemi engeller, ancak son sınır veritabanı yetkileridir. Üretime aktarmadan önce hedef bağlantıyı, hedef tabloyu ve yedekleme planınızı doğrulayın. ## Hedef Modları diff --git a/src-tauri/src/commands/mod.rs b/src-tauri/src/commands/mod.rs index da25644c7..275c64191 100644 --- a/src-tauri/src/commands/mod.rs +++ b/src-tauri/src/commands/mod.rs @@ -29,6 +29,7 @@ pub mod mcp; pub mod mcp_bridge; pub mod mcp_http_server; pub mod mongo_cmd; +pub mod mongodb_import_export; #[cfg(feature = "mq-admin")] pub mod mq_cmd; #[cfg(feature = "mq-admin")] diff --git a/src-tauri/src/commands/mongodb_import_export.rs b/src-tauri/src/commands/mongodb_import_export.rs new file mode 100644 index 000000000..a07a15870 --- /dev/null +++ b/src-tauri/src/commands/mongodb_import_export.rs @@ -0,0 +1,92 @@ +use std::collections::HashSet; +use std::sync::{Arc, OnceLock}; + +use tauri::{AppHandle, Emitter, State}; +use tokio::sync::RwLock; + +use crate::commands::connection::{ensure_connection_writable, AppState}; + +pub use dbx_core::mongodb_import_export::{ + MongoExportProgress, MongoExportRequest, MongoExportSummary, MongoImportPreview, MongoImportPreviewRequest, + MongoImportProgress, MongoImportRequest, MongoImportSummary, +}; + +static CANCELLED_IMPORTS: OnceLock>> = OnceLock::new(); + +fn cancelled_imports() -> &'static RwLock> { + CANCELLED_IMPORTS.get_or_init(|| RwLock::new(HashSet::new())) +} + +fn emit_import_progress(app: &AppHandle, progress: MongoImportProgress) { + let _ = app.emit("mongo-import-progress", progress); +} + +fn emit_export_progress(app: &AppHandle, progress: MongoExportProgress) { + let _ = app.emit("mongo-export-progress", progress); +} + +async fn is_cancelled(import_id: &str) -> bool { + cancelled_imports().read().await.contains(import_id) +} + +async fn clear_cancelled(import_id: &str) { + cancelled_imports().write().await.remove(import_id); +} + +#[tauri::command] +pub async fn preview_mongodb_import_file(request: MongoImportPreviewRequest) -> Result { + dbx_core::mongodb_import_export::preview_mongodb_import_file_core(request).await +} + +#[tauri::command] +pub async fn import_mongodb_file( + app: AppHandle, + state: State<'_, Arc>, + request: MongoImportRequest, +) -> Result { + clear_cancelled(&request.import_id).await; + ensure_connection_writable(&state, &request.connection_id, "Import").await?; + let result = dbx_core::mongodb_import_export::import_mongodb_file_core( + &state, + &request, + |import_id| { + let import_id = import_id.to_string(); + Box::pin(async move { is_cancelled(&import_id).await }) + }, + |progress| emit_import_progress(&app, progress), + ) + .await + .map_err(|error| error.display_message()); + clear_cancelled(&request.import_id).await; + result +} + +#[tauri::command] +pub async fn cancel_mongodb_import(import_id: String) -> Result { + cancelled_imports().write().await.insert(import_id); + Ok(true) +} + +#[tauri::command] +pub async fn export_mongodb_query( + app: AppHandle, + state: State<'_, Arc>, + request: MongoExportRequest, +) -> Result { + dbx_core::mongodb_import_export::export_mongodb_query_core( + &state, + &request, + |export_id| { + let export_id = export_id.to_string(); + Box::pin(async move { dbx_core::transfer::is_cancelled(&export_id).await }) + }, + |progress| emit_export_progress(&app, progress), + ) + .await +} + +#[tauri::command] +pub async fn cancel_mongodb_export(export_id: String) -> Result { + dbx_core::transfer::set_cancelled(&export_id).await; + Ok(true) +} diff --git a/src-tauri/src/lib.rs b/src-tauri/src/lib.rs index 9c4fcdf99..9bc0d598a 100644 --- a/src-tauri/src/lib.rs +++ b/src-tauri/src/lib.rs @@ -1981,6 +1981,11 @@ pub fn run() { commands::table_import::preview_table_import_file, commands::table_import::import_table_file, commands::table_import::cancel_table_import, + commands::mongodb_import_export::preview_mongodb_import_file, + commands::mongodb_import_export::import_mongodb_file, + commands::mongodb_import_export::cancel_mongodb_import, + commands::mongodb_import_export::export_mongodb_query, + commands::mongodb_import_export::cancel_mongodb_export, commands::redis_cmd::redis_list_databases, commands::redis_cmd::redis_scan_keys, commands::redis_cmd::redis_scan_keys_batch,