From d5fd45bb41c15a0cd087ad04f3efd83cd400848d Mon Sep 17 00:00:00 2001 From: Stephen Zhou <38493346+hyoban@users.noreply.github.com> Date: Tue, 25 Aug 2026 19:58:15 +0800 Subject: [PATCH] feat(knowledge_fs): expand supported document upload formats --- api/services/file_service.py | 2 +- .../knowledge_fs/staged_upload_service.py | 36 ++++- ...test_knowledge_fs_staged_upload_service.py | 93 ++++++++++++- ...-08-25-expanded-document-upload-formats.md | 73 ++++++++++ .../apps/api/src/parser-options.test.ts | 2 +- .../api/src/document-upload-utils.test.ts | 45 +++++++ .../packages/api/src/document-upload-utils.ts | 113 +++++++--------- knowledge-fs/packages/parsers/src/index.ts | 96 +++++++++++--- .../packages/parsers/src/parser.test.ts | 125 +++++++++++++++++- .../new-rag/__tests__/documents-page.spec.tsx | 2 +- .../new-rag/document-upload-policy.ts | 8 ++ web/i18n/ar-TN/dataset.json | 2 +- web/i18n/de-DE/dataset.json | 2 +- web/i18n/en-US/dataset.json | 2 +- web/i18n/es-ES/dataset.json | 2 +- web/i18n/fa-IR/dataset.json | 2 +- web/i18n/fr-FR/dataset.json | 2 +- web/i18n/hi-IN/dataset.json | 2 +- web/i18n/id-ID/dataset.json | 2 +- web/i18n/it-IT/dataset.json | 2 +- web/i18n/ja-JP/dataset.json | 2 +- web/i18n/ko-KR/dataset.json | 2 +- web/i18n/lo-LA/dataset.json | 2 +- web/i18n/nl-NL/dataset.json | 2 +- web/i18n/pl-PL/dataset.json | 2 +- web/i18n/pt-BR/dataset.json | 2 +- web/i18n/ro-RO/dataset.json | 2 +- web/i18n/ru-RU/dataset.json | 2 +- web/i18n/sl-SI/dataset.json | 2 +- web/i18n/th-TH/dataset.json | 2 +- web/i18n/tr-TR/dataset.json | 2 +- web/i18n/uk-UA/dataset.json | 2 +- web/i18n/vi-VN/dataset.json | 2 +- web/i18n/zh-Hans/dataset.json | 2 +- web/i18n/zh-Hant/dataset.json | 2 +- 35 files changed, 530 insertions(+), 113 deletions(-) create mode 100644 knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md diff --git a/api/services/file_service.py b/api/services/file_service.py index 4497639eb25..5d1261c0575 100644 --- a/api/services/file_service.py +++ b/api/services/file_service.py @@ -55,7 +55,7 @@ class FileService: mimetype: str, user: Account | EndUser, tenant_id: str | None = None, - source: Literal["datasets"] | None = None, + source: Literal["datasets", "knowledge_fs"] | None = None, source_url: str = "", default_file_size_limit: int | None = None, ) -> UploadFile: diff --git a/api/services/knowledge_fs/staged_upload_service.py b/api/services/knowledge_fs/staged_upload_service.py index c9e21f6849a..571fb893a2f 100644 --- a/api/services/knowledge_fs/staged_upload_service.py +++ b/api/services/knowledge_fs/staged_upload_service.py @@ -32,6 +32,33 @@ from services.knowledge_fs.product_dto import ( ) STAGED_UPLOAD_TTL = timedelta(hours=24) +_KNOWLEDGE_FS_DOCUMENT_MIME_TYPES = { + "csv": "text/csv", + "doc": "application/msword", + "docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document", + "eml": "message/rfc822", + "epub": "application/epub+zip", + "htm": "text/html", + "html": "text/html", + "json": "application/json", + "jsonl": "application/x-ndjson", + "markdown": "text/markdown", + "md": "text/markdown", + "mdx": "text/mdx", + "msg": "application/vnd.ms-outlook", + "odt": "application/vnd.oasis.opendocument.text", + "pdf": "application/pdf", + "ppt": "application/vnd.ms-powerpoint", + "pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation", + "properties": "text/x-java-properties", + "rtf": "application/rtf", + "text": "text/plain", + "txt": "text/plain", + "vtt": "text/vtt", + "xls": "application/vnd.ms-excel", + "xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", + "xml": "application/xml", +} class KnowledgeFSStagedUploadError(ValueError): @@ -78,7 +105,12 @@ class KnowledgeFSStagedUploadService: ) -> KnowledgeFSStagedUploadResponse: if not body: raise KnowledgeFSStagedUploadInvalidError("KnowledgeFS staged upload is empty") - normalized_content_type = content_type.strip() or "application/octet-stream" + _, separator, extension = file_name.strip().lower().rpartition(".") + if not separator or extension not in _KNOWLEDGE_FS_DOCUMENT_MIME_TYPES: + raise KnowledgeFSStagedUploadInvalidError("KnowledgeFS staged upload is invalid") + # Browser/OS MIME declarations are inconsistent and can route a binary document through a + # text parser. The admitted extension is the product contract, so persist its canonical MIME. + normalized_content_type = _KNOWLEDGE_FS_DOCUMENT_MIME_TYPES[extension] checksum = b64encode(sha256(body).digest()).decode() try: upload_file = FileService(self._session_maker).upload_file( @@ -87,7 +119,7 @@ class KnowledgeFSStagedUploadService: mimetype=normalized_content_type, user=account, tenant_id=tenant_id, - source="datasets", + source="knowledge_fs", default_file_size_limit=file_size_limit_mb, ) except FileTooLargeError as exc: diff --git a/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py b/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py index e196cfd457f..d4599208155 100644 --- a/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py +++ b/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py @@ -265,10 +265,10 @@ def test_stage_persists_workspace_owned_upload( file_service.upload_file.assert_called_once_with( filename="guide.pdf", content=_BODY, - mimetype="application/octet-stream", + mimetype="application/pdf", user=account, tenant_id=_TENANT_ID, - source="datasets", + source="knowledge_fs", default_file_size_limit=15, ) with sqlite_session_factory() as session: @@ -277,14 +277,73 @@ def test_stage_persists_workspace_owned_upload( assert persisted.checksum_sha256_base64 == b64encode(sha256(_BODY).digest()).decode() +@pytest.mark.parametrize( + ("file_name", "content_type", "expected_content_type"), + [ + ("report.pdf", "text/plain", "application/pdf"), + ("formatted.rtf", "text/rtf", "application/rtf"), + ("message.msg", "application/x-msg", "application/vnd.ms-outlook"), + ], +) +def test_stage_canonicalizes_content_type_from_the_supported_extension( + sqlite_session_factory: sessionmaker[Session], + monkeypatch: pytest.MonkeyPatch, + file_name: str, + content_type: str, + expected_content_type: str, +) -> None: + upload_file = _upload_file() + upload_file.name = file_name + upload_file.extension = file_name.rsplit(".", 1)[1] + upload_file.mime_type = expected_content_type + with sqlite_session_factory.begin() as session: + session.add(upload_file) + file_service = MagicMock() + file_service.upload_file.return_value = upload_file + monkeypatch.setattr(staged_upload_module, "FileService", lambda _: file_service) + service = KnowledgeFSStagedUploadService( + sqlite_session_factory, + facade=cast(KnowledgeFSDataFacade, MagicMock()), + ) + account = cast(Account, SimpleNamespace(id=_ACCOUNT_ID)) + + response = service.stage( + tenant_id=_TENANT_ID, + account=account, + file_name=file_name, + content_type=content_type, + body=_BODY, + file_size_limit_mb=15, + ) + + assert response.content_type == expected_content_type + file_service.upload_file.assert_called_once_with( + filename=file_name, + content=_BODY, + mimetype=expected_content_type, + user=account, + tenant_id=_TENANT_ID, + source="knowledge_fs", + default_file_size_limit=15, + ) + + @pytest.mark.parametrize( ("file_name", "content_type", "body"), [ ("notes.txt", "text/plain", b"KnowledgeFS notes"), ("guide.md", "text/markdown", b"# KnowledgeFS guide"), + ("README.markdown", "text/markdown", b"# KnowledgeFS guide"), + ("component.mdx", "text/mdx", b"# KnowledgeFS component"), + ("captions.vtt", "text/vtt", b"WEBVTT\n\n00:00.000 --> 00:01.000\nKnowledgeFS"), + ("application.properties", "text/x-java-properties", b"knowledge.fs=enabled"), + ("feed.xml", "application/xml", b"KnowledgeFS"), + ("manual.odt", "application/vnd.oasis.opendocument.text", b"odt"), + ("message.eml", "message/rfc822", b"Subject: KnowledgeFS\n\nBody"), + ("message.msg", "application/vnd.ms-outlook", b"msg"), ], ) -def test_stage_accepts_supported_text_files_with_the_real_file_service( +def test_stage_accepts_knowledge_fs_document_formats_with_the_real_file_service( sqlite_session_factory: sessionmaker[Session], monkeypatch: pytest.MonkeyPatch, file_name: str, @@ -322,6 +381,34 @@ def test_stage_accepts_supported_text_files_with_the_real_file_service( assert persisted.upload_file_id +@pytest.mark.parametrize("file_name", ["malware.exe", "md"]) +def test_stage_rejects_an_unsupported_filename_with_the_real_file_service( + sqlite_session_factory: sessionmaker[Session], monkeypatch: pytest.MonkeyPatch, file_name: str +) -> None: + backend = FakeStorage() + monkeypatch.setattr(file_service_module, "storage", backend) + monkeypatch.setattr(staged_upload_module, "storage", backend) + monkeypatch.setattr(file_service_module.file_helpers, "get_signed_file_url", lambda **_: "signed") + account = Account(name="KnowledgeFS tester", email="knowledge-fs@example.com") + account.id = _ACCOUNT_ID + service = KnowledgeFSStagedUploadService( + sqlite_session_factory, + facade=cast(KnowledgeFSDataFacade, MagicMock()), + ) + + with pytest.raises(KnowledgeFSStagedUploadInvalidError, match="invalid"): + service.stage( + tenant_id=_TENANT_ID, + account=account, + file_name=file_name, + content_type="application/octet-stream", + body=b"not executable content", + file_size_limit_mb=15, + ) + + assert backend.objects == {} + + def test_stage_rejects_empty_and_maps_file_service_errors( sqlite_session_factory: sessionmaker[Session], monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md b/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md new file mode 100644 index 00000000000..034ce69b3ac --- /dev/null +++ b/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md @@ -0,0 +1,73 @@ +# Expanded document upload formats + +## What changed + +- Added upload admission, MIME validation, and octet-stream inference for the legacy-compatible + `.markdown`, `.mdx`, `.vtt`, `.properties`, `.xml`, `.odt`, `.eml`, and `.msg` formats. +- Routed VTT and Java properties files through the bounded native text parser. Markdown aliases and + XML continue to use the existing native Markdown and structured-data parsers; ODT, EML, and MSG + use the existing Unstructured parser boundary. +- Kept the Dify New RAG file picker and local upload policy in sync with the KnowledgeFS service + allowlist. +- Gave the Dify KnowledgeFS staging service the same explicit extension allowlist and a dedicated + `knowledge_fs` upload source, so staging no longer inherits the legacy knowledge-base `ETL_TYPE` + whitelist. Unsupported extensions are still rejected before storage writes. +- Canonicalized staged-upload MIME types from the admitted extension, and changed direct + KnowledgeFS admission from two independent allowlists to an extension-to-MIME contract. Common + aliases such as `text/rtf` and JSONL declared as `application/json` remain accepted, while + unrelated pairs such as PDF plus `text/plain` are rejected. +- Prioritized complex binary extensions in parser routing so an inaccurate browser MIME declaration + cannot send PDF, Office, EPUB, RTF, ODT, EML, or MSG content through the native text parser. +- Preserved visible text inside MDX JSX blocks instead of silently dropping Marked's block HTML + tokens. MDX now carries its own `native-mdx@1` parser version so the behavior does not invalidate + existing plain-Markdown artifact hashes. +- Updated upload guidance in every supported locale to describe the supported format groups and + disclose the complex-document parser dependency. +- Added behavior tests for declared MIME types, octet-stream inference, native lightweight-text + routing, and the browser file-picker contract. The new tests were observed failing before the + implementation and passing afterward. + +## Why + +The new knowledge base rejected several formats already accepted by the legacy knowledge base even +though its parser stack could process them. Expanding the allowlists and using the lightest existing +parser restores compatibility without adding a new parser, storage path, or network dependency. + +## Verification + +- `pnpm --filter @knowledge/api exec vitest run src/document-upload-utils.test.ts` — passed (21 tests). +- `pnpm --filter @knowledge/parsers exec vitest run src/parser.test.ts` — passed (55 tests). +- `pnpm --filter @knowledge/parsers test:coverage` — passed with 95.69% statements/lines, + 90.02% branches, and 97.52% functions. +- `pnpm --filter @knowledge/api-app exec vitest run src/parser-options.test.ts` — passed (5 tests). +- `vp test run --project unit features/new-rag/__tests__/documents-page.spec.tsx` — passed (203 tests). +- KnowledgeFS typechecks — passed; the full Turbo test pipeline passed (22 tasks), including the API + suite with 4,640 tests passed and 3 skipped. +- Targeted KnowledgeFS Biome check for the five changed TypeScript files — passed. +- Targeted Dify `vp check` for the two changed Web files — passed. +- All 24 localized `dataset.json` files parsed successfully and contain the updated upload-format + guidance. The repository-wide dataset i18n alignment check remains blocked by pre-existing + missing KnowledgeFS quality-evaluation, task-failure, and related keys outside this change. +- Dify KnowledgeFS staged-upload service test — passed (42 tests), including real `FileService` + coverage for canonical MIME persistence, expanded formats, and rejection of unsupported or + extensionless filenames. +- Targeted Ruff format and lint checks for the three changed Python files — passed. +- Targeted Pyrefly checks for the changed Python service files — passed. +- Targeted Mypy was attempted but the installed Mypy 1.20.2 failed internally while reading its own + `typeshed/stdlib/zipimport.pyi`, before reporting project diagnostics. +- KnowledgeFS `pnpm build` — passed; the existing Next.js multiple-lockfile and ESLint-plugin warnings + remain unchanged. +- KnowledgeFS `pnpm lint` — attempted but remains blocked by pre-existing formatting/lint failures in + unrelated Admin, test setup, and generated contract files. No unrelated files were modified; the + targeted Biome check above covers every KnowledgeFS source and test file changed here. + +## Risks and follow-up + +- ODT, EML, and MSG parsing still requires a configured and capable Unstructured service, matching + other complex document types such as DOC and PPT. The upload guidance now calls out this + dependency; upload admission remains independent, while downstream parser failures continue to + use the existing failed-document lifecycle. +- The added allowlist entries are fixed-size `Set` members. Admission remains constant-time and does + not change upload byte limits, buffering, database access, or object-storage behavior. +- MDX JSX tags and attributes remain syntax rather than searchable text; visible child text is + retained, while `script`, `style`, and `noscript` contents remain excluded. diff --git a/knowledge-fs/apps/api/src/parser-options.test.ts b/knowledge-fs/apps/api/src/parser-options.test.ts index f53007e0982..5c0e9db1eee 100644 --- a/knowledge-fs/apps/api/src/parser-options.test.ts +++ b/knowledge-fs/apps/api/src/parser-options.test.ts @@ -67,7 +67,7 @@ describe("createApiDocumentParser", () => { expect(requestedUrl).toBe("https://unstructured.example.test/general/v0/general"); expect(artifact).toMatchObject({ metadata: { - routeReason: "unsupported-file-type", + routeReason: "complex-file-type", routedParser: "unstructured", }, parser: "unstructured", diff --git a/knowledge-fs/packages/api/src/document-upload-utils.test.ts b/knowledge-fs/packages/api/src/document-upload-utils.test.ts index 695376e2ee8..50b7de7ac98 100644 --- a/knowledge-fs/packages/api/src/document-upload-utils.test.ts +++ b/knowledge-fs/packages/api/src/document-upload-utils.test.ts @@ -280,6 +280,7 @@ describe("document upload utilities", () => { "application/x-ndjson", "application/jsonl", "application/ndjson", + "application/json", "application/octet-stream", ]) { const result = await readBulkDocumentUploadWithAdmission( @@ -302,6 +303,50 @@ describe("document upload utilities", () => { } }); + it.each([ + ["README.markdown", "text/markdown"], + ["component.mdx", "text/mdx"], + ["captions.vtt", "text/vtt"], + ["application.properties", "text/x-java-properties"], + ["formatted.rtf", "text/rtf"], + ["feed.xml", "application/xml"], + ["manual.odt", "application/vnd.oasis.opendocument.text"], + ["message.eml", "message/rfc822"], + ["message.msg", "application/vnd.ms-outlook"], + ])("accepts legacy-compatible document upload %s", async (filename, declaredMimeType) => { + for (const type of [declaredMimeType, "application/octet-stream"]) { + const result = await readBulkDocumentUploadWithAdmission( + { + parseBody: async () => ({ + files: [new File(["content"], filename, { type })], + }), + }, + { + maxAcceptedBytesByQuota: null, + maxBulkUploadBytes: 100, + maxBulkUploadFiles: 20, + maxUploadBytes: 100, + }, + ); + + expect(result.accepted).toHaveLength(1); + expect(result.accepted[0]?.filename).toBe(filename); + } + }); + + it("rejects a supported extension paired with an unrelated MIME type", async () => { + await expect( + readDocumentUpload( + { + parseBody: async () => ({ + file: new File(["%PDF-1.7"], "report.pdf", { type: "text/plain" }), + }), + }, + 100, + ), + ).rejects.toThrow(DocumentUploadValidationError); + }); + it("reports quota, aggregate-byte, and count exclusions without discarding earlier files", async () => { const files = [ new File(["aa"], "a.txt", { type: "text/plain" }), diff --git a/knowledge-fs/packages/api/src/document-upload-utils.ts b/knowledge-fs/packages/api/src/document-upload-utils.ts index 1f42cf8e512..2171d5b749e 100644 --- a/knowledge-fs/packages/api/src/document-upload-utils.ts +++ b/knowledge-fs/packages/api/src/document-upload-utils.ts @@ -20,25 +20,42 @@ export interface BulkDocumentRevisionTarget { readonly expectedDocumentRowVersion: number; } -export const SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES = new Set([ - "application/jsonl", - "application/ndjson", - "application/epub+zip", - "application/json", - "application/msword", - "application/pdf", - "application/rtf", - "application/vnd.ms-excel", - "application/vnd.ms-powerpoint", - "application/vnd.openxmlformats-officedocument.presentationml.presentation", - "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", - "application/vnd.openxmlformats-officedocument.wordprocessingml.document", - "application/x-ndjson", - "text/csv", - "text/html", - "text/markdown", - "text/plain", -]); +const DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION = { + csv: ["text/csv", "application/vnd.ms-excel"], + doc: ["application/msword"], + docx: ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"], + eml: ["message/rfc822"], + epub: ["application/epub+zip"], + htm: ["text/html"], + html: ["text/html"], + json: ["application/json"], + jsonl: ["application/x-ndjson", "application/jsonl", "application/ndjson", "application/json"], + markdown: ["text/markdown", "text/x-markdown", "text/plain"], + md: ["text/markdown", "text/x-markdown", "text/plain"], + mdx: ["text/mdx", "text/markdown", "text/plain"], + msg: ["application/vnd.ms-outlook", "application/x-msg"], + odt: ["application/vnd.oasis.opendocument.text"], + pdf: ["application/pdf"], + ppt: ["application/vnd.ms-powerpoint", "application/mspowerpoint", "application/x-mspowerpoint"], + pptx: ["application/vnd.openxmlformats-officedocument.presentationml.presentation"], + properties: ["text/x-java-properties", "text/plain"], + rtf: ["application/rtf", "text/rtf", "application/x-rtf"], + text: ["text/plain"], + txt: ["text/plain"], + vtt: ["text/vtt", "text/plain"], + xls: [ + "application/vnd.ms-excel", + "application/excel", + "application/x-excel", + "application/x-msexcel", + ], + xlsx: ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"], + xml: ["application/xml", "text/xml"], +} as const satisfies Readonly>; + +export const SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES = new Set( + Object.values(DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION).flat(), +); export const DEFAULT_DOCUMENT_UPLOAD_MAX_BYTES = 15 * 1024 * 1024; export const DEFAULT_BULK_DOCUMENT_UPLOAD_MAX_BYTES = 50 * 1024 * 1024; @@ -48,25 +65,9 @@ export const HARD_BULK_DOCUMENT_UPLOAD_MAX_FILES = 25; export const HARD_BULK_DOCUMENT_UPLOAD_MAX_BYTES = HARD_DOCUMENT_UPLOAD_MAX_BYTES * HARD_BULK_DOCUMENT_UPLOAD_MAX_FILES; -export const SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS = new Set([ - "csv", - "doc", - "docx", - "epub", - "htm", - "html", - "json", - "jsonl", - "md", - "pdf", - "ppt", - "pptx", - "rtf", - "text", - "txt", - "xls", - "xlsx", -]); +export const SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS = new Set( + Object.keys(DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION), +); export type DocumentUploadExclusionReason = | "batch_byte_limit_exceeded" @@ -539,27 +540,11 @@ function isJsonObject(value: unknown): value is Record { export function normalizeDocumentMimeType(file: File): string { const declared = file.type.trim().toLocaleLowerCase(); const extension = documentExtension(file.name); - const inferred = ( - { - csv: "text/csv", - doc: "application/msword", - docx: "application/vnd.openxmlformats-officedocument.wordprocessingml.document", - epub: "application/epub+zip", - html: "text/html", - htm: "text/html", - json: "application/json", - jsonl: "application/x-ndjson", - md: "text/markdown", - pdf: "application/pdf", - ppt: "application/vnd.ms-powerpoint", - pptx: "application/vnd.openxmlformats-officedocument.presentationml.presentation", - rtf: "application/rtf", - text: "text/plain", - txt: "text/plain", - xls: "application/vnd.ms-excel", - xlsx: "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet", - } as Readonly> - )[extension ?? ""]; + const inferred = extension + ? (DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION as Readonly>)[ + extension + ]?.[0] + : undefined; return !declared || declared === "application/octet-stream" ? (inferred ?? "application/octet-stream") : declared; @@ -567,11 +552,11 @@ export function normalizeDocumentMimeType(file: File): string { function isSupportedDocumentUpload(file: File, mimeType: string): boolean { const extension = documentExtension(file.name); - return ( - SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES.has(mimeType) && - extension !== undefined && - SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS.has(extension) - ); + if (extension === undefined) return false; + const allowedMimeTypes = ( + DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION as Readonly> + )[extension]; + return allowedMimeTypes?.includes(mimeType) === true; } function documentExtension(filename: string): string | undefined { diff --git a/knowledge-fs/packages/parsers/src/index.ts b/knowledge-fs/packages/parsers/src/index.ts index c95e498228c..ad29251aec4 100644 --- a/knowledge-fs/packages/parsers/src/index.ts +++ b/knowledge-fs/packages/parsers/src/index.ts @@ -191,6 +191,20 @@ const defaultMaxRows = 20_000; const defaultRetryDelayMs = 100; const defaultNow = () => new Date().toISOString(); const defaultGenerateId = () => crypto.randomUUID(); +const unstructuredDocumentExtensions = new Set([ + "doc", + "docx", + "eml", + "epub", + "msg", + "odt", + "pdf", + "ppt", + "pptx", + "rtf", + "xls", + "xlsx", +]); const UnstructuredElementSchema = z.object({ element_id: z.string().min(1).max(512).optional(), @@ -209,11 +223,12 @@ export function createNativeMarkdownParser(options: NativeParserOptions = {}): P return { kind: "native-markdown", parse: async (input) => { - const parserVersion = options.parserVersion ?? "native-markdown@1"; + const isMdx = isMdxInput(input); + const parserVersion = options.parserVersion ?? (isMdx ? "native-mdx@1" : "native-markdown@1"); assertInputBounds(input.body, options.maxInputBytes ?? defaultMaxInputBytes); const text = decodeUtf8(input.body); const tokens = marked.lexer(text, { gfm: true }); - const elements = markdownTokensToElements(tokens); + const elements = markdownTokensToElements(tokens, { preserveHtmlText: isMdx }); return createParseArtifact({ elements, @@ -237,7 +252,7 @@ export function createNativeHtmlParser(options: NativeParserOptions = {}): Parse lowerCaseAttributeNames: true, lowerCaseTags: true, }); - const nodes = (document.children ?? []) as HtmlNode[]; + const nodes = document.children as HtmlNode[]; const elements = htmlNodesToElements(nodes); const documentTitle = htmlDocumentTitle(nodes); @@ -742,6 +757,10 @@ function selectParser( return { parser: unstructured, reason: "unsupported-native-language" }; } + if (unstructuredDocumentExtensions.has(filename.split(".").at(-1) ?? "")) { + return { parser: unstructured, reason: "complex-file-type" }; + } + const structuredFormat = structuredDataFormat(input); if (structuredFormat && input.body.byteLength > maxNativeInputBytes) { @@ -754,10 +773,15 @@ function selectParser( const nativeParser = mimeType === "text/markdown" || + mimeType === "text/mdx" || mimeType === "text/plain" || + mimeType === "text/vtt" || + mimeType === "text/x-java-properties" || filename.endsWith(".md") || filename.endsWith(".markdown") || - filename.endsWith(".mdx") + filename.endsWith(".mdx") || + filename.endsWith(".properties") || + filename.endsWith(".vtt") ? markdown : mimeType === "text/html" || mimeType === "application/xhtml+xml" || @@ -841,23 +865,22 @@ function structuredDataFormat({ return "csv"; } - if ( - normalizedMime === "application/json" || - normalizedMime === "text/json" || - normalizedFilename.endsWith(".json") - ) { + if (normalizedFilename.endsWith(".jsonl") || normalizedFilename.endsWith(".ndjson")) { + return "jsonl"; + } + + if (normalizedFilename.endsWith(".json")) { return "json"; } - if ( - normalizedMime === "application/x-ndjson" || - normalizedMime === "application/jsonl" || - normalizedFilename.endsWith(".jsonl") || - normalizedFilename.endsWith(".ndjson") - ) { + if (normalizedMime === "application/x-ndjson" || normalizedMime === "application/jsonl") { return "jsonl"; } + if (normalizedMime === "application/json" || normalizedMime === "text/json") { + return "json"; + } + if ( normalizedMime === "application/yaml" || normalizedMime === "text/yaml" || @@ -1024,7 +1047,10 @@ function uniqueStrings(values: readonly string[]): string[] { return [...new Set(values)]; } -function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[] { +function markdownTokensToElements( + tokens: readonly Token[], + { preserveHtmlText }: { readonly preserveHtmlText: boolean }, +): ParseElementInput[] { const elements: ParseElementInput[] = []; const sectionPath: string[] = []; @@ -1078,6 +1104,12 @@ function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[] continue; } + if (token.type === "html" && preserveHtmlText) { + const html = token as Tokens.HTML; + pushTextElement(elements, "paragraph", markdownHtmlBlockText(html.text), sectionPath); + continue; + } + if (token.type === "list") { const list = token as Tokens.List; pushTextElement( @@ -1106,6 +1138,38 @@ function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[] return elements; } +function isMdxInput({ + filename, + mimeType, +}: Pick): boolean { + return ( + mimeType.trim().toLowerCase() === "text/mdx" || filename.trim().toLowerCase().endsWith(".mdx") + ); +} + +function markdownHtmlBlockText(source: string): string { + const document = parseDocument(source, { + lowerCaseAttributeNames: true, + lowerCaseTags: true, + }); + const nodes = document.children as HtmlNode[]; + + return nodes.map(searchableMarkdownHtmlText).join("\n"); +} + +function searchableMarkdownHtmlText(node: HtmlNode): string { + const name = node.name?.toLowerCase(); + if (name && ["script", "style", "noscript"].includes(name)) { + return ""; + } + + if (!node.children?.length) { + return htmlText(node); + } + + return node.children.map(searchableMarkdownHtmlText).join("\n"); +} + function htmlNodesToElements(nodes: readonly HtmlNode[]): ParseElementInput[] { const elements: ParseElementInput[] = []; const sectionPath: string[] = []; diff --git a/knowledge-fs/packages/parsers/src/parser.test.ts b/knowledge-fs/packages/parsers/src/parser.test.ts index a8fd425766b..9aa7b1a4760 100644 --- a/knowledge-fs/packages/parsers/src/parser.test.ts +++ b/knowledge-fs/packages/parsers/src/parser.test.ts @@ -87,6 +87,10 @@ describe("parser adapters", () => { "const answer = 42;", "```", "", + "```", + "plain code block", + "```", + "", "| A | B |", "| - | - |", "| 1 | 2 |", @@ -143,12 +147,69 @@ describe("parser adapters", () => { id: "018f0d60-7a49-7cc2-9c1b-5b36f18f2c45:element-5", metadata: {}, sectionPath: ["Overview"], + text: "plain code block", + type: "code", + }, + { + id: "018f0d60-7a49-7cc2-9c1b-5b36f18f2c45:element-6", + metadata: {}, + sectionPath: ["Overview"], text: "A | B\n1 | 2", type: "table", }, ]); }); + it.each(["text/mdx", "text/plain"])( + "preserves searchable text inside MDX JSX blocks declared as %s", + async (mimeType) => { + const parser = createNativeMarkdownParser({ + generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c95", + now: () => createdAt, + }); + + const artifact = await parser.parse( + createParseInput({ + body: [ + "# Overview", + "", + '', + "MDX keeps this searchable.", + "Nested detail", + "", + "", + ].join("\n"), + filename: "guide.mdx", + mimeType, + }), + ); + + expect(artifact.elements.map((element) => element.text)).toEqual([ + "Overview", + "MDX keeps this searchable.\nNested detail", + ]); + expect(artifact.metadata.parserVersion).toBe("native-mdx@1"); + }, + ); + + it("keeps plain Markdown raw HTML behavior and parser version unchanged", async () => { + const parser = createNativeMarkdownParser({ + generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c96", + now: () => createdAt, + }); + + const artifact = await parser.parse( + createParseInput({ + body: ["", "Plain Markdown keeps its existing behavior.", ""].join("\n"), + filename: "guide.md", + mimeType: "text/markdown", + }), + ); + + expect(artifact.elements).toEqual([]); + expect(artifact.metadata.parserVersion).toBe("native-markdown@1"); + }); + it("normalizes Markdown image references into image parse elements", async () => { const parser = createNativeMarkdownParser({ generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2d45", @@ -541,8 +602,15 @@ describe("parser adapters", () => { mimeType: "application/vnd.openxmlformats-officedocument.presentationml.presentation", version: 1, }); + await router.parse({ + body: textBytes("%PDF-1.7"), + documentAssetId, + filename: "report.pdf", + mimeType: "text/plain", + version: 1, + }); - expect(selected).toEqual(["markdown", "html", "unstructured"]); + expect(selected).toEqual(["markdown", "html", "unstructured", "unstructured"]); }); it("routes by file size, OCR need, layout complexity, and language hints", async () => { @@ -756,6 +824,31 @@ describe("parser adapters", () => { }); }); + it("uses the JSONL extension when the declared MIME type is application/json", async () => { + const parser = createNativeStructuredDataParser({ + generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c5d", + now: () => createdAt, + }); + + await expect( + parser.parse( + createParseInput({ + body: '{"name":"Ada"}\n{"name":"Lin"}', + filename: "records.jsonl", + mimeType: "application/json", + }), + ), + ).resolves.toMatchObject({ + elements: [ + { + metadata: { columns: ["name"], format: "jsonl", rowCount: 2 }, + text: "name\nAda\nLin", + type: "table", + }, + ], + }); + }); + it("routes structured data formats to the native structured parser", async () => { const selected: string[] = []; const structured = createNativeStructuredDataParser({ @@ -824,6 +917,36 @@ describe("parser adapters", () => { }); }); + it.each([ + ["captions.vtt", "text/vtt"], + ["application.properties", "text/x-java-properties"], + ])("routes lightweight text format %s to the native text parser", async (filename, mimeType) => { + const router = createParserRouter({ + html: createNativeHtmlParser(), + markdown: createNativeMarkdownParser(), + structured: createNativeStructuredDataParser(), + unstructured: { + kind: "unstructured", + parse: async () => { + throw new Error("lightweight text should not require Unstructured"); + }, + }, + }); + + await expect( + router.parse( + createParseInput({ + body: "first line\nsecond line", + filename, + mimeType, + }), + ), + ).resolves.toMatchObject({ + metadata: { routeReason: "native-file-type", routedParser: "native-markdown" }, + parser: "native-markdown", + }); + }); + it("rejects invalid or unbounded structured data inputs", async () => { await expect( createNativeStructuredDataParser({ maxRows: 1 }).parse( diff --git a/web/features/new-rag/__tests__/documents-page.spec.tsx b/web/features/new-rag/__tests__/documents-page.spec.tsx index 520761880b4..3c3ebc4d1dd 100644 --- a/web/features/new-rag/__tests__/documents-page.spec.tsx +++ b/web/features/new-rag/__tests__/documents-page.spec.tsx @@ -2106,7 +2106,7 @@ describe('DocumentsPage', () => { expect(input).toHaveAttribute('tabindex', '-1') expect(input).toHaveAttribute( 'accept', - '.csv,.doc,.docx,.epub,.htm,.html,.json,.jsonl,.md,.pdf,.ppt,.pptx,.rtf,.text,.txt,.xls,.xlsx', + '.csv,.doc,.docx,.eml,.epub,.htm,.html,.json,.jsonl,.markdown,.md,.mdx,.msg,.odt,.pdf,.ppt,.pptx,.properties,.rtf,.text,.txt,.vtt,.xls,.xlsx,.xml', ) await user.upload(input, new File(['one'], 'one.md', { type: 'text/markdown' })) diff --git a/web/features/new-rag/document-upload-policy.ts b/web/features/new-rag/document-upload-policy.ts index 2af46f765f3..82ddbd1ecc1 100644 --- a/web/features/new-rag/document-upload-policy.ts +++ b/web/features/new-rag/document-upload-policy.ts @@ -4,20 +4,28 @@ const DOCUMENT_UPLOAD_EXTENSIONS = [ 'csv', 'doc', 'docx', + 'eml', 'epub', 'htm', 'html', 'json', 'jsonl', + 'markdown', 'md', + 'mdx', + 'msg', + 'odt', 'pdf', 'ppt', 'pptx', + 'properties', 'rtf', 'text', 'txt', + 'vtt', 'xls', 'xlsx', + 'xml', ] as const const documentUploadExtensionSet = new Set(DOCUMENT_UPLOAD_EXTENSIONS) diff --git a/web/i18n/ar-TN/dataset.json b/web/i18n/ar-TN/dataset.json index cff14dfa622..b0320d4bfc8 100644 --- a/web/i18n/ar-TN/dataset.json +++ b/web/i18n/ar-TN/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "تم تجاوز حصة مساحة العمل", "newKnowledge.documentUploadExclusion.target": "لم يعد هدف المستند صالحًا", "newKnowledge.documentUploadFailed": "لم نتمكن من تحميل هذه المستندات. حاول ثانية.", - "newKnowledge.documentUploadFormats": "يدعم TXT وMarkdown وPDF وHTML وXLSX وCSV وJSONL · حتى 15 ميغابايت لكل ملف", + "newKnowledge.documentUploadFormats": "يدعم النصوص وMarkdown وHTML وPDF وOffice وEPUB والبريد الإلكتروني والبيانات المنظمة (CSV وJSON/JSONL وXML) · تتطلب التنسيقات المعقدة محلل مستندات · حتى 15 ميغابايت لكل ملف", "newKnowledge.documentUploadPartial": "بدأت معالجة {{accepted}} مستندًا؛ تعذرت إضافة {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "لم يتم قبول أي مستند: {{details}}", "newKnowledge.documentUploadStarted": "بدأت معالجة المستندات.", diff --git a/web/i18n/de-DE/dataset.json b/web/i18n/de-DE/dataset.json index 068b3f6acb9..8d3d778d028 100644 --- a/web/i18n/de-DE/dataset.json +++ b/web/i18n/de-DE/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "Arbeitsbereichskontingent überschritten", "newKnowledge.documentUploadExclusion.target": "Dokumentziel ist nicht mehr gültig", "newKnowledge.documentUploadFailed": "Wir konnten diese Dokumente nicht hochladen. Versuchen Sie es erneut.", - "newKnowledge.documentUploadFormats": "Unterstützt TXT, Markdown, PDF, HTML, XLSX, CSV und JSONL · jeweils bis zu 15 MB", + "newKnowledge.documentUploadFormats": "Unterstützt Text, Markdown, HTML, PDF, Office, EPUB, E-Mail und strukturierte Daten (CSV, JSON/JSONL, XML) · komplexe Formate erfordern einen Dokumentparser · bis zu 15 MB pro Datei", "newKnowledge.documentUploadPartial": "{{accepted}} Dokumente wurden gestartet; {{excluded}} konnten nicht hinzugefügt werden: {{details}}", "newKnowledge.documentUploadRejected": "Keine Dokumente wurden akzeptiert: {{details}}", "newKnowledge.documentUploadStarted": "Die Dokumentenverarbeitung wurde gestartet.", diff --git a/web/i18n/en-US/dataset.json b/web/i18n/en-US/dataset.json index 54e04b143fc..6d7e62c6139 100644 --- a/web/i18n/en-US/dataset.json +++ b/web/i18n/en-US/dataset.json @@ -264,7 +264,7 @@ "newKnowledge.documentUploadExclusion.quota": "workspace quota exceeded", "newKnowledge.documentUploadExclusion.target": "document target is no longer valid", "newKnowledge.documentUploadFailed": "We couldn't upload these documents. Try again.", - "newKnowledge.documentUploadFormats": "Supports TXT, Markdown, PDF, HTML, XLSX, CSV, and JSONL · up to 15MB each", + "newKnowledge.documentUploadFormats": "Supports text, Markdown, HTML, PDF, Office, EPUB, email, and structured data (CSV, JSON/JSONL, XML) · complex formats require a document parser · up to 15MB each", "newKnowledge.documentUploadPartial": "{{accepted}} documents started; {{excluded}} could not be added: {{details}}", "newKnowledge.documentUploadRejected": "No documents were accepted: {{details}}", "newKnowledge.documentUploadStarted": "Document processing started.", diff --git a/web/i18n/es-ES/dataset.json b/web/i18n/es-ES/dataset.json index 68e0b880ce2..53e8477a3d5 100644 --- a/web/i18n/es-ES/dataset.json +++ b/web/i18n/es-ES/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "se superó la cuota del espacio de trabajo", "newKnowledge.documentUploadExclusion.target": "el destino del documento ya no es válido", "newKnowledge.documentUploadFailed": "No pudimos cargar estos documentos. Intentar otra vez.", - "newKnowledge.documentUploadFormats": "Admite TXT, Markdown, PDF, HTML, XLSX, CSV y JSONL · hasta 15 MB cada uno", + "newKnowledge.documentUploadFormats": "Admite texto, Markdown, HTML, PDF, Office, EPUB, correo electrónico y datos estructurados (CSV, JSON/JSONL, XML) · los formatos complejos requieren un analizador de documentos · hasta 15 MB por archivo", "newKnowledge.documentUploadPartial": "Se inició el procesamiento de {{accepted}} documentos; no se pudieron añadir {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "No se aceptó ningún documento: {{details}}", "newKnowledge.documentUploadStarted": "Se inició el procesamiento de documentos.", diff --git a/web/i18n/fa-IR/dataset.json b/web/i18n/fa-IR/dataset.json index b2542bc7fb6..6693908c642 100644 --- a/web/i18n/fa-IR/dataset.json +++ b/web/i18n/fa-IR/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "سهمیه فضای کاری رد شده است", "newKnowledge.documentUploadExclusion.target": "هدف سند دیگر معتبر نیست", "newKnowledge.documentUploadFailed": "ما نتوانستیم این اسناد را آپلود کنیم. دوباره امتحان کنید.", - "newKnowledge.documentUploadFormats": "پشتیبانی از TXT، Markdown، PDF، HTML، XLSX، CSV و JSONL · هر فایل تا ۱۵ مگابایت", + "newKnowledge.documentUploadFormats": "پشتیبانی از متن، Markdown، HTML، PDF، Office، EPUB، ایمیل و داده‌های ساخت‌یافته (CSV، JSON/JSONL، XML) · قالب‌های پیچیده به تجزیه‌گر سند نیاز دارند · هر فایل تا ۱۵ مگابایت", "newKnowledge.documentUploadPartial": "پردازش {{accepted}} سند آغاز شد؛ {{excluded}} سند افزوده نشد: {{details}}", "newKnowledge.documentUploadRejected": "هیچ سندی پذیرفته نشد: {{details}}", "newKnowledge.documentUploadStarted": "پردازش اسناد آغاز شد.", diff --git a/web/i18n/fr-FR/dataset.json b/web/i18n/fr-FR/dataset.json index cdea99f344f..12d4ab82d2e 100644 --- a/web/i18n/fr-FR/dataset.json +++ b/web/i18n/fr-FR/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "quota de l’espace de travail dépassé", "newKnowledge.documentUploadExclusion.target": "la cible du document n’est plus valide", "newKnowledge.documentUploadFailed": "Nous n'avons pas pu télécharger ces documents. Essayer à nouveau.", - "newKnowledge.documentUploadFormats": "Formats pris en charge : TXT, Markdown, PDF, HTML, XLSX, CSV et JSONL · 15 Mo max par fichier", + "newKnowledge.documentUploadFormats": "Formats pris en charge : texte, Markdown, HTML, PDF, Office, EPUB, e-mail et données structurées (CSV, JSON/JSONL, XML) · les formats complexes nécessitent un analyseur de documents · 15 Mo max par fichier", "newKnowledge.documentUploadPartial": "Le traitement de {{accepted}} documents a démarré ; {{excluded}} n’ont pas pu être ajoutés : {{details}}", "newKnowledge.documentUploadRejected": "Aucun document n’a été accepté : {{details}}", "newKnowledge.documentUploadStarted": "Le traitement du document a commencé.", diff --git a/web/i18n/hi-IN/dataset.json b/web/i18n/hi-IN/dataset.json index d0493265c5c..4d525fcca1d 100644 --- a/web/i18n/hi-IN/dataset.json +++ b/web/i18n/hi-IN/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "कार्यस्थान कोटा पार हो गया", "newKnowledge.documentUploadExclusion.target": "दस्तावेज़ लक्ष्य अब मान्य नहीं है", "newKnowledge.documentUploadFailed": "हम ये दस्तावेज़ अपलोड नहीं कर सके. पुनः प्रयास करें।", - "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV और JSONL समर्थित · प्रत्येक फ़ाइल अधिकतम 15 MB", + "newKnowledge.documentUploadFormats": "टेक्स्ट, Markdown, HTML, PDF, Office, EPUB, ईमेल और संरचित डेटा (CSV, JSON/JSONL, XML) समर्थित · जटिल फ़ॉर्मैट के लिए दस्तावेज़ पार्सर आवश्यक है · प्रत्येक फ़ाइल अधिकतम 15 MB", "newKnowledge.documentUploadPartial": "{{accepted}} दस्तावेज़ों की प्रोसेसिंग शुरू हुई; {{excluded}} जोड़े नहीं जा सके: {{details}}", "newKnowledge.documentUploadRejected": "कोई दस्तावेज़ स्वीकार नहीं किया गया: {{details}}", "newKnowledge.documentUploadStarted": "दस्तावेज़ प्रसंस्करण शुरू हुआ.", diff --git a/web/i18n/id-ID/dataset.json b/web/i18n/id-ID/dataset.json index 5a6fa2704e2..ca6da2e1e2a 100644 --- a/web/i18n/id-ID/dataset.json +++ b/web/i18n/id-ID/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "kuota ruang kerja terlampaui", "newKnowledge.documentUploadExclusion.target": "target dokumen tidak lagi valid", "newKnowledge.documentUploadFailed": "Kami tidak dapat mengunggah dokumen-dokumen ini. Coba lagi.", - "newKnowledge.documentUploadFormats": "Mendukung TXT, Markdown, PDF, HTML, XLSX, CSV, dan JSONL · masing-masing hingga 15 MB", + "newKnowledge.documentUploadFormats": "Mendukung teks, Markdown, HTML, PDF, Office, EPUB, email, dan data terstruktur (CSV, JSON/JSONL, XML) · format kompleks memerlukan pengurai dokumen · hingga 15 MB per file", "newKnowledge.documentUploadPartial": "Pemrosesan {{accepted}} dokumen dimulai; {{excluded}} tidak dapat ditambahkan: {{details}}", "newKnowledge.documentUploadRejected": "Tidak ada dokumen yang diterima: {{details}}", "newKnowledge.documentUploadStarted": "Pemrosesan dokumen dimulai.", diff --git a/web/i18n/it-IT/dataset.json b/web/i18n/it-IT/dataset.json index 2b22b492764..f47cf80cc6f 100644 --- a/web/i18n/it-IT/dataset.json +++ b/web/i18n/it-IT/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "quota dell'area di lavoro superata", "newKnowledge.documentUploadExclusion.target": "la destinazione del documento non è più valida", "newKnowledge.documentUploadFailed": "Non è stato possibile caricare questi documenti. Riprova.", - "newKnowledge.documentUploadFormats": "Supporta TXT, Markdown, PDF, HTML, XLSX, CSV e JSONL · fino a 15 MB ciascuno", + "newKnowledge.documentUploadFormats": "Supporta testo, Markdown, HTML, PDF, Office, EPUB, e-mail e dati strutturati (CSV, JSON/JSONL, XML) · i formati complessi richiedono un parser di documenti · fino a 15 MB per file", "newKnowledge.documentUploadPartial": "Elaborazione avviata per {{accepted}} documenti; impossibile aggiungerne {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "Nessun documento è stato accettato: {{details}}", "newKnowledge.documentUploadStarted": "È iniziata l'elaborazione del documento.", diff --git a/web/i18n/ja-JP/dataset.json b/web/i18n/ja-JP/dataset.json index 79b000d6a0b..11410bc2269 100644 --- a/web/i18n/ja-JP/dataset.json +++ b/web/i18n/ja-JP/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "ワークスペースのクォータを超えています", "newKnowledge.documentUploadExclusion.target": "ドキュメントの対象が無効になっています", "newKnowledge.documentUploadFailed": "これらの書類をアップロードできませんでした。もう一度やり直してください。", - "newKnowledge.documentUploadFormats": "TXT、Markdown、PDF、HTML、XLSX、CSV、JSONL に対応 · 1ファイル最大15 MB", + "newKnowledge.documentUploadFormats": "テキスト、Markdown、HTML、PDF、Office、EPUB、メール、構造化データ(CSV、JSON/JSONL、XML)に対応 · 複雑な形式にはドキュメントパーサーが必要 · 1ファイル最大15 MB", "newKnowledge.documentUploadPartial": "{{accepted}} 件のドキュメントの処理を開始しました。{{excluded}} 件は追加できませんでした:{{details}}", "newKnowledge.documentUploadRejected": "ドキュメントを受け付けられませんでした:{{details}}", "newKnowledge.documentUploadStarted": "文書処理が開始されました。", diff --git a/web/i18n/ko-KR/dataset.json b/web/i18n/ko-KR/dataset.json index f23c0e7bfde..f05689aebca 100644 --- a/web/i18n/ko-KR/dataset.json +++ b/web/i18n/ko-KR/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "워크스페이스 할당량을 초과했습니다", "newKnowledge.documentUploadExclusion.target": "문서 대상이 더 이상 유효하지 않습니다", "newKnowledge.documentUploadFailed": "이 문서를 업로드할 수 없습니다. 다시 시도해 보세요.", - "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV, JSONL 지원 · 파일당 최대 15 MB", + "newKnowledge.documentUploadFormats": "텍스트, Markdown, HTML, PDF, Office, EPUB, 이메일 및 구조화 데이터(CSV, JSON/JSONL, XML) 지원 · 복잡한 형식에는 문서 파서 필요 · 파일당 최대 15 MB", "newKnowledge.documentUploadPartial": "문서 {{accepted}}개의 처리를 시작했습니다. {{excluded}}개는 추가하지 못했습니다: {{details}}", "newKnowledge.documentUploadRejected": "수락된 문서가 없습니다: {{details}}", "newKnowledge.documentUploadStarted": "문서 처리가 시작되었습니다.", diff --git a/web/i18n/lo-LA/dataset.json b/web/i18n/lo-LA/dataset.json index b3316d16c03..aa1aa6468e1 100644 --- a/web/i18n/lo-LA/dataset.json +++ b/web/i18n/lo-LA/dataset.json @@ -253,7 +253,7 @@ "newKnowledge.documentUploadExclusion.quota": "ເກີນໂຄຕ້າພື້ນທີ່ເຮັດວຽກ", "newKnowledge.documentUploadExclusion.target": "ເປົ້າໝາຍເອກະສານບໍ່ຖືກຕ້ອງອີກຕໍ່ໄປ", "newKnowledge.documentUploadFailed": "ພວກເຮົາບໍ່ສາມາດອັບໂຫລດເອກະສານເຫຼົ່ານີ້ໄດ້. ລອງອີກຄັ້ງ.", - "newKnowledge.documentUploadFormats": "ຮອງຮັບ TXT, Markdown, PDF, HTML, XLSX, CSV ແລະ JSONL · ສູງສຸດ 15 MB ຕໍ່ໄຟລ໌", + "newKnowledge.documentUploadFormats": "ຮອງຮັບຂໍ້ຄວາມ, Markdown, HTML, PDF, Office, EPUB, ອີເມວ ແລະຂໍ້ມູນທີ່ມີໂຄງສ້າງ (CSV, JSON/JSONL, XML) · ຮູບແບບທີ່ຊັບຊ້ອນຕ້ອງໃຊ້ຕົວແຍກວິເຄາະເອກະສານ · ສູງສຸດ 15 MB ຕໍ່ໄຟລ໌", "newKnowledge.documentUploadPartial": "ເອກະສານ {{accepted}} ລາຍການເລີ່ມແລ້ວ; ບໍ່ສາມາດເພີ່ມ {{excluded}} ລາຍການໄດ້: {{details}}", "newKnowledge.documentUploadRejected": "ບໍ່ມີເອກະສານໄດ້ຮັບການຍອມຮັບ: {{details}}", "newKnowledge.documentUploadStarted": "ການປະມວນຜົນເອກະສານໄດ້ເລີ່ມຕົ້ນ.", diff --git a/web/i18n/nl-NL/dataset.json b/web/i18n/nl-NL/dataset.json index 834a25b1ca8..702b45a1dff 100644 --- a/web/i18n/nl-NL/dataset.json +++ b/web/i18n/nl-NL/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "werkruimtequotum overschreden", "newKnowledge.documentUploadExclusion.target": "documentdoel is niet meer geldig", "newKnowledge.documentUploadFailed": "We konden deze documenten niet uploaden. Probeer het opnieuw.", - "newKnowledge.documentUploadFormats": "Ondersteunt TXT, Markdown, PDF, HTML, XLSX, CSV en JSONL · maximaal 15 MB per bestand", + "newKnowledge.documentUploadFormats": "Ondersteunt tekst, Markdown, HTML, PDF, Office, EPUB, e-mail en gestructureerde gegevens (CSV, JSON/JSONL, XML) · complexe indelingen vereisen een documentparser · maximaal 15 MB per bestand", "newKnowledge.documentUploadPartial": "Verwerking van {{accepted}} documenten gestart; {{excluded}} konden niet worden toegevoegd: {{details}}", "newKnowledge.documentUploadRejected": "Er zijn geen documenten geaccepteerd: {{details}}", "newKnowledge.documentUploadStarted": "Documentverwerking gestart.", diff --git a/web/i18n/pl-PL/dataset.json b/web/i18n/pl-PL/dataset.json index 2610f798340..e4d4b8be0c0 100644 --- a/web/i18n/pl-PL/dataset.json +++ b/web/i18n/pl-PL/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "przekroczono limit obszaru roboczego", "newKnowledge.documentUploadExclusion.target": "miejsce docelowe dokumentu jest już nieprawidłowe", "newKnowledge.documentUploadFailed": "Nie mogliśmy przesłać tych dokumentów. Spróbuj ponownie.", - "newKnowledge.documentUploadFormats": "Obsługuje TXT, Markdown, PDF, HTML, XLSX, CSV i JSONL · do 15 MB na plik", + "newKnowledge.documentUploadFormats": "Obsługuje tekst, Markdown, HTML, PDF, Office, EPUB, e-mail i dane strukturalne (CSV, JSON/JSONL, XML) · złożone formaty wymagają parsera dokumentów · do 15 MB na plik", "newKnowledge.documentUploadPartial": "Rozpoczęto przetwarzanie {{accepted}} dokumentów; nie udało się dodać {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "Nie zaakceptowano żadnych dokumentów: {{details}}", "newKnowledge.documentUploadStarted": "Rozpoczęto przetwarzanie dokumentu.", diff --git a/web/i18n/pt-BR/dataset.json b/web/i18n/pt-BR/dataset.json index 69757610d00..f6d48327b0a 100644 --- a/web/i18n/pt-BR/dataset.json +++ b/web/i18n/pt-BR/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "cota do espaço de trabalho excedida", "newKnowledge.documentUploadExclusion.target": "o destino do documento não é mais válido", "newKnowledge.documentUploadFailed": "Não foi possível fazer upload desses documentos. Tente novamente.", - "newKnowledge.documentUploadFormats": "Compatível com TXT, Markdown, PDF, HTML, XLSX, CSV e JSONL · até 15 MB cada", + "newKnowledge.documentUploadFormats": "Compatível com texto, Markdown, HTML, PDF, Office, EPUB, e-mail e dados estruturados (CSV, JSON/JSONL, XML) · formatos complexos exigem um analisador de documentos · até 15 MB por arquivo", "newKnowledge.documentUploadPartial": "O processamento de {{accepted}} documentos foi iniciado; não foi possível adicionar {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "Nenhum documento foi aceito: {{details}}", "newKnowledge.documentUploadStarted": "O processamento do documento foi iniciado.", diff --git a/web/i18n/ro-RO/dataset.json b/web/i18n/ro-RO/dataset.json index 3a1a14457e5..5418c796379 100644 --- a/web/i18n/ro-RO/dataset.json +++ b/web/i18n/ro-RO/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "cota spațiului de lucru a fost depășită", "newKnowledge.documentUploadExclusion.target": "destinația documentului nu mai este validă", "newKnowledge.documentUploadFailed": "Nu am putut încărca aceste documente. Încearcă din nou.", - "newKnowledge.documentUploadFormats": "Acceptă TXT, Markdown, PDF, HTML, XLSX, CSV și JSONL · maximum 15 MB fiecare", + "newKnowledge.documentUploadFormats": "Acceptă text, Markdown, HTML, PDF, Office, EPUB, e-mail și date structurate (CSV, JSON/JSONL, XML) · formatele complexe necesită un analizor de documente · maximum 15 MB per fișier", "newKnowledge.documentUploadPartial": "Procesarea a {{accepted}} documente a început; {{excluded}} nu au putut fi adăugate: {{details}}", "newKnowledge.documentUploadRejected": "Nu a fost acceptat niciun document: {{details}}", "newKnowledge.documentUploadStarted": "Procesarea documentelor a început.", diff --git a/web/i18n/ru-RU/dataset.json b/web/i18n/ru-RU/dataset.json index 0bc40b06848..13c87dcf8e6 100644 --- a/web/i18n/ru-RU/dataset.json +++ b/web/i18n/ru-RU/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "превышена квота рабочей области", "newKnowledge.documentUploadExclusion.target": "назначение документа больше недействительно", "newKnowledge.documentUploadFailed": "Нам не удалось загрузить эти документы. Попробуйте еще раз.", - "newKnowledge.documentUploadFormats": "Поддерживаются TXT, Markdown, PDF, HTML, XLSX, CSV и JSONL · до 15 МБ на файл", + "newKnowledge.documentUploadFormats": "Поддерживаются текст, Markdown, HTML, PDF, Office, EPUB, электронная почта и структурированные данные (CSV, JSON/JSONL, XML) · для сложных форматов требуется анализатор документов · до 15 МБ на файл", "newKnowledge.documentUploadPartial": "Начата обработка {{accepted}} документов; не удалось добавить {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "Ни один документ не принят: {{details}}", "newKnowledge.documentUploadStarted": "Началась обработка документов.", diff --git a/web/i18n/sl-SI/dataset.json b/web/i18n/sl-SI/dataset.json index 637e2c8356f..4211289e80f 100644 --- a/web/i18n/sl-SI/dataset.json +++ b/web/i18n/sl-SI/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "kvota delovnega prostora je presežena", "newKnowledge.documentUploadExclusion.target": "cilj dokumenta ni več veljaven", "newKnowledge.documentUploadFailed": "Teh dokumentov nismo mogli naložiti. poskusi ponovno", - "newKnowledge.documentUploadFormats": "Podpira TXT, Markdown, PDF, HTML, XLSX, CSV in JSONL · do 15 MB na datoteko", + "newKnowledge.documentUploadFormats": "Podpira besedilo, Markdown, HTML, PDF, Office, EPUB, e-pošto in strukturirane podatke (CSV, JSON/JSONL, XML) · zapletene oblike zahtevajo razčlenjevalnik dokumentov · do 15 MB na datoteko", "newKnowledge.documentUploadPartial": "Obdelava {{accepted}} dokumentov se je začela; {{excluded}} jih ni bilo mogoče dodati: {{details}}", "newKnowledge.documentUploadRejected": "Noben dokument ni bil sprejet: {{details}}", "newKnowledge.documentUploadStarted": "Začela se je obdelava dokumentov.", diff --git a/web/i18n/th-TH/dataset.json b/web/i18n/th-TH/dataset.json index 437a9c27602..9a3a7998854 100644 --- a/web/i18n/th-TH/dataset.json +++ b/web/i18n/th-TH/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "เกินโควตาพื้นที่ทำงาน", "newKnowledge.documentUploadExclusion.target": "เป้าหมายเอกสารไม่ถูกต้องอีกต่อไป", "newKnowledge.documentUploadFailed": "เราไม่สามารถอัปโหลดเอกสารเหล่านี้ได้ ลองอีกครั้ง", - "newKnowledge.documentUploadFormats": "รองรับ TXT, Markdown, PDF, HTML, XLSX, CSV และ JSONL · สูงสุดไฟล์ละ 15 MB", + "newKnowledge.documentUploadFormats": "รองรับข้อความ, Markdown, HTML, PDF, Office, EPUB, อีเมล และข้อมูลที่มีโครงสร้าง (CSV, JSON/JSONL, XML) · รูปแบบที่ซับซ้อนต้องใช้ตัวแยกวิเคราะห์เอกสาร · สูงสุดไฟล์ละ 15 MB", "newKnowledge.documentUploadPartial": "เริ่มประมวลผลเอกสาร {{accepted}} รายการแล้ว; เพิ่มไม่ได้ {{excluded}} รายการ: {{details}}", "newKnowledge.documentUploadRejected": "ไม่มีเอกสารที่ได้รับการยอมรับ: {{details}}", "newKnowledge.documentUploadStarted": "การประมวลผลเอกสารเริ่มต้นขึ้น", diff --git a/web/i18n/tr-TR/dataset.json b/web/i18n/tr-TR/dataset.json index e02dae071a4..b05a2973383 100644 --- a/web/i18n/tr-TR/dataset.json +++ b/web/i18n/tr-TR/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "çalışma alanı kotası aşıldı", "newKnowledge.documentUploadExclusion.target": "belge hedefi artık geçerli değil", "newKnowledge.documentUploadFailed": "Bu belgeleri yükleyemedik. Tekrar deneyin.", - "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV ve JSONL destekler · dosya başına en fazla 15 MB", + "newKnowledge.documentUploadFormats": "Metin, Markdown, HTML, PDF, Office, EPUB, e-posta ve yapılandırılmış verileri (CSV, JSON/JSONL, XML) destekler · karmaşık biçimler belge ayrıştırıcısı gerektirir · dosya başına en fazla 15 MB", "newKnowledge.documentUploadPartial": "{{accepted}} belgenin işlenmesi başladı; {{excluded}} belge eklenemedi: {{details}}", "newKnowledge.documentUploadRejected": "Hiçbir belge kabul edilmedi: {{details}}", "newKnowledge.documentUploadStarted": "Evrak işlemleri başlatıldı.", diff --git a/web/i18n/uk-UA/dataset.json b/web/i18n/uk-UA/dataset.json index 6476ad0779e..ae5f9a0a46a 100644 --- a/web/i18n/uk-UA/dataset.json +++ b/web/i18n/uk-UA/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "перевищено квоту робочої області", "newKnowledge.documentUploadExclusion.target": "ціль документа більше недійсна", "newKnowledge.documentUploadFailed": "Нам не вдалося завантажити ці документи. Спробуйте знову.", - "newKnowledge.documentUploadFormats": "Підтримуються TXT, Markdown, PDF, HTML, XLSX, CSV і JSONL · до 15 МБ на файл", + "newKnowledge.documentUploadFormats": "Підтримуються текст, Markdown, HTML, PDF, Office, EPUB, електронна пошта та структуровані дані (CSV, JSON/JSONL, XML) · для складних форматів потрібен аналізатор документів · до 15 МБ на файл", "newKnowledge.documentUploadPartial": "Розпочато обробку {{accepted}} документів; не вдалося додати {{excluded}}: {{details}}", "newKnowledge.documentUploadRejected": "Жодного документа не прийнято: {{details}}", "newKnowledge.documentUploadStarted": "Розпочато обробку документів.", diff --git a/web/i18n/vi-VN/dataset.json b/web/i18n/vi-VN/dataset.json index ab20e96870b..fd003efb9c1 100644 --- a/web/i18n/vi-VN/dataset.json +++ b/web/i18n/vi-VN/dataset.json @@ -260,7 +260,7 @@ "newKnowledge.documentUploadExclusion.quota": "đã vượt quá hạn mức không gian làm việc", "newKnowledge.documentUploadExclusion.target": "đích tài liệu không còn hợp lệ", "newKnowledge.documentUploadFailed": "Chúng tôi không thể tải lên những tài liệu này. Hãy thử lại.", - "newKnowledge.documentUploadFormats": "Hỗ trợ TXT, Markdown, PDF, HTML, XLSX, CSV và JSONL · tối đa 15 MB mỗi tệp", + "newKnowledge.documentUploadFormats": "Hỗ trợ văn bản, Markdown, HTML, PDF, Office, EPUB, email và dữ liệu có cấu trúc (CSV, JSON/JSONL, XML) · định dạng phức tạp cần trình phân tích tài liệu · tối đa 15 MB mỗi tệp", "newKnowledge.documentUploadPartial": "Đã bắt đầu xử lý {{accepted}} tài liệu; không thể thêm {{excluded}} tài liệu: {{details}}", "newKnowledge.documentUploadRejected": "Không có tài liệu nào được chấp nhận: {{details}}", "newKnowledge.documentUploadStarted": "Quá trình xử lý tài liệu bắt đầu.", diff --git a/web/i18n/zh-Hans/dataset.json b/web/i18n/zh-Hans/dataset.json index 09f2eda354a..f0411de2225 100644 --- a/web/i18n/zh-Hans/dataset.json +++ b/web/i18n/zh-Hans/dataset.json @@ -264,7 +264,7 @@ "newKnowledge.documentUploadExclusion.quota": "超出工作区配额", "newKnowledge.documentUploadExclusion.target": "文档目标已失效", "newKnowledge.documentUploadFailed": "无法上传这些文档,请重试。", - "newKnowledge.documentUploadFormats": "支持 TXT、Markdown、PDF、HTML、XLSX、CSV 和 JSONL · 每个文件不超过 15 MB", + "newKnowledge.documentUploadFormats": "支持文本、Markdown、HTML、PDF、Office、EPUB、邮件及结构化数据(CSV、JSON/JSONL、XML)· 复杂格式需要文档解析服务 · 每个文件不超过 15 MB", "newKnowledge.documentUploadPartial": "已开始处理 {{accepted}} 个文档;{{excluded}} 个无法添加:{{details}}", "newKnowledge.documentUploadRejected": "所有文档均未能添加:{{details}}", "newKnowledge.documentUploadStarted": "文档处理已开始。", diff --git a/web/i18n/zh-Hant/dataset.json b/web/i18n/zh-Hant/dataset.json index 06374e028fd..9075255a10d 100644 --- a/web/i18n/zh-Hant/dataset.json +++ b/web/i18n/zh-Hant/dataset.json @@ -263,7 +263,7 @@ "newKnowledge.documentUploadExclusion.quota": "超出工作區配額", "newKnowledge.documentUploadExclusion.target": "文件目標已失效", "newKnowledge.documentUploadFailed": "無法上傳這些文件,請再試一次。", - "newKnowledge.documentUploadFormats": "支援 TXT、Markdown、PDF、HTML、XLSX、CSV 和 JSONL · 每個檔案不超過 15 MB", + "newKnowledge.documentUploadFormats": "支援文字、Markdown、HTML、PDF、Office、EPUB、郵件及結構化資料(CSV、JSON/JSONL、XML)· 複雜格式需要文件解析服務 · 每個檔案不超過 15 MB", "newKnowledge.documentUploadPartial": "已開始處理 {{accepted}} 個文件;{{excluded}} 個無法新增:{{details}}", "newKnowledge.documentUploadRejected": "未接受任何文件:{{details}}", "newKnowledge.documentUploadStarted": "文件處理已開始。",