From d5fd45bb41c15a0cd087ad04f3efd83cd400848d Mon Sep 17 00:00:00 2001
From: Stephen Zhou <38493346+hyoban@users.noreply.github.com>
Date: Tue, 25 Aug 2026 19:58:15 +0800
Subject: [PATCH] feat(knowledge_fs): expand supported document upload formats
---
api/services/file_service.py | 2 +-
.../knowledge_fs/staged_upload_service.py | 36 ++++-
...test_knowledge_fs_staged_upload_service.py | 93 ++++++++++++-
...-08-25-expanded-document-upload-formats.md | 73 ++++++++++
.../apps/api/src/parser-options.test.ts | 2 +-
.../api/src/document-upload-utils.test.ts | 45 +++++++
.../packages/api/src/document-upload-utils.ts | 113 +++++++---------
knowledge-fs/packages/parsers/src/index.ts | 96 +++++++++++---
.../packages/parsers/src/parser.test.ts | 125 +++++++++++++++++-
.../new-rag/__tests__/documents-page.spec.tsx | 2 +-
.../new-rag/document-upload-policy.ts | 8 ++
web/i18n/ar-TN/dataset.json | 2 +-
web/i18n/de-DE/dataset.json | 2 +-
web/i18n/en-US/dataset.json | 2 +-
web/i18n/es-ES/dataset.json | 2 +-
web/i18n/fa-IR/dataset.json | 2 +-
web/i18n/fr-FR/dataset.json | 2 +-
web/i18n/hi-IN/dataset.json | 2 +-
web/i18n/id-ID/dataset.json | 2 +-
web/i18n/it-IT/dataset.json | 2 +-
web/i18n/ja-JP/dataset.json | 2 +-
web/i18n/ko-KR/dataset.json | 2 +-
web/i18n/lo-LA/dataset.json | 2 +-
web/i18n/nl-NL/dataset.json | 2 +-
web/i18n/pl-PL/dataset.json | 2 +-
web/i18n/pt-BR/dataset.json | 2 +-
web/i18n/ro-RO/dataset.json | 2 +-
web/i18n/ru-RU/dataset.json | 2 +-
web/i18n/sl-SI/dataset.json | 2 +-
web/i18n/th-TH/dataset.json | 2 +-
web/i18n/tr-TR/dataset.json | 2 +-
web/i18n/uk-UA/dataset.json | 2 +-
web/i18n/vi-VN/dataset.json | 2 +-
web/i18n/zh-Hans/dataset.json | 2 +-
web/i18n/zh-Hant/dataset.json | 2 +-
35 files changed, 530 insertions(+), 113 deletions(-)
create mode 100644 knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md
diff --git a/api/services/file_service.py b/api/services/file_service.py
index 4497639eb25..5d1261c0575 100644
--- a/api/services/file_service.py
+++ b/api/services/file_service.py
@@ -55,7 +55,7 @@ class FileService:
mimetype: str,
user: Account | EndUser,
tenant_id: str | None = None,
- source: Literal["datasets"] | None = None,
+ source: Literal["datasets", "knowledge_fs"] | None = None,
source_url: str = "",
default_file_size_limit: int | None = None,
) -> UploadFile:
diff --git a/api/services/knowledge_fs/staged_upload_service.py b/api/services/knowledge_fs/staged_upload_service.py
index c9e21f6849a..571fb893a2f 100644
--- a/api/services/knowledge_fs/staged_upload_service.py
+++ b/api/services/knowledge_fs/staged_upload_service.py
@@ -32,6 +32,33 @@ from services.knowledge_fs.product_dto import (
)
STAGED_UPLOAD_TTL = timedelta(hours=24)
+_KNOWLEDGE_FS_DOCUMENT_MIME_TYPES = {
+ "csv": "text/csv",
+ "doc": "application/msword",
+ "docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
+ "eml": "message/rfc822",
+ "epub": "application/epub+zip",
+ "htm": "text/html",
+ "html": "text/html",
+ "json": "application/json",
+ "jsonl": "application/x-ndjson",
+ "markdown": "text/markdown",
+ "md": "text/markdown",
+ "mdx": "text/mdx",
+ "msg": "application/vnd.ms-outlook",
+ "odt": "application/vnd.oasis.opendocument.text",
+ "pdf": "application/pdf",
+ "ppt": "application/vnd.ms-powerpoint",
+ "pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
+ "properties": "text/x-java-properties",
+ "rtf": "application/rtf",
+ "text": "text/plain",
+ "txt": "text/plain",
+ "vtt": "text/vtt",
+ "xls": "application/vnd.ms-excel",
+ "xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
+ "xml": "application/xml",
+}
class KnowledgeFSStagedUploadError(ValueError):
@@ -78,7 +105,12 @@ class KnowledgeFSStagedUploadService:
) -> KnowledgeFSStagedUploadResponse:
if not body:
raise KnowledgeFSStagedUploadInvalidError("KnowledgeFS staged upload is empty")
- normalized_content_type = content_type.strip() or "application/octet-stream"
+ _, separator, extension = file_name.strip().lower().rpartition(".")
+ if not separator or extension not in _KNOWLEDGE_FS_DOCUMENT_MIME_TYPES:
+ raise KnowledgeFSStagedUploadInvalidError("KnowledgeFS staged upload is invalid")
+ # Browser/OS MIME declarations are inconsistent and can route a binary document through a
+ # text parser. The admitted extension is the product contract, so persist its canonical MIME.
+ normalized_content_type = _KNOWLEDGE_FS_DOCUMENT_MIME_TYPES[extension]
checksum = b64encode(sha256(body).digest()).decode()
try:
upload_file = FileService(self._session_maker).upload_file(
@@ -87,7 +119,7 @@ class KnowledgeFSStagedUploadService:
mimetype=normalized_content_type,
user=account,
tenant_id=tenant_id,
- source="datasets",
+ source="knowledge_fs",
default_file_size_limit=file_size_limit_mb,
)
except FileTooLargeError as exc:
diff --git a/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py b/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py
index e196cfd457f..d4599208155 100644
--- a/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py
+++ b/api/tests/unit_tests/services/test_knowledge_fs_staged_upload_service.py
@@ -265,10 +265,10 @@ def test_stage_persists_workspace_owned_upload(
file_service.upload_file.assert_called_once_with(
filename="guide.pdf",
content=_BODY,
- mimetype="application/octet-stream",
+ mimetype="application/pdf",
user=account,
tenant_id=_TENANT_ID,
- source="datasets",
+ source="knowledge_fs",
default_file_size_limit=15,
)
with sqlite_session_factory() as session:
@@ -277,14 +277,73 @@ def test_stage_persists_workspace_owned_upload(
assert persisted.checksum_sha256_base64 == b64encode(sha256(_BODY).digest()).decode()
+@pytest.mark.parametrize(
+ ("file_name", "content_type", "expected_content_type"),
+ [
+ ("report.pdf", "text/plain", "application/pdf"),
+ ("formatted.rtf", "text/rtf", "application/rtf"),
+ ("message.msg", "application/x-msg", "application/vnd.ms-outlook"),
+ ],
+)
+def test_stage_canonicalizes_content_type_from_the_supported_extension(
+ sqlite_session_factory: sessionmaker[Session],
+ monkeypatch: pytest.MonkeyPatch,
+ file_name: str,
+ content_type: str,
+ expected_content_type: str,
+) -> None:
+ upload_file = _upload_file()
+ upload_file.name = file_name
+ upload_file.extension = file_name.rsplit(".", 1)[1]
+ upload_file.mime_type = expected_content_type
+ with sqlite_session_factory.begin() as session:
+ session.add(upload_file)
+ file_service = MagicMock()
+ file_service.upload_file.return_value = upload_file
+ monkeypatch.setattr(staged_upload_module, "FileService", lambda _: file_service)
+ service = KnowledgeFSStagedUploadService(
+ sqlite_session_factory,
+ facade=cast(KnowledgeFSDataFacade, MagicMock()),
+ )
+ account = cast(Account, SimpleNamespace(id=_ACCOUNT_ID))
+
+ response = service.stage(
+ tenant_id=_TENANT_ID,
+ account=account,
+ file_name=file_name,
+ content_type=content_type,
+ body=_BODY,
+ file_size_limit_mb=15,
+ )
+
+ assert response.content_type == expected_content_type
+ file_service.upload_file.assert_called_once_with(
+ filename=file_name,
+ content=_BODY,
+ mimetype=expected_content_type,
+ user=account,
+ tenant_id=_TENANT_ID,
+ source="knowledge_fs",
+ default_file_size_limit=15,
+ )
+
+
@pytest.mark.parametrize(
("file_name", "content_type", "body"),
[
("notes.txt", "text/plain", b"KnowledgeFS notes"),
("guide.md", "text/markdown", b"# KnowledgeFS guide"),
+ ("README.markdown", "text/markdown", b"# KnowledgeFS guide"),
+ ("component.mdx", "text/mdx", b"# KnowledgeFS component"),
+ ("captions.vtt", "text/vtt", b"WEBVTT\n\n00:00.000 --> 00:01.000\nKnowledgeFS"),
+ ("application.properties", "text/x-java-properties", b"knowledge.fs=enabled"),
+ ("feed.xml", "application/xml", b"KnowledgeFS"),
+ ("manual.odt", "application/vnd.oasis.opendocument.text", b"odt"),
+ ("message.eml", "message/rfc822", b"Subject: KnowledgeFS\n\nBody"),
+ ("message.msg", "application/vnd.ms-outlook", b"msg"),
],
)
-def test_stage_accepts_supported_text_files_with_the_real_file_service(
+def test_stage_accepts_knowledge_fs_document_formats_with_the_real_file_service(
sqlite_session_factory: sessionmaker[Session],
monkeypatch: pytest.MonkeyPatch,
file_name: str,
@@ -322,6 +381,34 @@ def test_stage_accepts_supported_text_files_with_the_real_file_service(
assert persisted.upload_file_id
+@pytest.mark.parametrize("file_name", ["malware.exe", "md"])
+def test_stage_rejects_an_unsupported_filename_with_the_real_file_service(
+ sqlite_session_factory: sessionmaker[Session], monkeypatch: pytest.MonkeyPatch, file_name: str
+) -> None:
+ backend = FakeStorage()
+ monkeypatch.setattr(file_service_module, "storage", backend)
+ monkeypatch.setattr(staged_upload_module, "storage", backend)
+ monkeypatch.setattr(file_service_module.file_helpers, "get_signed_file_url", lambda **_: "signed")
+ account = Account(name="KnowledgeFS tester", email="knowledge-fs@example.com")
+ account.id = _ACCOUNT_ID
+ service = KnowledgeFSStagedUploadService(
+ sqlite_session_factory,
+ facade=cast(KnowledgeFSDataFacade, MagicMock()),
+ )
+
+ with pytest.raises(KnowledgeFSStagedUploadInvalidError, match="invalid"):
+ service.stage(
+ tenant_id=_TENANT_ID,
+ account=account,
+ file_name=file_name,
+ content_type="application/octet-stream",
+ body=b"not executable content",
+ file_size_limit_mb=15,
+ )
+
+ assert backend.objects == {}
+
+
def test_stage_rejects_empty_and_maps_file_service_errors(
sqlite_session_factory: sessionmaker[Session], monkeypatch: pytest.MonkeyPatch
) -> None:
diff --git a/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md b/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md
new file mode 100644
index 00000000000..034ce69b3ac
--- /dev/null
+++ b/knowledge-fs/.harness/changes/2026-08-25-expanded-document-upload-formats.md
@@ -0,0 +1,73 @@
+# Expanded document upload formats
+
+## What changed
+
+- Added upload admission, MIME validation, and octet-stream inference for the legacy-compatible
+ `.markdown`, `.mdx`, `.vtt`, `.properties`, `.xml`, `.odt`, `.eml`, and `.msg` formats.
+- Routed VTT and Java properties files through the bounded native text parser. Markdown aliases and
+ XML continue to use the existing native Markdown and structured-data parsers; ODT, EML, and MSG
+ use the existing Unstructured parser boundary.
+- Kept the Dify New RAG file picker and local upload policy in sync with the KnowledgeFS service
+ allowlist.
+- Gave the Dify KnowledgeFS staging service the same explicit extension allowlist and a dedicated
+ `knowledge_fs` upload source, so staging no longer inherits the legacy knowledge-base `ETL_TYPE`
+ whitelist. Unsupported extensions are still rejected before storage writes.
+- Canonicalized staged-upload MIME types from the admitted extension, and changed direct
+ KnowledgeFS admission from two independent allowlists to an extension-to-MIME contract. Common
+ aliases such as `text/rtf` and JSONL declared as `application/json` remain accepted, while
+ unrelated pairs such as PDF plus `text/plain` are rejected.
+- Prioritized complex binary extensions in parser routing so an inaccurate browser MIME declaration
+ cannot send PDF, Office, EPUB, RTF, ODT, EML, or MSG content through the native text parser.
+- Preserved visible text inside MDX JSX blocks instead of silently dropping Marked's block HTML
+ tokens. MDX now carries its own `native-mdx@1` parser version so the behavior does not invalidate
+ existing plain-Markdown artifact hashes.
+- Updated upload guidance in every supported locale to describe the supported format groups and
+ disclose the complex-document parser dependency.
+- Added behavior tests for declared MIME types, octet-stream inference, native lightweight-text
+ routing, and the browser file-picker contract. The new tests were observed failing before the
+ implementation and passing afterward.
+
+## Why
+
+The new knowledge base rejected several formats already accepted by the legacy knowledge base even
+though its parser stack could process them. Expanding the allowlists and using the lightest existing
+parser restores compatibility without adding a new parser, storage path, or network dependency.
+
+## Verification
+
+- `pnpm --filter @knowledge/api exec vitest run src/document-upload-utils.test.ts` — passed (21 tests).
+- `pnpm --filter @knowledge/parsers exec vitest run src/parser.test.ts` — passed (55 tests).
+- `pnpm --filter @knowledge/parsers test:coverage` — passed with 95.69% statements/lines,
+ 90.02% branches, and 97.52% functions.
+- `pnpm --filter @knowledge/api-app exec vitest run src/parser-options.test.ts` — passed (5 tests).
+- `vp test run --project unit features/new-rag/__tests__/documents-page.spec.tsx` — passed (203 tests).
+- KnowledgeFS typechecks — passed; the full Turbo test pipeline passed (22 tasks), including the API
+ suite with 4,640 tests passed and 3 skipped.
+- Targeted KnowledgeFS Biome check for the five changed TypeScript files — passed.
+- Targeted Dify `vp check` for the two changed Web files — passed.
+- All 24 localized `dataset.json` files parsed successfully and contain the updated upload-format
+ guidance. The repository-wide dataset i18n alignment check remains blocked by pre-existing
+ missing KnowledgeFS quality-evaluation, task-failure, and related keys outside this change.
+- Dify KnowledgeFS staged-upload service test — passed (42 tests), including real `FileService`
+ coverage for canonical MIME persistence, expanded formats, and rejection of unsupported or
+ extensionless filenames.
+- Targeted Ruff format and lint checks for the three changed Python files — passed.
+- Targeted Pyrefly checks for the changed Python service files — passed.
+- Targeted Mypy was attempted but the installed Mypy 1.20.2 failed internally while reading its own
+ `typeshed/stdlib/zipimport.pyi`, before reporting project diagnostics.
+- KnowledgeFS `pnpm build` — passed; the existing Next.js multiple-lockfile and ESLint-plugin warnings
+ remain unchanged.
+- KnowledgeFS `pnpm lint` — attempted but remains blocked by pre-existing formatting/lint failures in
+ unrelated Admin, test setup, and generated contract files. No unrelated files were modified; the
+ targeted Biome check above covers every KnowledgeFS source and test file changed here.
+
+## Risks and follow-up
+
+- ODT, EML, and MSG parsing still requires a configured and capable Unstructured service, matching
+ other complex document types such as DOC and PPT. The upload guidance now calls out this
+ dependency; upload admission remains independent, while downstream parser failures continue to
+ use the existing failed-document lifecycle.
+- The added allowlist entries are fixed-size `Set` members. Admission remains constant-time and does
+ not change upload byte limits, buffering, database access, or object-storage behavior.
+- MDX JSX tags and attributes remain syntax rather than searchable text; visible child text is
+ retained, while `script`, `style`, and `noscript` contents remain excluded.
diff --git a/knowledge-fs/apps/api/src/parser-options.test.ts b/knowledge-fs/apps/api/src/parser-options.test.ts
index f53007e0982..5c0e9db1eee 100644
--- a/knowledge-fs/apps/api/src/parser-options.test.ts
+++ b/knowledge-fs/apps/api/src/parser-options.test.ts
@@ -67,7 +67,7 @@ describe("createApiDocumentParser", () => {
expect(requestedUrl).toBe("https://unstructured.example.test/general/v0/general");
expect(artifact).toMatchObject({
metadata: {
- routeReason: "unsupported-file-type",
+ routeReason: "complex-file-type",
routedParser: "unstructured",
},
parser: "unstructured",
diff --git a/knowledge-fs/packages/api/src/document-upload-utils.test.ts b/knowledge-fs/packages/api/src/document-upload-utils.test.ts
index 695376e2ee8..50b7de7ac98 100644
--- a/knowledge-fs/packages/api/src/document-upload-utils.test.ts
+++ b/knowledge-fs/packages/api/src/document-upload-utils.test.ts
@@ -280,6 +280,7 @@ describe("document upload utilities", () => {
"application/x-ndjson",
"application/jsonl",
"application/ndjson",
+ "application/json",
"application/octet-stream",
]) {
const result = await readBulkDocumentUploadWithAdmission(
@@ -302,6 +303,50 @@ describe("document upload utilities", () => {
}
});
+ it.each([
+ ["README.markdown", "text/markdown"],
+ ["component.mdx", "text/mdx"],
+ ["captions.vtt", "text/vtt"],
+ ["application.properties", "text/x-java-properties"],
+ ["formatted.rtf", "text/rtf"],
+ ["feed.xml", "application/xml"],
+ ["manual.odt", "application/vnd.oasis.opendocument.text"],
+ ["message.eml", "message/rfc822"],
+ ["message.msg", "application/vnd.ms-outlook"],
+ ])("accepts legacy-compatible document upload %s", async (filename, declaredMimeType) => {
+ for (const type of [declaredMimeType, "application/octet-stream"]) {
+ const result = await readBulkDocumentUploadWithAdmission(
+ {
+ parseBody: async () => ({
+ files: [new File(["content"], filename, { type })],
+ }),
+ },
+ {
+ maxAcceptedBytesByQuota: null,
+ maxBulkUploadBytes: 100,
+ maxBulkUploadFiles: 20,
+ maxUploadBytes: 100,
+ },
+ );
+
+ expect(result.accepted).toHaveLength(1);
+ expect(result.accepted[0]?.filename).toBe(filename);
+ }
+ });
+
+ it("rejects a supported extension paired with an unrelated MIME type", async () => {
+ await expect(
+ readDocumentUpload(
+ {
+ parseBody: async () => ({
+ file: new File(["%PDF-1.7"], "report.pdf", { type: "text/plain" }),
+ }),
+ },
+ 100,
+ ),
+ ).rejects.toThrow(DocumentUploadValidationError);
+ });
+
it("reports quota, aggregate-byte, and count exclusions without discarding earlier files", async () => {
const files = [
new File(["aa"], "a.txt", { type: "text/plain" }),
diff --git a/knowledge-fs/packages/api/src/document-upload-utils.ts b/knowledge-fs/packages/api/src/document-upload-utils.ts
index 1f42cf8e512..2171d5b749e 100644
--- a/knowledge-fs/packages/api/src/document-upload-utils.ts
+++ b/knowledge-fs/packages/api/src/document-upload-utils.ts
@@ -20,25 +20,42 @@ export interface BulkDocumentRevisionTarget {
readonly expectedDocumentRowVersion: number;
}
-export const SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES = new Set([
- "application/jsonl",
- "application/ndjson",
- "application/epub+zip",
- "application/json",
- "application/msword",
- "application/pdf",
- "application/rtf",
- "application/vnd.ms-excel",
- "application/vnd.ms-powerpoint",
- "application/vnd.openxmlformats-officedocument.presentationml.presentation",
- "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
- "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
- "application/x-ndjson",
- "text/csv",
- "text/html",
- "text/markdown",
- "text/plain",
-]);
+const DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION = {
+ csv: ["text/csv", "application/vnd.ms-excel"],
+ doc: ["application/msword"],
+ docx: ["application/vnd.openxmlformats-officedocument.wordprocessingml.document"],
+ eml: ["message/rfc822"],
+ epub: ["application/epub+zip"],
+ htm: ["text/html"],
+ html: ["text/html"],
+ json: ["application/json"],
+ jsonl: ["application/x-ndjson", "application/jsonl", "application/ndjson", "application/json"],
+ markdown: ["text/markdown", "text/x-markdown", "text/plain"],
+ md: ["text/markdown", "text/x-markdown", "text/plain"],
+ mdx: ["text/mdx", "text/markdown", "text/plain"],
+ msg: ["application/vnd.ms-outlook", "application/x-msg"],
+ odt: ["application/vnd.oasis.opendocument.text"],
+ pdf: ["application/pdf"],
+ ppt: ["application/vnd.ms-powerpoint", "application/mspowerpoint", "application/x-mspowerpoint"],
+ pptx: ["application/vnd.openxmlformats-officedocument.presentationml.presentation"],
+ properties: ["text/x-java-properties", "text/plain"],
+ rtf: ["application/rtf", "text/rtf", "application/x-rtf"],
+ text: ["text/plain"],
+ txt: ["text/plain"],
+ vtt: ["text/vtt", "text/plain"],
+ xls: [
+ "application/vnd.ms-excel",
+ "application/excel",
+ "application/x-excel",
+ "application/x-msexcel",
+ ],
+ xlsx: ["application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"],
+ xml: ["application/xml", "text/xml"],
+} as const satisfies Readonly>;
+
+export const SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES = new Set(
+ Object.values(DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION).flat(),
+);
export const DEFAULT_DOCUMENT_UPLOAD_MAX_BYTES = 15 * 1024 * 1024;
export const DEFAULT_BULK_DOCUMENT_UPLOAD_MAX_BYTES = 50 * 1024 * 1024;
@@ -48,25 +65,9 @@ export const HARD_BULK_DOCUMENT_UPLOAD_MAX_FILES = 25;
export const HARD_BULK_DOCUMENT_UPLOAD_MAX_BYTES =
HARD_DOCUMENT_UPLOAD_MAX_BYTES * HARD_BULK_DOCUMENT_UPLOAD_MAX_FILES;
-export const SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS = new Set([
- "csv",
- "doc",
- "docx",
- "epub",
- "htm",
- "html",
- "json",
- "jsonl",
- "md",
- "pdf",
- "ppt",
- "pptx",
- "rtf",
- "text",
- "txt",
- "xls",
- "xlsx",
-]);
+export const SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS = new Set(
+ Object.keys(DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION),
+);
export type DocumentUploadExclusionReason =
| "batch_byte_limit_exceeded"
@@ -539,27 +540,11 @@ function isJsonObject(value: unknown): value is Record {
export function normalizeDocumentMimeType(file: File): string {
const declared = file.type.trim().toLocaleLowerCase();
const extension = documentExtension(file.name);
- const inferred = (
- {
- csv: "text/csv",
- doc: "application/msword",
- docx: "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
- epub: "application/epub+zip",
- html: "text/html",
- htm: "text/html",
- json: "application/json",
- jsonl: "application/x-ndjson",
- md: "text/markdown",
- pdf: "application/pdf",
- ppt: "application/vnd.ms-powerpoint",
- pptx: "application/vnd.openxmlformats-officedocument.presentationml.presentation",
- rtf: "application/rtf",
- text: "text/plain",
- txt: "text/plain",
- xls: "application/vnd.ms-excel",
- xlsx: "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
- } as Readonly>
- )[extension ?? ""];
+ const inferred = extension
+ ? (DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION as Readonly>)[
+ extension
+ ]?.[0]
+ : undefined;
return !declared || declared === "application/octet-stream"
? (inferred ?? "application/octet-stream")
: declared;
@@ -567,11 +552,11 @@ export function normalizeDocumentMimeType(file: File): string {
function isSupportedDocumentUpload(file: File, mimeType: string): boolean {
const extension = documentExtension(file.name);
- return (
- SUPPORTED_DOCUMENT_UPLOAD_MIME_TYPES.has(mimeType) &&
- extension !== undefined &&
- SUPPORTED_DOCUMENT_UPLOAD_EXTENSIONS.has(extension)
- );
+ if (extension === undefined) return false;
+ const allowedMimeTypes = (
+ DOCUMENT_UPLOAD_MIME_TYPES_BY_EXTENSION as Readonly>
+ )[extension];
+ return allowedMimeTypes?.includes(mimeType) === true;
}
function documentExtension(filename: string): string | undefined {
diff --git a/knowledge-fs/packages/parsers/src/index.ts b/knowledge-fs/packages/parsers/src/index.ts
index c95e498228c..ad29251aec4 100644
--- a/knowledge-fs/packages/parsers/src/index.ts
+++ b/knowledge-fs/packages/parsers/src/index.ts
@@ -191,6 +191,20 @@ const defaultMaxRows = 20_000;
const defaultRetryDelayMs = 100;
const defaultNow = () => new Date().toISOString();
const defaultGenerateId = () => crypto.randomUUID();
+const unstructuredDocumentExtensions = new Set([
+ "doc",
+ "docx",
+ "eml",
+ "epub",
+ "msg",
+ "odt",
+ "pdf",
+ "ppt",
+ "pptx",
+ "rtf",
+ "xls",
+ "xlsx",
+]);
const UnstructuredElementSchema = z.object({
element_id: z.string().min(1).max(512).optional(),
@@ -209,11 +223,12 @@ export function createNativeMarkdownParser(options: NativeParserOptions = {}): P
return {
kind: "native-markdown",
parse: async (input) => {
- const parserVersion = options.parserVersion ?? "native-markdown@1";
+ const isMdx = isMdxInput(input);
+ const parserVersion = options.parserVersion ?? (isMdx ? "native-mdx@1" : "native-markdown@1");
assertInputBounds(input.body, options.maxInputBytes ?? defaultMaxInputBytes);
const text = decodeUtf8(input.body);
const tokens = marked.lexer(text, { gfm: true });
- const elements = markdownTokensToElements(tokens);
+ const elements = markdownTokensToElements(tokens, { preserveHtmlText: isMdx });
return createParseArtifact({
elements,
@@ -237,7 +252,7 @@ export function createNativeHtmlParser(options: NativeParserOptions = {}): Parse
lowerCaseAttributeNames: true,
lowerCaseTags: true,
});
- const nodes = (document.children ?? []) as HtmlNode[];
+ const nodes = document.children as HtmlNode[];
const elements = htmlNodesToElements(nodes);
const documentTitle = htmlDocumentTitle(nodes);
@@ -742,6 +757,10 @@ function selectParser(
return { parser: unstructured, reason: "unsupported-native-language" };
}
+ if (unstructuredDocumentExtensions.has(filename.split(".").at(-1) ?? "")) {
+ return { parser: unstructured, reason: "complex-file-type" };
+ }
+
const structuredFormat = structuredDataFormat(input);
if (structuredFormat && input.body.byteLength > maxNativeInputBytes) {
@@ -754,10 +773,15 @@ function selectParser(
const nativeParser =
mimeType === "text/markdown" ||
+ mimeType === "text/mdx" ||
mimeType === "text/plain" ||
+ mimeType === "text/vtt" ||
+ mimeType === "text/x-java-properties" ||
filename.endsWith(".md") ||
filename.endsWith(".markdown") ||
- filename.endsWith(".mdx")
+ filename.endsWith(".mdx") ||
+ filename.endsWith(".properties") ||
+ filename.endsWith(".vtt")
? markdown
: mimeType === "text/html" ||
mimeType === "application/xhtml+xml" ||
@@ -841,23 +865,22 @@ function structuredDataFormat({
return "csv";
}
- if (
- normalizedMime === "application/json" ||
- normalizedMime === "text/json" ||
- normalizedFilename.endsWith(".json")
- ) {
+ if (normalizedFilename.endsWith(".jsonl") || normalizedFilename.endsWith(".ndjson")) {
+ return "jsonl";
+ }
+
+ if (normalizedFilename.endsWith(".json")) {
return "json";
}
- if (
- normalizedMime === "application/x-ndjson" ||
- normalizedMime === "application/jsonl" ||
- normalizedFilename.endsWith(".jsonl") ||
- normalizedFilename.endsWith(".ndjson")
- ) {
+ if (normalizedMime === "application/x-ndjson" || normalizedMime === "application/jsonl") {
return "jsonl";
}
+ if (normalizedMime === "application/json" || normalizedMime === "text/json") {
+ return "json";
+ }
+
if (
normalizedMime === "application/yaml" ||
normalizedMime === "text/yaml" ||
@@ -1024,7 +1047,10 @@ function uniqueStrings(values: readonly string[]): string[] {
return [...new Set(values)];
}
-function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[] {
+function markdownTokensToElements(
+ tokens: readonly Token[],
+ { preserveHtmlText }: { readonly preserveHtmlText: boolean },
+): ParseElementInput[] {
const elements: ParseElementInput[] = [];
const sectionPath: string[] = [];
@@ -1078,6 +1104,12 @@ function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[]
continue;
}
+ if (token.type === "html" && preserveHtmlText) {
+ const html = token as Tokens.HTML;
+ pushTextElement(elements, "paragraph", markdownHtmlBlockText(html.text), sectionPath);
+ continue;
+ }
+
if (token.type === "list") {
const list = token as Tokens.List;
pushTextElement(
@@ -1106,6 +1138,38 @@ function markdownTokensToElements(tokens: readonly Token[]): ParseElementInput[]
return elements;
}
+function isMdxInput({
+ filename,
+ mimeType,
+}: Pick): boolean {
+ return (
+ mimeType.trim().toLowerCase() === "text/mdx" || filename.trim().toLowerCase().endsWith(".mdx")
+ );
+}
+
+function markdownHtmlBlockText(source: string): string {
+ const document = parseDocument(source, {
+ lowerCaseAttributeNames: true,
+ lowerCaseTags: true,
+ });
+ const nodes = document.children as HtmlNode[];
+
+ return nodes.map(searchableMarkdownHtmlText).join("\n");
+}
+
+function searchableMarkdownHtmlText(node: HtmlNode): string {
+ const name = node.name?.toLowerCase();
+ if (name && ["script", "style", "noscript"].includes(name)) {
+ return "";
+ }
+
+ if (!node.children?.length) {
+ return htmlText(node);
+ }
+
+ return node.children.map(searchableMarkdownHtmlText).join("\n");
+}
+
function htmlNodesToElements(nodes: readonly HtmlNode[]): ParseElementInput[] {
const elements: ParseElementInput[] = [];
const sectionPath: string[] = [];
diff --git a/knowledge-fs/packages/parsers/src/parser.test.ts b/knowledge-fs/packages/parsers/src/parser.test.ts
index a8fd425766b..9aa7b1a4760 100644
--- a/knowledge-fs/packages/parsers/src/parser.test.ts
+++ b/knowledge-fs/packages/parsers/src/parser.test.ts
@@ -87,6 +87,10 @@ describe("parser adapters", () => {
"const answer = 42;",
"```",
"",
+ "```",
+ "plain code block",
+ "```",
+ "",
"| A | B |",
"| - | - |",
"| 1 | 2 |",
@@ -143,12 +147,69 @@ describe("parser adapters", () => {
id: "018f0d60-7a49-7cc2-9c1b-5b36f18f2c45:element-5",
metadata: {},
sectionPath: ["Overview"],
+ text: "plain code block",
+ type: "code",
+ },
+ {
+ id: "018f0d60-7a49-7cc2-9c1b-5b36f18f2c45:element-6",
+ metadata: {},
+ sectionPath: ["Overview"],
text: "A | B\n1 | 2",
type: "table",
},
]);
});
+ it.each(["text/mdx", "text/plain"])(
+ "preserves searchable text inside MDX JSX blocks declared as %s",
+ async (mimeType) => {
+ const parser = createNativeMarkdownParser({
+ generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c95",
+ now: () => createdAt,
+ });
+
+ const artifact = await parser.parse(
+ createParseInput({
+ body: [
+ "# Overview",
+ "",
+ '',
+ "MDX keeps this searchable.",
+ "Nested detail",
+ "",
+ "",
+ ].join("\n"),
+ filename: "guide.mdx",
+ mimeType,
+ }),
+ );
+
+ expect(artifact.elements.map((element) => element.text)).toEqual([
+ "Overview",
+ "MDX keeps this searchable.\nNested detail",
+ ]);
+ expect(artifact.metadata.parserVersion).toBe("native-mdx@1");
+ },
+ );
+
+ it("keeps plain Markdown raw HTML behavior and parser version unchanged", async () => {
+ const parser = createNativeMarkdownParser({
+ generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c96",
+ now: () => createdAt,
+ });
+
+ const artifact = await parser.parse(
+ createParseInput({
+ body: ["", "Plain Markdown keeps its existing behavior.", ""].join("\n"),
+ filename: "guide.md",
+ mimeType: "text/markdown",
+ }),
+ );
+
+ expect(artifact.elements).toEqual([]);
+ expect(artifact.metadata.parserVersion).toBe("native-markdown@1");
+ });
+
it("normalizes Markdown image references into image parse elements", async () => {
const parser = createNativeMarkdownParser({
generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2d45",
@@ -541,8 +602,15 @@ describe("parser adapters", () => {
mimeType: "application/vnd.openxmlformats-officedocument.presentationml.presentation",
version: 1,
});
+ await router.parse({
+ body: textBytes("%PDF-1.7"),
+ documentAssetId,
+ filename: "report.pdf",
+ mimeType: "text/plain",
+ version: 1,
+ });
- expect(selected).toEqual(["markdown", "html", "unstructured"]);
+ expect(selected).toEqual(["markdown", "html", "unstructured", "unstructured"]);
});
it("routes by file size, OCR need, layout complexity, and language hints", async () => {
@@ -756,6 +824,31 @@ describe("parser adapters", () => {
});
});
+ it("uses the JSONL extension when the declared MIME type is application/json", async () => {
+ const parser = createNativeStructuredDataParser({
+ generateId: () => "018f0d60-7a49-7cc2-9c1b-5b36f18f2c5d",
+ now: () => createdAt,
+ });
+
+ await expect(
+ parser.parse(
+ createParseInput({
+ body: '{"name":"Ada"}\n{"name":"Lin"}',
+ filename: "records.jsonl",
+ mimeType: "application/json",
+ }),
+ ),
+ ).resolves.toMatchObject({
+ elements: [
+ {
+ metadata: { columns: ["name"], format: "jsonl", rowCount: 2 },
+ text: "name\nAda\nLin",
+ type: "table",
+ },
+ ],
+ });
+ });
+
it("routes structured data formats to the native structured parser", async () => {
const selected: string[] = [];
const structured = createNativeStructuredDataParser({
@@ -824,6 +917,36 @@ describe("parser adapters", () => {
});
});
+ it.each([
+ ["captions.vtt", "text/vtt"],
+ ["application.properties", "text/x-java-properties"],
+ ])("routes lightweight text format %s to the native text parser", async (filename, mimeType) => {
+ const router = createParserRouter({
+ html: createNativeHtmlParser(),
+ markdown: createNativeMarkdownParser(),
+ structured: createNativeStructuredDataParser(),
+ unstructured: {
+ kind: "unstructured",
+ parse: async () => {
+ throw new Error("lightweight text should not require Unstructured");
+ },
+ },
+ });
+
+ await expect(
+ router.parse(
+ createParseInput({
+ body: "first line\nsecond line",
+ filename,
+ mimeType,
+ }),
+ ),
+ ).resolves.toMatchObject({
+ metadata: { routeReason: "native-file-type", routedParser: "native-markdown" },
+ parser: "native-markdown",
+ });
+ });
+
it("rejects invalid or unbounded structured data inputs", async () => {
await expect(
createNativeStructuredDataParser({ maxRows: 1 }).parse(
diff --git a/web/features/new-rag/__tests__/documents-page.spec.tsx b/web/features/new-rag/__tests__/documents-page.spec.tsx
index 520761880b4..3c3ebc4d1dd 100644
--- a/web/features/new-rag/__tests__/documents-page.spec.tsx
+++ b/web/features/new-rag/__tests__/documents-page.spec.tsx
@@ -2106,7 +2106,7 @@ describe('DocumentsPage', () => {
expect(input).toHaveAttribute('tabindex', '-1')
expect(input).toHaveAttribute(
'accept',
- '.csv,.doc,.docx,.epub,.htm,.html,.json,.jsonl,.md,.pdf,.ppt,.pptx,.rtf,.text,.txt,.xls,.xlsx',
+ '.csv,.doc,.docx,.eml,.epub,.htm,.html,.json,.jsonl,.markdown,.md,.mdx,.msg,.odt,.pdf,.ppt,.pptx,.properties,.rtf,.text,.txt,.vtt,.xls,.xlsx,.xml',
)
await user.upload(input, new File(['one'], 'one.md', { type: 'text/markdown' }))
diff --git a/web/features/new-rag/document-upload-policy.ts b/web/features/new-rag/document-upload-policy.ts
index 2af46f765f3..82ddbd1ecc1 100644
--- a/web/features/new-rag/document-upload-policy.ts
+++ b/web/features/new-rag/document-upload-policy.ts
@@ -4,20 +4,28 @@ const DOCUMENT_UPLOAD_EXTENSIONS = [
'csv',
'doc',
'docx',
+ 'eml',
'epub',
'htm',
'html',
'json',
'jsonl',
+ 'markdown',
'md',
+ 'mdx',
+ 'msg',
+ 'odt',
'pdf',
'ppt',
'pptx',
+ 'properties',
'rtf',
'text',
'txt',
+ 'vtt',
'xls',
'xlsx',
+ 'xml',
] as const
const documentUploadExtensionSet = new Set(DOCUMENT_UPLOAD_EXTENSIONS)
diff --git a/web/i18n/ar-TN/dataset.json b/web/i18n/ar-TN/dataset.json
index cff14dfa622..b0320d4bfc8 100644
--- a/web/i18n/ar-TN/dataset.json
+++ b/web/i18n/ar-TN/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "تم تجاوز حصة مساحة العمل",
"newKnowledge.documentUploadExclusion.target": "لم يعد هدف المستند صالحًا",
"newKnowledge.documentUploadFailed": "لم نتمكن من تحميل هذه المستندات. حاول ثانية.",
- "newKnowledge.documentUploadFormats": "يدعم TXT وMarkdown وPDF وHTML وXLSX وCSV وJSONL · حتى 15 ميغابايت لكل ملف",
+ "newKnowledge.documentUploadFormats": "يدعم النصوص وMarkdown وHTML وPDF وOffice وEPUB والبريد الإلكتروني والبيانات المنظمة (CSV وJSON/JSONL وXML) · تتطلب التنسيقات المعقدة محلل مستندات · حتى 15 ميغابايت لكل ملف",
"newKnowledge.documentUploadPartial": "بدأت معالجة {{accepted}} مستندًا؛ تعذرت إضافة {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "لم يتم قبول أي مستند: {{details}}",
"newKnowledge.documentUploadStarted": "بدأت معالجة المستندات.",
diff --git a/web/i18n/de-DE/dataset.json b/web/i18n/de-DE/dataset.json
index 068b3f6acb9..8d3d778d028 100644
--- a/web/i18n/de-DE/dataset.json
+++ b/web/i18n/de-DE/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "Arbeitsbereichskontingent überschritten",
"newKnowledge.documentUploadExclusion.target": "Dokumentziel ist nicht mehr gültig",
"newKnowledge.documentUploadFailed": "Wir konnten diese Dokumente nicht hochladen. Versuchen Sie es erneut.",
- "newKnowledge.documentUploadFormats": "Unterstützt TXT, Markdown, PDF, HTML, XLSX, CSV und JSONL · jeweils bis zu 15 MB",
+ "newKnowledge.documentUploadFormats": "Unterstützt Text, Markdown, HTML, PDF, Office, EPUB, E-Mail und strukturierte Daten (CSV, JSON/JSONL, XML) · komplexe Formate erfordern einen Dokumentparser · bis zu 15 MB pro Datei",
"newKnowledge.documentUploadPartial": "{{accepted}} Dokumente wurden gestartet; {{excluded}} konnten nicht hinzugefügt werden: {{details}}",
"newKnowledge.documentUploadRejected": "Keine Dokumente wurden akzeptiert: {{details}}",
"newKnowledge.documentUploadStarted": "Die Dokumentenverarbeitung wurde gestartet.",
diff --git a/web/i18n/en-US/dataset.json b/web/i18n/en-US/dataset.json
index 54e04b143fc..6d7e62c6139 100644
--- a/web/i18n/en-US/dataset.json
+++ b/web/i18n/en-US/dataset.json
@@ -264,7 +264,7 @@
"newKnowledge.documentUploadExclusion.quota": "workspace quota exceeded",
"newKnowledge.documentUploadExclusion.target": "document target is no longer valid",
"newKnowledge.documentUploadFailed": "We couldn't upload these documents. Try again.",
- "newKnowledge.documentUploadFormats": "Supports TXT, Markdown, PDF, HTML, XLSX, CSV, and JSONL · up to 15MB each",
+ "newKnowledge.documentUploadFormats": "Supports text, Markdown, HTML, PDF, Office, EPUB, email, and structured data (CSV, JSON/JSONL, XML) · complex formats require a document parser · up to 15MB each",
"newKnowledge.documentUploadPartial": "{{accepted}} documents started; {{excluded}} could not be added: {{details}}",
"newKnowledge.documentUploadRejected": "No documents were accepted: {{details}}",
"newKnowledge.documentUploadStarted": "Document processing started.",
diff --git a/web/i18n/es-ES/dataset.json b/web/i18n/es-ES/dataset.json
index 68e0b880ce2..53e8477a3d5 100644
--- a/web/i18n/es-ES/dataset.json
+++ b/web/i18n/es-ES/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "se superó la cuota del espacio de trabajo",
"newKnowledge.documentUploadExclusion.target": "el destino del documento ya no es válido",
"newKnowledge.documentUploadFailed": "No pudimos cargar estos documentos. Intentar otra vez.",
- "newKnowledge.documentUploadFormats": "Admite TXT, Markdown, PDF, HTML, XLSX, CSV y JSONL · hasta 15 MB cada uno",
+ "newKnowledge.documentUploadFormats": "Admite texto, Markdown, HTML, PDF, Office, EPUB, correo electrónico y datos estructurados (CSV, JSON/JSONL, XML) · los formatos complejos requieren un analizador de documentos · hasta 15 MB por archivo",
"newKnowledge.documentUploadPartial": "Se inició el procesamiento de {{accepted}} documentos; no se pudieron añadir {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "No se aceptó ningún documento: {{details}}",
"newKnowledge.documentUploadStarted": "Se inició el procesamiento de documentos.",
diff --git a/web/i18n/fa-IR/dataset.json b/web/i18n/fa-IR/dataset.json
index b2542bc7fb6..6693908c642 100644
--- a/web/i18n/fa-IR/dataset.json
+++ b/web/i18n/fa-IR/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "سهمیه فضای کاری رد شده است",
"newKnowledge.documentUploadExclusion.target": "هدف سند دیگر معتبر نیست",
"newKnowledge.documentUploadFailed": "ما نتوانستیم این اسناد را آپلود کنیم. دوباره امتحان کنید.",
- "newKnowledge.documentUploadFormats": "پشتیبانی از TXT، Markdown، PDF، HTML، XLSX، CSV و JSONL · هر فایل تا ۱۵ مگابایت",
+ "newKnowledge.documentUploadFormats": "پشتیبانی از متن، Markdown، HTML، PDF، Office، EPUB، ایمیل و دادههای ساختیافته (CSV، JSON/JSONL، XML) · قالبهای پیچیده به تجزیهگر سند نیاز دارند · هر فایل تا ۱۵ مگابایت",
"newKnowledge.documentUploadPartial": "پردازش {{accepted}} سند آغاز شد؛ {{excluded}} سند افزوده نشد: {{details}}",
"newKnowledge.documentUploadRejected": "هیچ سندی پذیرفته نشد: {{details}}",
"newKnowledge.documentUploadStarted": "پردازش اسناد آغاز شد.",
diff --git a/web/i18n/fr-FR/dataset.json b/web/i18n/fr-FR/dataset.json
index cdea99f344f..12d4ab82d2e 100644
--- a/web/i18n/fr-FR/dataset.json
+++ b/web/i18n/fr-FR/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "quota de l’espace de travail dépassé",
"newKnowledge.documentUploadExclusion.target": "la cible du document n’est plus valide",
"newKnowledge.documentUploadFailed": "Nous n'avons pas pu télécharger ces documents. Essayer à nouveau.",
- "newKnowledge.documentUploadFormats": "Formats pris en charge : TXT, Markdown, PDF, HTML, XLSX, CSV et JSONL · 15 Mo max par fichier",
+ "newKnowledge.documentUploadFormats": "Formats pris en charge : texte, Markdown, HTML, PDF, Office, EPUB, e-mail et données structurées (CSV, JSON/JSONL, XML) · les formats complexes nécessitent un analyseur de documents · 15 Mo max par fichier",
"newKnowledge.documentUploadPartial": "Le traitement de {{accepted}} documents a démarré ; {{excluded}} n’ont pas pu être ajoutés : {{details}}",
"newKnowledge.documentUploadRejected": "Aucun document n’a été accepté : {{details}}",
"newKnowledge.documentUploadStarted": "Le traitement du document a commencé.",
diff --git a/web/i18n/hi-IN/dataset.json b/web/i18n/hi-IN/dataset.json
index d0493265c5c..4d525fcca1d 100644
--- a/web/i18n/hi-IN/dataset.json
+++ b/web/i18n/hi-IN/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "कार्यस्थान कोटा पार हो गया",
"newKnowledge.documentUploadExclusion.target": "दस्तावेज़ लक्ष्य अब मान्य नहीं है",
"newKnowledge.documentUploadFailed": "हम ये दस्तावेज़ अपलोड नहीं कर सके. पुनः प्रयास करें।",
- "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV और JSONL समर्थित · प्रत्येक फ़ाइल अधिकतम 15 MB",
+ "newKnowledge.documentUploadFormats": "टेक्स्ट, Markdown, HTML, PDF, Office, EPUB, ईमेल और संरचित डेटा (CSV, JSON/JSONL, XML) समर्थित · जटिल फ़ॉर्मैट के लिए दस्तावेज़ पार्सर आवश्यक है · प्रत्येक फ़ाइल अधिकतम 15 MB",
"newKnowledge.documentUploadPartial": "{{accepted}} दस्तावेज़ों की प्रोसेसिंग शुरू हुई; {{excluded}} जोड़े नहीं जा सके: {{details}}",
"newKnowledge.documentUploadRejected": "कोई दस्तावेज़ स्वीकार नहीं किया गया: {{details}}",
"newKnowledge.documentUploadStarted": "दस्तावेज़ प्रसंस्करण शुरू हुआ.",
diff --git a/web/i18n/id-ID/dataset.json b/web/i18n/id-ID/dataset.json
index 5a6fa2704e2..ca6da2e1e2a 100644
--- a/web/i18n/id-ID/dataset.json
+++ b/web/i18n/id-ID/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "kuota ruang kerja terlampaui",
"newKnowledge.documentUploadExclusion.target": "target dokumen tidak lagi valid",
"newKnowledge.documentUploadFailed": "Kami tidak dapat mengunggah dokumen-dokumen ini. Coba lagi.",
- "newKnowledge.documentUploadFormats": "Mendukung TXT, Markdown, PDF, HTML, XLSX, CSV, dan JSONL · masing-masing hingga 15 MB",
+ "newKnowledge.documentUploadFormats": "Mendukung teks, Markdown, HTML, PDF, Office, EPUB, email, dan data terstruktur (CSV, JSON/JSONL, XML) · format kompleks memerlukan pengurai dokumen · hingga 15 MB per file",
"newKnowledge.documentUploadPartial": "Pemrosesan {{accepted}} dokumen dimulai; {{excluded}} tidak dapat ditambahkan: {{details}}",
"newKnowledge.documentUploadRejected": "Tidak ada dokumen yang diterima: {{details}}",
"newKnowledge.documentUploadStarted": "Pemrosesan dokumen dimulai.",
diff --git a/web/i18n/it-IT/dataset.json b/web/i18n/it-IT/dataset.json
index 2b22b492764..f47cf80cc6f 100644
--- a/web/i18n/it-IT/dataset.json
+++ b/web/i18n/it-IT/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "quota dell'area di lavoro superata",
"newKnowledge.documentUploadExclusion.target": "la destinazione del documento non è più valida",
"newKnowledge.documentUploadFailed": "Non è stato possibile caricare questi documenti. Riprova.",
- "newKnowledge.documentUploadFormats": "Supporta TXT, Markdown, PDF, HTML, XLSX, CSV e JSONL · fino a 15 MB ciascuno",
+ "newKnowledge.documentUploadFormats": "Supporta testo, Markdown, HTML, PDF, Office, EPUB, e-mail e dati strutturati (CSV, JSON/JSONL, XML) · i formati complessi richiedono un parser di documenti · fino a 15 MB per file",
"newKnowledge.documentUploadPartial": "Elaborazione avviata per {{accepted}} documenti; impossibile aggiungerne {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "Nessun documento è stato accettato: {{details}}",
"newKnowledge.documentUploadStarted": "È iniziata l'elaborazione del documento.",
diff --git a/web/i18n/ja-JP/dataset.json b/web/i18n/ja-JP/dataset.json
index 79b000d6a0b..11410bc2269 100644
--- a/web/i18n/ja-JP/dataset.json
+++ b/web/i18n/ja-JP/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "ワークスペースのクォータを超えています",
"newKnowledge.documentUploadExclusion.target": "ドキュメントの対象が無効になっています",
"newKnowledge.documentUploadFailed": "これらの書類をアップロードできませんでした。もう一度やり直してください。",
- "newKnowledge.documentUploadFormats": "TXT、Markdown、PDF、HTML、XLSX、CSV、JSONL に対応 · 1ファイル最大15 MB",
+ "newKnowledge.documentUploadFormats": "テキスト、Markdown、HTML、PDF、Office、EPUB、メール、構造化データ(CSV、JSON/JSONL、XML)に対応 · 複雑な形式にはドキュメントパーサーが必要 · 1ファイル最大15 MB",
"newKnowledge.documentUploadPartial": "{{accepted}} 件のドキュメントの処理を開始しました。{{excluded}} 件は追加できませんでした:{{details}}",
"newKnowledge.documentUploadRejected": "ドキュメントを受け付けられませんでした:{{details}}",
"newKnowledge.documentUploadStarted": "文書処理が開始されました。",
diff --git a/web/i18n/ko-KR/dataset.json b/web/i18n/ko-KR/dataset.json
index f23c0e7bfde..f05689aebca 100644
--- a/web/i18n/ko-KR/dataset.json
+++ b/web/i18n/ko-KR/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "워크스페이스 할당량을 초과했습니다",
"newKnowledge.documentUploadExclusion.target": "문서 대상이 더 이상 유효하지 않습니다",
"newKnowledge.documentUploadFailed": "이 문서를 업로드할 수 없습니다. 다시 시도해 보세요.",
- "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV, JSONL 지원 · 파일당 최대 15 MB",
+ "newKnowledge.documentUploadFormats": "텍스트, Markdown, HTML, PDF, Office, EPUB, 이메일 및 구조화 데이터(CSV, JSON/JSONL, XML) 지원 · 복잡한 형식에는 문서 파서 필요 · 파일당 최대 15 MB",
"newKnowledge.documentUploadPartial": "문서 {{accepted}}개의 처리를 시작했습니다. {{excluded}}개는 추가하지 못했습니다: {{details}}",
"newKnowledge.documentUploadRejected": "수락된 문서가 없습니다: {{details}}",
"newKnowledge.documentUploadStarted": "문서 처리가 시작되었습니다.",
diff --git a/web/i18n/lo-LA/dataset.json b/web/i18n/lo-LA/dataset.json
index b3316d16c03..aa1aa6468e1 100644
--- a/web/i18n/lo-LA/dataset.json
+++ b/web/i18n/lo-LA/dataset.json
@@ -253,7 +253,7 @@
"newKnowledge.documentUploadExclusion.quota": "ເກີນໂຄຕ້າພື້ນທີ່ເຮັດວຽກ",
"newKnowledge.documentUploadExclusion.target": "ເປົ້າໝາຍເອກະສານບໍ່ຖືກຕ້ອງອີກຕໍ່ໄປ",
"newKnowledge.documentUploadFailed": "ພວກເຮົາບໍ່ສາມາດອັບໂຫລດເອກະສານເຫຼົ່ານີ້ໄດ້. ລອງອີກຄັ້ງ.",
- "newKnowledge.documentUploadFormats": "ຮອງຮັບ TXT, Markdown, PDF, HTML, XLSX, CSV ແລະ JSONL · ສູງສຸດ 15 MB ຕໍ່ໄຟລ໌",
+ "newKnowledge.documentUploadFormats": "ຮອງຮັບຂໍ້ຄວາມ, Markdown, HTML, PDF, Office, EPUB, ອີເມວ ແລະຂໍ້ມູນທີ່ມີໂຄງສ້າງ (CSV, JSON/JSONL, XML) · ຮູບແບບທີ່ຊັບຊ້ອນຕ້ອງໃຊ້ຕົວແຍກວິເຄາະເອກະສານ · ສູງສຸດ 15 MB ຕໍ່ໄຟລ໌",
"newKnowledge.documentUploadPartial": "ເອກະສານ {{accepted}} ລາຍການເລີ່ມແລ້ວ; ບໍ່ສາມາດເພີ່ມ {{excluded}} ລາຍການໄດ້: {{details}}",
"newKnowledge.documentUploadRejected": "ບໍ່ມີເອກະສານໄດ້ຮັບການຍອມຮັບ: {{details}}",
"newKnowledge.documentUploadStarted": "ການປະມວນຜົນເອກະສານໄດ້ເລີ່ມຕົ້ນ.",
diff --git a/web/i18n/nl-NL/dataset.json b/web/i18n/nl-NL/dataset.json
index 834a25b1ca8..702b45a1dff 100644
--- a/web/i18n/nl-NL/dataset.json
+++ b/web/i18n/nl-NL/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "werkruimtequotum overschreden",
"newKnowledge.documentUploadExclusion.target": "documentdoel is niet meer geldig",
"newKnowledge.documentUploadFailed": "We konden deze documenten niet uploaden. Probeer het opnieuw.",
- "newKnowledge.documentUploadFormats": "Ondersteunt TXT, Markdown, PDF, HTML, XLSX, CSV en JSONL · maximaal 15 MB per bestand",
+ "newKnowledge.documentUploadFormats": "Ondersteunt tekst, Markdown, HTML, PDF, Office, EPUB, e-mail en gestructureerde gegevens (CSV, JSON/JSONL, XML) · complexe indelingen vereisen een documentparser · maximaal 15 MB per bestand",
"newKnowledge.documentUploadPartial": "Verwerking van {{accepted}} documenten gestart; {{excluded}} konden niet worden toegevoegd: {{details}}",
"newKnowledge.documentUploadRejected": "Er zijn geen documenten geaccepteerd: {{details}}",
"newKnowledge.documentUploadStarted": "Documentverwerking gestart.",
diff --git a/web/i18n/pl-PL/dataset.json b/web/i18n/pl-PL/dataset.json
index 2610f798340..e4d4b8be0c0 100644
--- a/web/i18n/pl-PL/dataset.json
+++ b/web/i18n/pl-PL/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "przekroczono limit obszaru roboczego",
"newKnowledge.documentUploadExclusion.target": "miejsce docelowe dokumentu jest już nieprawidłowe",
"newKnowledge.documentUploadFailed": "Nie mogliśmy przesłać tych dokumentów. Spróbuj ponownie.",
- "newKnowledge.documentUploadFormats": "Obsługuje TXT, Markdown, PDF, HTML, XLSX, CSV i JSONL · do 15 MB na plik",
+ "newKnowledge.documentUploadFormats": "Obsługuje tekst, Markdown, HTML, PDF, Office, EPUB, e-mail i dane strukturalne (CSV, JSON/JSONL, XML) · złożone formaty wymagają parsera dokumentów · do 15 MB na plik",
"newKnowledge.documentUploadPartial": "Rozpoczęto przetwarzanie {{accepted}} dokumentów; nie udało się dodać {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "Nie zaakceptowano żadnych dokumentów: {{details}}",
"newKnowledge.documentUploadStarted": "Rozpoczęto przetwarzanie dokumentu.",
diff --git a/web/i18n/pt-BR/dataset.json b/web/i18n/pt-BR/dataset.json
index 69757610d00..f6d48327b0a 100644
--- a/web/i18n/pt-BR/dataset.json
+++ b/web/i18n/pt-BR/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "cota do espaço de trabalho excedida",
"newKnowledge.documentUploadExclusion.target": "o destino do documento não é mais válido",
"newKnowledge.documentUploadFailed": "Não foi possível fazer upload desses documentos. Tente novamente.",
- "newKnowledge.documentUploadFormats": "Compatível com TXT, Markdown, PDF, HTML, XLSX, CSV e JSONL · até 15 MB cada",
+ "newKnowledge.documentUploadFormats": "Compatível com texto, Markdown, HTML, PDF, Office, EPUB, e-mail e dados estruturados (CSV, JSON/JSONL, XML) · formatos complexos exigem um analisador de documentos · até 15 MB por arquivo",
"newKnowledge.documentUploadPartial": "O processamento de {{accepted}} documentos foi iniciado; não foi possível adicionar {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "Nenhum documento foi aceito: {{details}}",
"newKnowledge.documentUploadStarted": "O processamento do documento foi iniciado.",
diff --git a/web/i18n/ro-RO/dataset.json b/web/i18n/ro-RO/dataset.json
index 3a1a14457e5..5418c796379 100644
--- a/web/i18n/ro-RO/dataset.json
+++ b/web/i18n/ro-RO/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "cota spațiului de lucru a fost depășită",
"newKnowledge.documentUploadExclusion.target": "destinația documentului nu mai este validă",
"newKnowledge.documentUploadFailed": "Nu am putut încărca aceste documente. Încearcă din nou.",
- "newKnowledge.documentUploadFormats": "Acceptă TXT, Markdown, PDF, HTML, XLSX, CSV și JSONL · maximum 15 MB fiecare",
+ "newKnowledge.documentUploadFormats": "Acceptă text, Markdown, HTML, PDF, Office, EPUB, e-mail și date structurate (CSV, JSON/JSONL, XML) · formatele complexe necesită un analizor de documente · maximum 15 MB per fișier",
"newKnowledge.documentUploadPartial": "Procesarea a {{accepted}} documente a început; {{excluded}} nu au putut fi adăugate: {{details}}",
"newKnowledge.documentUploadRejected": "Nu a fost acceptat niciun document: {{details}}",
"newKnowledge.documentUploadStarted": "Procesarea documentelor a început.",
diff --git a/web/i18n/ru-RU/dataset.json b/web/i18n/ru-RU/dataset.json
index 0bc40b06848..13c87dcf8e6 100644
--- a/web/i18n/ru-RU/dataset.json
+++ b/web/i18n/ru-RU/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "превышена квота рабочей области",
"newKnowledge.documentUploadExclusion.target": "назначение документа больше недействительно",
"newKnowledge.documentUploadFailed": "Нам не удалось загрузить эти документы. Попробуйте еще раз.",
- "newKnowledge.documentUploadFormats": "Поддерживаются TXT, Markdown, PDF, HTML, XLSX, CSV и JSONL · до 15 МБ на файл",
+ "newKnowledge.documentUploadFormats": "Поддерживаются текст, Markdown, HTML, PDF, Office, EPUB, электронная почта и структурированные данные (CSV, JSON/JSONL, XML) · для сложных форматов требуется анализатор документов · до 15 МБ на файл",
"newKnowledge.documentUploadPartial": "Начата обработка {{accepted}} документов; не удалось добавить {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "Ни один документ не принят: {{details}}",
"newKnowledge.documentUploadStarted": "Началась обработка документов.",
diff --git a/web/i18n/sl-SI/dataset.json b/web/i18n/sl-SI/dataset.json
index 637e2c8356f..4211289e80f 100644
--- a/web/i18n/sl-SI/dataset.json
+++ b/web/i18n/sl-SI/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "kvota delovnega prostora je presežena",
"newKnowledge.documentUploadExclusion.target": "cilj dokumenta ni več veljaven",
"newKnowledge.documentUploadFailed": "Teh dokumentov nismo mogli naložiti. poskusi ponovno",
- "newKnowledge.documentUploadFormats": "Podpira TXT, Markdown, PDF, HTML, XLSX, CSV in JSONL · do 15 MB na datoteko",
+ "newKnowledge.documentUploadFormats": "Podpira besedilo, Markdown, HTML, PDF, Office, EPUB, e-pošto in strukturirane podatke (CSV, JSON/JSONL, XML) · zapletene oblike zahtevajo razčlenjevalnik dokumentov · do 15 MB na datoteko",
"newKnowledge.documentUploadPartial": "Obdelava {{accepted}} dokumentov se je začela; {{excluded}} jih ni bilo mogoče dodati: {{details}}",
"newKnowledge.documentUploadRejected": "Noben dokument ni bil sprejet: {{details}}",
"newKnowledge.documentUploadStarted": "Začela se je obdelava dokumentov.",
diff --git a/web/i18n/th-TH/dataset.json b/web/i18n/th-TH/dataset.json
index 437a9c27602..9a3a7998854 100644
--- a/web/i18n/th-TH/dataset.json
+++ b/web/i18n/th-TH/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "เกินโควตาพื้นที่ทำงาน",
"newKnowledge.documentUploadExclusion.target": "เป้าหมายเอกสารไม่ถูกต้องอีกต่อไป",
"newKnowledge.documentUploadFailed": "เราไม่สามารถอัปโหลดเอกสารเหล่านี้ได้ ลองอีกครั้ง",
- "newKnowledge.documentUploadFormats": "รองรับ TXT, Markdown, PDF, HTML, XLSX, CSV และ JSONL · สูงสุดไฟล์ละ 15 MB",
+ "newKnowledge.documentUploadFormats": "รองรับข้อความ, Markdown, HTML, PDF, Office, EPUB, อีเมล และข้อมูลที่มีโครงสร้าง (CSV, JSON/JSONL, XML) · รูปแบบที่ซับซ้อนต้องใช้ตัวแยกวิเคราะห์เอกสาร · สูงสุดไฟล์ละ 15 MB",
"newKnowledge.documentUploadPartial": "เริ่มประมวลผลเอกสาร {{accepted}} รายการแล้ว; เพิ่มไม่ได้ {{excluded}} รายการ: {{details}}",
"newKnowledge.documentUploadRejected": "ไม่มีเอกสารที่ได้รับการยอมรับ: {{details}}",
"newKnowledge.documentUploadStarted": "การประมวลผลเอกสารเริ่มต้นขึ้น",
diff --git a/web/i18n/tr-TR/dataset.json b/web/i18n/tr-TR/dataset.json
index e02dae071a4..b05a2973383 100644
--- a/web/i18n/tr-TR/dataset.json
+++ b/web/i18n/tr-TR/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "çalışma alanı kotası aşıldı",
"newKnowledge.documentUploadExclusion.target": "belge hedefi artık geçerli değil",
"newKnowledge.documentUploadFailed": "Bu belgeleri yükleyemedik. Tekrar deneyin.",
- "newKnowledge.documentUploadFormats": "TXT, Markdown, PDF, HTML, XLSX, CSV ve JSONL destekler · dosya başına en fazla 15 MB",
+ "newKnowledge.documentUploadFormats": "Metin, Markdown, HTML, PDF, Office, EPUB, e-posta ve yapılandırılmış verileri (CSV, JSON/JSONL, XML) destekler · karmaşık biçimler belge ayrıştırıcısı gerektirir · dosya başına en fazla 15 MB",
"newKnowledge.documentUploadPartial": "{{accepted}} belgenin işlenmesi başladı; {{excluded}} belge eklenemedi: {{details}}",
"newKnowledge.documentUploadRejected": "Hiçbir belge kabul edilmedi: {{details}}",
"newKnowledge.documentUploadStarted": "Evrak işlemleri başlatıldı.",
diff --git a/web/i18n/uk-UA/dataset.json b/web/i18n/uk-UA/dataset.json
index 6476ad0779e..ae5f9a0a46a 100644
--- a/web/i18n/uk-UA/dataset.json
+++ b/web/i18n/uk-UA/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "перевищено квоту робочої області",
"newKnowledge.documentUploadExclusion.target": "ціль документа більше недійсна",
"newKnowledge.documentUploadFailed": "Нам не вдалося завантажити ці документи. Спробуйте знову.",
- "newKnowledge.documentUploadFormats": "Підтримуються TXT, Markdown, PDF, HTML, XLSX, CSV і JSONL · до 15 МБ на файл",
+ "newKnowledge.documentUploadFormats": "Підтримуються текст, Markdown, HTML, PDF, Office, EPUB, електронна пошта та структуровані дані (CSV, JSON/JSONL, XML) · для складних форматів потрібен аналізатор документів · до 15 МБ на файл",
"newKnowledge.documentUploadPartial": "Розпочато обробку {{accepted}} документів; не вдалося додати {{excluded}}: {{details}}",
"newKnowledge.documentUploadRejected": "Жодного документа не прийнято: {{details}}",
"newKnowledge.documentUploadStarted": "Розпочато обробку документів.",
diff --git a/web/i18n/vi-VN/dataset.json b/web/i18n/vi-VN/dataset.json
index ab20e96870b..fd003efb9c1 100644
--- a/web/i18n/vi-VN/dataset.json
+++ b/web/i18n/vi-VN/dataset.json
@@ -260,7 +260,7 @@
"newKnowledge.documentUploadExclusion.quota": "đã vượt quá hạn mức không gian làm việc",
"newKnowledge.documentUploadExclusion.target": "đích tài liệu không còn hợp lệ",
"newKnowledge.documentUploadFailed": "Chúng tôi không thể tải lên những tài liệu này. Hãy thử lại.",
- "newKnowledge.documentUploadFormats": "Hỗ trợ TXT, Markdown, PDF, HTML, XLSX, CSV và JSONL · tối đa 15 MB mỗi tệp",
+ "newKnowledge.documentUploadFormats": "Hỗ trợ văn bản, Markdown, HTML, PDF, Office, EPUB, email và dữ liệu có cấu trúc (CSV, JSON/JSONL, XML) · định dạng phức tạp cần trình phân tích tài liệu · tối đa 15 MB mỗi tệp",
"newKnowledge.documentUploadPartial": "Đã bắt đầu xử lý {{accepted}} tài liệu; không thể thêm {{excluded}} tài liệu: {{details}}",
"newKnowledge.documentUploadRejected": "Không có tài liệu nào được chấp nhận: {{details}}",
"newKnowledge.documentUploadStarted": "Quá trình xử lý tài liệu bắt đầu.",
diff --git a/web/i18n/zh-Hans/dataset.json b/web/i18n/zh-Hans/dataset.json
index 09f2eda354a..f0411de2225 100644
--- a/web/i18n/zh-Hans/dataset.json
+++ b/web/i18n/zh-Hans/dataset.json
@@ -264,7 +264,7 @@
"newKnowledge.documentUploadExclusion.quota": "超出工作区配额",
"newKnowledge.documentUploadExclusion.target": "文档目标已失效",
"newKnowledge.documentUploadFailed": "无法上传这些文档,请重试。",
- "newKnowledge.documentUploadFormats": "支持 TXT、Markdown、PDF、HTML、XLSX、CSV 和 JSONL · 每个文件不超过 15 MB",
+ "newKnowledge.documentUploadFormats": "支持文本、Markdown、HTML、PDF、Office、EPUB、邮件及结构化数据(CSV、JSON/JSONL、XML)· 复杂格式需要文档解析服务 · 每个文件不超过 15 MB",
"newKnowledge.documentUploadPartial": "已开始处理 {{accepted}} 个文档;{{excluded}} 个无法添加:{{details}}",
"newKnowledge.documentUploadRejected": "所有文档均未能添加:{{details}}",
"newKnowledge.documentUploadStarted": "文档处理已开始。",
diff --git a/web/i18n/zh-Hant/dataset.json b/web/i18n/zh-Hant/dataset.json
index 06374e028fd..9075255a10d 100644
--- a/web/i18n/zh-Hant/dataset.json
+++ b/web/i18n/zh-Hant/dataset.json
@@ -263,7 +263,7 @@
"newKnowledge.documentUploadExclusion.quota": "超出工作區配額",
"newKnowledge.documentUploadExclusion.target": "文件目標已失效",
"newKnowledge.documentUploadFailed": "無法上傳這些文件,請再試一次。",
- "newKnowledge.documentUploadFormats": "支援 TXT、Markdown、PDF、HTML、XLSX、CSV 和 JSONL · 每個檔案不超過 15 MB",
+ "newKnowledge.documentUploadFormats": "支援文字、Markdown、HTML、PDF、Office、EPUB、郵件及結構化資料(CSV、JSON/JSONL、XML)· 複雜格式需要文件解析服務 · 每個檔案不超過 15 MB",
"newKnowledge.documentUploadPartial": "已開始處理 {{accepted}} 個文件;{{excluded}} 個無法新增:{{details}}",
"newKnowledge.documentUploadRejected": "未接受任何文件:{{details}}",
"newKnowledge.documentUploadStarted": "文件處理已開始。",