Feat/document bundles milestone 3 (#844)

* feat(document): support document bundles

* fix(document): avoid server extractor imports in client bundle

* fix(pdf): default mineru backend to pipeline

* fix(document): address bundle review findings

* fix(document): harden extraction limits and storage errors

* fix(document): make analysis step format agnostic

---------

Co-authored-by: Rowan_lxb <Lxb_savior@163.com>
This commit is contained in:
jackefn
2026-07-13 23:01:40 +08:00
committed by GitHub
co-authored by Rowan_lxb
parent c8a638a101
commit baff0998bb
28 changed files with 980 additions and 250 deletions
+3
View File
@@ -132,6 +132,9 @@ PDF_UNPDF_BASE_URL=
PDF_MINERU_API_KEY=
PDF_MINERU_BASE_URL=
# Optional. Defaults to "pipeline"; use "hybrid-auto-engine" only when your MinerU
# service has the required GPU/device configuration.
PDF_MINERU_BACKEND=
# --- Image Generation ---------------------------------------------------------
+10
View File
@@ -18,6 +18,7 @@ import { apiError, apiSuccess } from '@/lib/server/api-response';
import { validateUrlForSSRF } from '@/lib/server/ssrf-guard';
const log = createLogger('Extract Document');
const MAX_EXTRACT_DOCUMENT_FILE_SIZE_BYTES = 50 * 1024 * 1024;
function isPdfProviderId(providerId: string): providerId is PDFProviderId {
return providerId in PDF_PROVIDERS;
@@ -82,6 +83,15 @@ export async function POST(req: NextRequest) {
`Unsupported course material type for "${documentFile.name}"`,
);
}
if (documentFile.size > MAX_EXTRACT_DOCUMENT_FILE_SIZE_BYTES) {
return apiError(
'INVALID_REQUEST',
413,
`Course material file is too large. Maximum size is ${Math.floor(
MAX_EXTRACT_DOCUMENT_FILE_SIZE_BYTES / 1024 / 1024,
)}MB.`,
);
}
let provider = preferredProviderId
? getDocumentExtractorProvider(preferredProviderId)
+4 -1
View File
@@ -25,6 +25,7 @@ import { apiError, apiSuccess } from '@/lib/server/api-response';
import { llmApiError } from '@/lib/server/llm-error-response';
import { resolveModelFromRequest } from '@/lib/server/resolve-model';
import { resolveVocationalActive } from '@/lib/config/feature-flags';
import { sortDocumentImagesForVision } from '@/lib/document/bundle';
const log = createLogger('Scene Content API');
@@ -150,7 +151,9 @@ export async function POST(req: NextRequest) {
effectiveOutline.suggestedImageIds.length > 0
) {
const suggestedIds = new Set(effectiveOutline.suggestedImageIds);
assignedImages = pdfImages.filter((img) => suggestedIds.has(img.id));
assignedImages = sortDocumentImagesForVision(
pdfImages.filter((img) => suggestedIds.has(img.id)),
);
}
// ── Media generation is handled client-side in parallel (media-orchestrator.ts) ──
@@ -36,6 +36,7 @@ import type {
import { apiError } from '@/lib/server/api-response';
import { createLogger } from '@/lib/logger';
import { resolveModelFromRequest } from '@/lib/server/resolve-model';
import { sortDocumentImagesForVision } from '@/lib/document/bundle';
import { resolveVocationalActive } from '@/lib/config/feature-flags';
const log = createLogger('Outlines Stream');
@@ -326,10 +327,11 @@ export async function POST(req: NextRequest) {
if (pdfImages && pdfImages.length > 0) {
if (hasVision && imageMapping) {
// Vision mode: split into vision images (first N) and text-only (rest)
const allWithSrc = pdfImages.filter((img) => imageMapping[img.id]);
const sortedImages = sortDocumentImagesForVision(pdfImages);
const allWithSrc = sortedImages.filter((img) => imageMapping[img.id]);
const visionSlice = allWithSrc.slice(0, MAX_VISION_IMAGES);
const textOnlySlice = allWithSrc.slice(MAX_VISION_IMAGES);
const noSrcImages = pdfImages.filter((img) => !imageMapping[img.id]);
const noSrcImages = sortedImages.filter((img) => !imageMapping[img.id]);
const visionDescriptions = visionSlice.map((img) => formatImagePlaceholder(img));
const textDescriptions = [...textOnlySlice, ...noSrcImages].map((img) =>
+165 -104
View File
@@ -25,16 +25,27 @@ import { isAbortError } from '@/lib/generation/generation-retry';
import { FOREGROUND_SCENE_RETRY_OPTIONS } from './foreground-retry';
import {
loadImageMapping,
loadPdfBlob,
loadDocumentBlob,
cleanupOldImages,
storeImages,
} from '@/lib/utils/image-storage';
import { getCurrentModelConfig } from '@/lib/utils/model-config';
import { MAX_PDF_CONTENT_CHARS, MAX_VISION_IMAGES } from '@/lib/constants/generation';
import { MAX_VISION_IMAGES } from '@/lib/constants/generation';
import {
MAX_DOCUMENT_BUNDLE_FILES,
MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES,
buildDocumentBundle,
type ParsedDocumentPart,
} from '@/lib/document/bundle';
import { buildVideoManifestFromOutlines } from '@/lib/media/video-manifest';
import { nanoid } from 'nanoid';
import type { Stage } from '@/lib/types/stage';
import type { SceneOutline, PdfImage, ImageMapping } from '@/lib/types/generation';
import type {
SceneOutline,
PdfImage,
ImageMapping,
SessionDocumentSource,
} from '@/lib/types/generation';
import { AgentRevealModal } from '@/components/agent/agent-reveal-modal';
import { createLogger } from '@/lib/logger';
import {
@@ -49,6 +60,49 @@ import { resolveTaskEngineModeFromOutlineDoneEvent } from './vocational-mode';
const log = createLogger('GenerationPreview');
const OUTLINE_REVIEW_AUTO_CONTINUE_MS = 2500;
type ParsedDocumentResponseImage = {
id: string;
src?: string;
pageNumber?: number;
description?: string;
width?: number;
height?: number;
};
function legacySourceFromSession(session: GenerationSessionState): SessionDocumentSource[] {
if (session.documentSources?.length) return session.documentSources;
if (!session.pdfStorageKey) return [];
return [
{
id: 'source_1',
name: session.pdfFileName || 'document.pdf',
size: 0,
mimeType: session.documentMimeType || 'application/pdf',
order: 1,
storageKey: session.pdfStorageKey,
providerId: session.pdfProviderId,
},
];
}
function validateDocumentSources(
sources: SessionDocumentSource[],
t: (key: string, values?: Record<string, unknown>) => string,
) {
if (sources.length > MAX_DOCUMENT_BUNDLE_FILES) {
throw new Error(t('upload.courseMaterialCountLimit', { n: MAX_DOCUMENT_BUNDLE_FILES }));
}
const totalSize = sources.reduce((sum, source) => sum + source.size, 0);
if (totalSize > MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES) {
throw new Error(
t('upload.courseMaterialTotalSizeLimit', {
n: Math.floor(MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES / 1024 / 1024),
}),
);
}
}
type SceneGenerationFailure = {
error?: string;
errorCode?: string;
@@ -281,7 +335,8 @@ function GenerationPreviewContent() {
let activeSteps = getActiveSteps(currentSession);
// Determine if we need the document analysis step
const hasPdfToAnalyze = !!currentSession.pdfStorageKey && !currentSession.pdfText;
const documentSources = legacySourceFromSession(currentSession);
const hasPdfToAnalyze = documentSources.length > 0 && !currentSession.pdfText;
// If no document to analyze, skip to the next available step
if (!hasPdfToAnalyze) {
const firstNonPdfIdx = activeSteps.findIndex((s) => s.id !== 'pdf-analysis');
@@ -290,117 +345,120 @@ function GenerationPreviewContent() {
// Step 0: Extract uploaded course material if needed
if (hasPdfToAnalyze) {
log.debug('=== Generation Preview: Extracting course material ===');
const pdfBlob = await loadPdfBlob(currentSession.pdfStorageKey!);
if (!pdfBlob) {
throw new Error(t('generation.courseMaterialLoadFailed'));
}
log.debug('=== Generation Preview: Extracting course material bundle ===');
validateDocumentSources(documentSources, t);
const sortedDocumentSources = [...documentSources].sort((a, b) => a.order - b.order);
const parsedParts = await Promise.all(
sortedDocumentSources.map(async (source): Promise<ParsedDocumentPart> => {
const documentBlob = await loadDocumentBlob(source.storageKey);
if (!documentBlob) {
throw new Error(t('generation.courseMaterialLoadFailed'));
}
// Ensure pdfBlob is a valid Blob with content
if (!(pdfBlob instanceof Blob) || pdfBlob.size === 0) {
log.error('Invalid course material blob:', {
type: typeof pdfBlob,
size: pdfBlob instanceof Blob ? pdfBlob.size : 'N/A',
});
throw new Error(t('generation.courseMaterialLoadFailed'));
}
if (!(documentBlob instanceof Blob) || documentBlob.size === 0) {
log.error('Invalid course material blob:', {
source: source.name,
type: typeof documentBlob,
size: documentBlob instanceof Blob ? documentBlob.size : 'N/A',
});
throw new Error(t('generation.courseMaterialLoadFailed'));
}
// Wrap as a File to guarantee multipart/form-data with correct content-type
const pdfFile = new File([pdfBlob], currentSession.pdfFileName || 'document.pdf', {
type: currentSession.documentMimeType || pdfBlob.type || 'application/pdf',
});
const documentFile = new File([documentBlob], source.name || 'document.pdf', {
type: source.mimeType || documentBlob.type || 'application/pdf',
});
const parseFormData = new FormData();
parseFormData.append('file', pdfFile);
const parseFormData = new FormData();
parseFormData.append('file', documentFile);
if (currentSession.pdfProviderId) {
parseFormData.append('providerId', currentSession.pdfProviderId);
}
if (currentSession.pdfProviderConfig?.apiKey?.trim()) {
parseFormData.append('apiKey', currentSession.pdfProviderConfig.apiKey);
}
if (currentSession.pdfProviderConfig?.baseUrl?.trim()) {
parseFormData.append('baseUrl', currentSession.pdfProviderConfig.baseUrl);
}
const providerId = source.providerId || currentSession.pdfProviderId;
const legacySourceConfig = (
source as SessionDocumentSource & {
providerConfig?: { apiKey?: string; baseUrl?: string };
}
).providerConfig;
const providerConfig = currentSession.pdfProviderConfig || legacySourceConfig;
if (providerId) parseFormData.append('providerId', providerId);
if (providerConfig?.apiKey?.trim()) {
parseFormData.append('apiKey', providerConfig.apiKey);
}
if (providerConfig?.baseUrl?.trim()) {
parseFormData.append('baseUrl', providerConfig.baseUrl);
}
const parseResponse = await fetch('/api/extract-document', {
method: 'POST',
body: parseFormData,
signal,
});
const parseResponse = await fetch('/api/extract-document', {
method: 'POST',
body: parseFormData,
signal,
});
if (!parseResponse.ok) {
const errorData = await parseResponse.json();
throw new Error(errorData.error || t('generation.courseMaterialParseFailed'));
}
if (!parseResponse.ok) {
const errorData = await parseResponse.json();
throw new Error(errorData.error || t('generation.courseMaterialParseFailed'));
}
const parseResult = await parseResponse.json();
if (!parseResult.success || !parseResult.data) {
throw new Error(t('generation.courseMaterialParseFailed'));
}
const parseResult = await parseResponse.json();
if (!parseResult.success || !parseResult.data) {
throw new Error(t('generation.courseMaterialParseFailed'));
}
let pdfText = parseResult.data.text as string;
const rawImages = parseResult.data.metadata?.pdfImages;
const images = rawImages
? rawImages.map((img: ParsedDocumentResponseImage) => ({
id: img.id,
src: img.src || '',
pageNumber: img.pageNumber ?? 1,
description: img.description,
width: img.width,
height: img.height,
}))
: ((parseResult.data.images as string[] | undefined) ?? []).map((src, i) => ({
id: `img_${i + 1}`,
src,
pageNumber: 1,
}));
// Truncate if needed
if (pdfText.length > MAX_PDF_CONTENT_CHARS) {
pdfText = pdfText.substring(0, MAX_PDF_CONTENT_CHARS);
}
// Create image metadata and store images
// Prefer metadata.pdfImages (both parsers now return this)
const rawPdfImages = parseResult.data.metadata?.pdfImages;
const images = rawPdfImages
? rawPdfImages.map(
(img: {
id: string;
src?: string;
pageNumber?: number;
description?: string;
width?: number;
height?: number;
}) => ({
id: img.id,
src: img.src || '',
pageNumber: img.pageNumber || 1,
description: img.description,
width: img.width,
height: img.height,
}),
)
: (parseResult.data.images as string[]).map((src: string, i: number) => ({
id: `img_${i + 1}`,
src,
pageNumber: 1,
}));
const imageStorageIds = await storeImages(images);
const pdfImages: PdfImage[] = images.map(
(
img: {
id: string;
src: string;
pageNumber: number;
description?: string;
width?: number;
height?: number;
},
i: number,
) => ({
id: img.id,
src: '',
pageNumber: img.pageNumber,
description: img.description,
width: img.width,
height: img.height,
storageId: imageStorageIds[i],
return {
source: {
id: source.id,
name: source.name,
size: source.size,
lastModified: source.lastModified,
mimeType: source.mimeType,
order: source.order,
providerId,
},
text: parseResult.data.text as string,
rawTextLength: (parseResult.data.text as string).length,
pageCount: parseResult.data.metadata?.pageCount,
images,
};
}),
);
const bundle = buildDocumentBundle(parsedParts);
const imageStorageIds = await storeImages(bundle.images);
const pdfImages: PdfImage[] = bundle.images.map((img, i) => ({
id: img.id,
src: '',
pageNumber: img.pageNumber,
description: img.description,
width: img.width,
height: img.height,
originalId: img.originalId,
sourceDocumentId: img.sourceDocumentId,
sourceDocumentName: img.sourceDocumentName,
sourceDocumentOrder: img.sourceDocumentOrder,
visionPriority: img.visionPriority,
storageId: imageStorageIds[i],
}));
// Update session with extracted document data
const updatedSession = {
...currentSession,
pdfText,
documentSources,
pdfText: bundle.text,
pdfImages,
imageStorageIds,
pdfStorageKey: undefined, // Clear so we don't re-parse
@@ -410,12 +468,15 @@ function GenerationPreviewContent() {
// Truncation warnings
const warnings: string[] = [];
if ((parseResult.data.text as string).length > MAX_PDF_CONTENT_CHARS) {
warnings.push(t('generation.textTruncated', { n: MAX_PDF_CONTENT_CHARS }));
if (bundle.totalRawTextLength > bundle.textContentBudget) {
warnings.push(t('generation.textTruncated', { n: bundle.textContentBudget }));
}
if (images.length > MAX_VISION_IMAGES) {
if (bundle.totalImageCount > MAX_VISION_IMAGES) {
warnings.push(
t('generation.imageTruncated', { total: images.length, max: MAX_VISION_IMAGES }),
t('generation.imageTruncated', {
total: bundle.totalImageCount,
max: MAX_VISION_IMAGES,
}),
);
}
if (warnings.length > 0) {
+10 -22
View File
@@ -5,6 +5,7 @@ import type {
UserRequirements,
PdfImage,
ImageMapping,
SessionDocumentSource,
} from '@/lib/types/generation';
// Session state stored in sessionStorage
@@ -12,6 +13,7 @@ export interface GenerationSessionState {
sessionId: string;
requirements: UserRequirements;
pdfText: string;
documentSources?: SessionDocumentSource[];
pdfImages?: PdfImage[];
imageStorageIds?: string[];
imageMapping?: ImageMapping;
@@ -43,33 +45,14 @@ export type GenerationStep = {
type: 'analysis' | 'writing' | 'visual';
};
function getDocumentTypeLabel(session: GenerationSessionState | null): string {
const mimeType = session?.documentMimeType;
if (mimeType) {
if (mimeType === 'application/pdf') return 'PDF';
if (mimeType.includes('wordprocessingml')) return 'DOCX';
if (mimeType.includes('presentationml')) return 'PPTX';
if (mimeType === 'text/plain') return 'TXT';
if (mimeType.includes('markdown')) return 'Markdown';
}
const extension = session?.pdfFileName?.split('.').pop()?.trim().toLowerCase();
if (extension === 'pdf') return 'PDF';
if (extension === 'docx') return 'DOCX';
if (extension === 'pptx') return 'PPTX';
if (extension === 'txt') return 'TXT';
if (extension === 'md' || extension === 'markdown') return 'Markdown';
return 'document';
}
export function getGenerationStepText(
step: GenerationStep,
session: GenerationSessionState | null,
_session: GenerationSessionState | null,
) {
if (step.id === 'pdf-analysis') {
const documentType = getDocumentTypeLabel(session);
return {
title: 'generation.analyzingCourseMaterial',
titleValues: { type: documentType },
titleValues: undefined,
description: 'generation.analyzingCourseMaterialDesc',
};
}
@@ -127,7 +110,12 @@ export const ALL_STEPS: GenerationStep[] = [
export const getActiveSteps = (session: GenerationSessionState | null) => {
return ALL_STEPS.filter((step) => {
if (step.id === 'pdf-analysis') return !!session?.pdfStorageKey;
if (step.id === 'pdf-analysis') {
return Boolean(
session?.pdfStorageKey ||
((session?.documentSources?.length ?? 0) > 0 && !session?.pdfText),
);
}
if (step.id === 'web-search') return !!session?.requirements?.webSearch;
if (step.id === 'agent-generation') return useSettingsStore.getState().agentMode === 'auto';
return true;
+74 -25
View File
@@ -36,9 +36,14 @@ import { GenerationToolbar } from '@/components/generation/generation-toolbar';
import { AgentBar } from '@/components/agent/agent-bar';
import { useTheme } from '@/lib/hooks/use-theme';
import { nanoid } from 'nanoid';
import { storePdfBlob } from '@/lib/utils/image-storage';
import { deleteDocumentBlob, storeDocumentBlob } from '@/lib/utils/image-storage';
import { normalizeDocumentMimeType } from '@/lib/document/mime';
import type { UserRequirements } from '@/lib/types/generation';
import { dedupeCourseMaterialFiles } from '@/lib/document/course-materials';
import type {
SelectedCourseMaterial,
SessionDocumentSource,
UserRequirements,
} from '@/lib/types/generation';
import { useSettingsStore } from '@/lib/store/settings';
import { hasUsableLLMProvider } from '@/lib/store/settings-validation';
import { useUserProfileStore, AVATAR_OPTIONS } from '@/lib/store/user-profile';
@@ -75,7 +80,7 @@ const INTERACTIVE_MODE_STORAGE_KEY = 'interactiveModeEnabled';
const PPTX_IMPORT_ENABLED = process.env.NEXT_PUBLIC_ENABLE_PPTX_IMPORT === 'true';
interface FormState {
pdfFile: File | null;
courseMaterials: SelectedCourseMaterial[];
requirement: string;
webSearch: boolean;
interactiveMode: boolean;
@@ -83,7 +88,7 @@ interface FormState {
}
const initialFormState: FormState = {
pdfFile: null,
courseMaterials: [],
requirement: '',
webSearch: false,
interactiveMode: false,
@@ -121,7 +126,6 @@ function HomePage() {
};
// Hydrate client-only state after mount (avoids SSR mismatch)
/* eslint-disable react-hooks/set-state-in-effect -- Hydration from localStorage must happen in effect */
useEffect(() => {
try {
const saved = localStorage.getItem(RECENT_OPEN_STORAGE_KEY);
@@ -142,21 +146,18 @@ function HomePage() {
/* localStorage unavailable */
}
}, []);
/* eslint-enable react-hooks/set-state-in-effect */
// Restore requirement draft from localStorage on mount. The previous derived-state
// pattern initialised `prev` from the cached value itself, so on the first client
// render the comparison was always equal and the restore never fired. Use an effect
// so the cache is hydrated into the form once we know the live requirement is empty.
const draftRestoredRef = useRef(false);
/* eslint-disable react-hooks/set-state-in-effect -- Hydration from localStorage must happen in effect */
useEffect(() => {
if (draftRestoredRef.current) return;
if (!cachedRequirement) return;
draftRestoredRef.current = true;
setForm((prev) => (prev.requirement ? prev : { ...prev, requirement: cachedRequirement }));
}, [cachedRequirement]);
/* eslint-enable react-hooks/set-state-in-effect */
const [themeOpen, setThemeOpen] = useState(false);
const [error, setError] = useState<string | null>(null);
@@ -226,7 +227,6 @@ function HomePage() {
useMediaGenerationStore.getState().revokeObjectUrls();
useMediaGenerationStore.setState({ tasks: {} });
// eslint-disable-next-line react-hooks/set-state-in-effect -- Store hydration on mount
loadClassrooms();
return () => {
@@ -284,6 +284,35 @@ function HomePage() {
}
};
const addCourseMaterials = (files: File[]) => {
setForm((prev) => {
const dedupedFiles = dedupeCourseMaterialFiles(prev.courseMaterials, files);
const startOrder = prev.courseMaterials.length + 1;
const additions = dedupedFiles.map((file, index) => ({
id: nanoid(8),
file,
name: file.name,
size: file.size,
lastModified: file.lastModified,
type: file.type,
order: startOrder + index,
}));
return additions.length > 0
? { ...prev, courseMaterials: [...prev.courseMaterials, ...additions] }
: prev;
});
};
const removeCourseMaterial = (id: string) => {
setForm((prev) => ({
...prev,
courseMaterials: prev.courseMaterials
.filter((item) => item.id !== id)
.map((item, index) => ({ ...item, order: index + 1 })),
}));
};
const handleGenerate = async () => {
// No model/provider guard here: generation is gated by `canGenerate`
// (requires a usable provider), and under the #580 invariant a usable
@@ -307,20 +336,11 @@ function HomePage() {
...(form.vocationalTestMode ? { taskEngineMode: true } : {}),
};
let pdfStorageKey: string | undefined;
let pdfFileName: string | undefined;
let documentMimeType: string | undefined;
let documentSources: SessionDocumentSource[] | undefined;
let pdfProviderId: string | undefined;
let pdfProviderConfig: { apiKey?: string; baseUrl?: string } | undefined;
if (form.pdfFile) {
pdfStorageKey = await storePdfBlob(form.pdfFile);
pdfFileName = form.pdfFile.name;
documentMimeType = normalizeDocumentMimeType({
mimeType: form.pdfFile.type,
fileName: form.pdfFile.name,
});
if (form.courseMaterials.length > 0) {
const settings = useSettingsStore.getState();
pdfProviderId = settings.pdfProviderId;
const providerCfg = settings.pdfProvidersConfig?.[settings.pdfProviderId];
@@ -330,6 +350,32 @@ function HomePage() {
baseUrl: providerCfg.baseUrl,
};
}
const storedDocumentKeys: string[] = [];
try {
documentSources = [];
const orderedMaterials = [...form.courseMaterials].sort((a, b) => a.order - b.order);
for (const [index, item] of orderedMaterials.entries()) {
const storageKey = await storeDocumentBlob(item.file);
storedDocumentKeys.push(storageKey);
documentSources.push({
id: item.id,
name: item.name,
size: item.size,
lastModified: item.lastModified,
mimeType: normalizeDocumentMimeType({
mimeType: item.file.type,
fileName: item.file.name,
}),
order: index + 1,
storageKey,
providerId: pdfProviderId,
});
}
} catch (error) {
await Promise.allSettled(storedDocumentKeys.map((key) => deleteDocumentBlob(key)));
throw error;
}
}
const sessionState = {
@@ -338,9 +384,11 @@ function HomePage() {
pdfText: '',
pdfImages: [],
imageStorageIds: [],
pdfStorageKey,
pdfFileName,
documentMimeType,
documentSources,
// Backward-compatible single-document fields for previously saved sessions.
pdfStorageKey: documentSources?.[0]?.storageKey,
pdfFileName: documentSources?.[0]?.name,
documentMimeType: documentSources?.[0]?.mimeType,
pdfProviderId,
pdfProviderConfig,
sceneOutlines: null,
@@ -569,8 +617,9 @@ function HomePage() {
setSettingsSection(section);
setSettingsOpen(true);
}}
pdfFile={form.pdfFile}
onPdfFileChange={(f) => updateForm('pdfFile', f)}
courseMaterials={form.courseMaterials}
onCourseMaterialsAdd={addCourseMaterials}
onCourseMaterialRemove={removeCourseMaterial}
onPdfError={setError}
/>
</div>
+125 -69
View File
@@ -37,6 +37,12 @@ import {
import type { SettingsSection } from '@/lib/types/settings';
import { MediaPopover } from '@/components/generation/media-popover';
import { getAcceptStringForProviders, isMimeSupportedByProviders } from '@/lib/document/mime';
import {
MAX_DOCUMENT_BUNDLE_FILES,
MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES,
} from '@/lib/document/bundle';
import { dedupeCourseMaterialFiles } from '@/lib/document/course-materials';
import type { SelectedCourseMaterial } from '@/lib/types/generation';
import { findModelById, modelIdsMatch } from '@/lib/ai/model-aliases';
// ─── Constants ───────────────────────────────────────────────
@@ -49,8 +55,9 @@ export interface GenerationToolbarProps {
onWebSearchChange: (v: boolean) => void;
onSettingsOpen: (section?: SettingsSection) => void;
// PDF
pdfFile: File | null;
onPdfFileChange: (file: File | null) => void;
courseMaterials: SelectedCourseMaterial[];
onCourseMaterialsAdd: (files: File[]) => void;
onCourseMaterialRemove: (id: string) => void;
onPdfError: (error: string | null) => void;
}
@@ -59,8 +66,9 @@ export function GenerationToolbar({
webSearch,
onWebSearchChange,
onSettingsOpen,
pdfFile,
onPdfFileChange,
courseMaterials,
onCourseMaterialsAdd,
onCourseMaterialRemove,
onPdfError,
}: GenerationToolbarProps) {
const { t } = useI18n();
@@ -125,46 +133,79 @@ export function GenerationToolbar({
// Course material handler. `plain-text` is always active alongside the
// user-selected extractor so txt/md files remain uploadable without
// configuring an external service.
const acceptForCurrentProvider = useMemo(
() => getAcceptStringForProviders([pdfProviderId, 'plain-text']),
const activeDocumentProviderIds = useMemo(
() => [pdfProviderId, 'plain-text'] as const,
[pdfProviderId],
);
const acceptForCurrentProvider = useMemo(
() => getAcceptStringForProviders(activeDocumentProviderIds),
[activeDocumentProviderIds],
);
// If the user switches to a provider that doesn't support the currently
// attached file, drop the file so we don't submit an unsupported upload
// that would only fail server-side.
// If the user switches to a provider that doesn't support already attached
// materials, drop only the incompatible files so the eventual extraction
// request matches the current provider capability.
useEffect(() => {
if (!pdfFile) return;
const stillSupported = isMimeSupportedByProviders(
{ mimeType: pdfFile.type, fileName: pdfFile.name },
[pdfProviderId, 'plain-text'],
const unsupportedMaterials = courseMaterials.filter(
(file) =>
!isMimeSupportedByProviders(
{ mimeType: file.type, fileName: file.name },
activeDocumentProviderIds,
),
);
if (!stillSupported) {
onPdfFileChange(null);
onPdfError(t('upload.unsupportedCourseMaterial'));
}
// Intentionally omit onPdfFileChange/onPdfError/t from deps: they are
// stable enough for this check and adding them would re-run the effect
// on unrelated parent re-renders.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [pdfProviderId, pdfFile]);
if (unsupportedMaterials.length === 0) return;
const handleFileSelect = (file: File) => {
if (
!isMimeSupportedByProviders({ mimeType: file.type, fileName: file.name }, [
pdfProviderId,
'plain-text',
])
) {
for (const file of unsupportedMaterials) {
onCourseMaterialRemove(file.id);
}
onPdfError(t('upload.unsupportedCourseMaterial'));
// Intentionally omit callbacks/t from deps: adding them would re-run this
// provider capability cleanup on unrelated parent re-renders.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [activeDocumentProviderIds, courseMaterials]);
const handleFilesSelect = (incomingFiles: File[]) => {
const supportedFiles = incomingFiles.filter((file) =>
isMimeSupportedByProviders(
{ mimeType: file.type, fileName: file.name },
activeDocumentProviderIds,
),
);
if (supportedFiles.length === 0) {
onPdfError(t('upload.unsupportedCourseMaterial'));
return;
}
if (file.size > MAX_COURSE_MATERIAL_SIZE_BYTES) {
if (supportedFiles.length !== incomingFiles.length) {
onPdfError(t('upload.unsupportedCourseMaterial'));
return;
}
if (supportedFiles.some((file) => file.size > MAX_COURSE_MATERIAL_SIZE_BYTES)) {
onPdfError(t('upload.fileTooLarge'));
return;
}
const dedupedFiles = dedupeCourseMaterialFiles(courseMaterials, supportedFiles);
if (dedupedFiles.length === 0) return;
if (courseMaterials.length + dedupedFiles.length > MAX_DOCUMENT_BUNDLE_FILES) {
onPdfError(t('upload.courseMaterialCountLimit', { n: MAX_DOCUMENT_BUNDLE_FILES }));
return;
}
const totalSize =
courseMaterials.reduce((sum, file) => sum + file.size, 0) +
dedupedFiles.reduce((sum, file) => sum + file.size, 0);
if (totalSize > MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES) {
onPdfError(
t('upload.courseMaterialTotalSizeLimit', {
n: Math.floor(MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES / 1024 / 1024),
}),
);
return;
}
onPdfError(null);
onPdfFileChange(file);
onCourseMaterialsAdd(dedupedFiles);
};
// ─── Pill button helper ─────────────────────────────
@@ -216,19 +257,13 @@ export function GenerationToolbar({
{/* ── Course material (extractor + upload) combined Popover ── */}
<Popover>
<PopoverTrigger asChild>
{pdfFile ? (
{courseMaterials.length > 0 ? (
<button className={pillActive}>
<Paperclip className="size-3.5" />
<span className="max-w-[100px] truncate">{pdfFile.name}</span>
<span
role="button"
className="size-4 rounded-full inline-flex items-center justify-center hover:bg-violet-200 dark:hover:bg-violet-800 transition-colors"
onClick={(e) => {
e.stopPropagation();
onPdfFileChange(null);
}}
>
<X className="size-2.5" />
<span className="max-w-[140px] truncate">
{courseMaterials.length === 1
? courseMaterials[0].name
: t('toolbar.courseMaterialsSelected', { n: courseMaterials.length })}
</span>
</button>
) : (
@@ -284,33 +319,14 @@ export function GenerationToolbar({
ref={fileInputRef}
className="hidden"
accept={acceptForCurrentProvider}
multiple
onChange={(e) => {
const f = e.target.files?.[0];
if (f) handleFileSelect(f);
const files = Array.from(e.target.files ?? []);
if (files.length > 0) handleFilesSelect(files);
e.target.value = '';
}}
/>
{pdfFile ? (
<div className="space-y-2">
<div className="flex items-center gap-2">
<div className="size-8 rounded-lg bg-violet-100 dark:bg-violet-900/30 flex items-center justify-center shrink-0">
<FileText className="size-4 text-violet-600 dark:text-violet-400" />
</div>
<div className="min-w-0 flex-1">
<p className="text-sm font-medium truncate">{pdfFile.name}</p>
<p className="text-xs text-muted-foreground">
{(pdfFile.size / 1024 / 1024).toFixed(2)} MB
</p>
</div>
</div>
<button
onClick={() => onPdfFileChange(null)}
className="w-full text-xs text-destructive hover:underline text-left"
>
{t('toolbar.removeCourseMaterial')}
</button>
</div>
) : (
<div className="space-y-3">
<div
className={cn(
'flex flex-col items-center justify-center rounded-lg border-2 border-dashed p-4 transition-colors cursor-pointer',
@@ -327,17 +343,57 @@ export function GenerationToolbar({
onDrop={(e) => {
e.preventDefault();
setIsDragging(false);
const f = e.dataTransfer.files?.[0];
if (f) handleFileSelect(f);
const files = Array.from(e.dataTransfer.files ?? []);
if (files.length > 0) handleFilesSelect(files);
}}
>
<Paperclip className="size-5 text-muted-foreground/50 mb-1.5" />
<p className="text-xs font-medium">{t('toolbar.courseMaterialUpload')}</p>
<p className="text-[10px] text-muted-foreground/60 mt-0.5">
<p className="text-[10px] text-muted-foreground/60 mt-0.5 text-center">
{t('upload.courseMaterialSizeLimit')}
</p>
<p className="text-[10px] text-muted-foreground/60 text-center">
{t('upload.courseMaterialCountLimit', { n: MAX_DOCUMENT_BUNDLE_FILES })}
</p>
</div>
)}
{courseMaterials.length > 0 && (
<div className="space-y-2">
<p className="text-[10px] text-muted-foreground/70">
{t('toolbar.courseMaterialMergeOrder')}
</p>
<div className="max-h-44 space-y-2 overflow-y-auto pr-1">
{[...courseMaterials]
.sort((a, b) => a.order - b.order)
.map((file) => (
<div
key={file.id}
className="flex items-center gap-2 rounded-lg border border-border/50 px-2 py-2"
>
<div className="size-8 rounded-lg bg-violet-100 dark:bg-violet-900/30 flex items-center justify-center shrink-0">
<FileText className="size-4 text-violet-600 dark:text-violet-400" />
</div>
<div className="min-w-0 flex-1">
<p className="text-sm font-medium truncate">
{file.order}. {file.name}
</p>
<p className="text-xs text-muted-foreground">
{(file.size / 1024 / 1024).toFixed(2)} MB
</p>
</div>
<button
onClick={() => onCourseMaterialRemove(file.id)}
className="size-6 rounded-full inline-flex items-center justify-center text-muted-foreground hover:bg-muted transition-colors"
aria-label={t('toolbar.removeCourseMaterial')}
>
<X className="size-3.5" />
</button>
</div>
))}
</div>
</div>
)}
</div>
</div>
</PopoverContent>
</Popover>
+256
View File
@@ -0,0 +1,256 @@
import { MAX_PDF_CONTENT_CHARS, MAX_VISION_IMAGES } from '@/lib/constants/generation';
import type { PdfImage, SessionDocumentSource } from '@/lib/types/generation';
export const MAX_DOCUMENT_BUNDLE_FILES = 5;
export const MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES = 150 * 1024 * 1024;
const BASE_BUDGET_PER_DOCUMENT = 1500;
const RESERVED_BUDGET_RATIO = 0.4;
const SECTION_SEPARATOR = '\n\n---\n\n';
export interface ParsedDocumentImage extends Omit<PdfImage, 'storageId' | 'visionPriority'> {
src: string;
}
export interface ParsedDocumentPart {
source: Omit<SessionDocumentSource, 'storageKey'>;
text: string;
rawTextLength: number;
pageCount?: number;
images: ParsedDocumentImage[];
}
export interface DocumentBundleResult {
text: string;
images: Array<ParsedDocumentImage & { visionPriority: number }>;
textContentBudget: number;
totalRawTextLength: number;
totalImageCount: number;
visionImageCount: number;
}
function escapeRegex(value: string): string {
return value.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
}
function replaceImageIds(text: string, idMap: ReadonlyMap<string, string>): string {
let nextText = text;
for (const [fromId, toId] of idMap.entries()) {
nextText = nextText.replace(
new RegExp(`(?<![\\w-])${escapeRegex(fromId)}(?![\\w-])`, 'g'),
toId,
);
}
return nextText;
}
function truncateTextAtBoundary(text: string, maxChars: number): string {
if (maxChars <= 0) return '';
if (text.length <= maxChars) return text;
const sliced = Array.from(text).slice(0, maxChars).join('');
let cut = sliced.length;
while (cut > 0 && /[\p{L}\p{N}_-]/u.test(sliced[cut - 1])) {
cut -= 1;
}
return cut > 0 ? sliced.slice(0, cut) : sliced;
}
function buildSectionHeader(part: ParsedDocumentPart, index: number): string {
const lines = [
`## Source Document ${index + 1}: ${part.source.name}`,
`- Order: ${part.source.order}`,
part.source.mimeType ? `- MIME type: ${part.source.mimeType}` : undefined,
typeof part.pageCount === 'number' ? `- Pages: ${part.pageCount}` : undefined,
'',
].filter((line): line is string => typeof line === 'string');
return `${lines.join('\n')}\n`;
}
export function allocateDocumentTextBudgets(lengths: number[], maxChars: number): number[] {
if (lengths.length === 0 || maxChars <= 0) return lengths.map(() => 0);
const reserved = Math.min(
lengths.length * BASE_BUDGET_PER_DOCUMENT,
Math.floor(maxChars * RESERVED_BUDGET_RATIO),
);
const basePerDocument = Math.floor(reserved / lengths.length);
const budgets = lengths.map((length) => Math.min(length, basePerDocument));
let remainingBudget = maxChars - budgets.reduce((sum, value) => sum + value, 0);
const unmet = lengths
.map((length, index) => ({ index, remaining: Math.max(0, length - budgets[index]) }))
.filter((entry) => entry.remaining > 0);
while (remainingBudget > 0 && unmet.length > 0) {
const totalRemaining = unmet.reduce((sum, entry) => sum + entry.remaining, 0);
if (totalRemaining === 0) break;
let distributed = 0;
for (const entry of unmet) {
if (remainingBudget === 0) break;
const share = Math.floor((remainingBudget * entry.remaining) / totalRemaining);
const allocation = Math.min(entry.remaining, share > 0 ? share : 1, remainingBudget);
budgets[entry.index] += allocation;
entry.remaining -= allocation;
remainingBudget -= allocation;
distributed += allocation;
}
if (distributed === 0) break;
for (let i = unmet.length - 1; i >= 0; i -= 1) {
if (unmet[i].remaining === 0) unmet.splice(i, 1);
}
}
return budgets;
}
function compareImagesForVision(a: ParsedDocumentImage, b: ParsedDocumentImage): number {
const aHasDescription = Number(Boolean(a.description));
const bHasDescription = Number(Boolean(b.description));
if (aHasDescription !== bHasDescription) return bHasDescription - aHasDescription;
const sourceDiff = (a.sourceDocumentOrder ?? 0) - (b.sourceDocumentOrder ?? 0);
if (sourceDiff !== 0) return sourceDiff;
const pageDiff = a.pageNumber - b.pageNumber;
if (pageDiff !== 0) return pageDiff;
const aArea = (a.width ?? 0) * (a.height ?? 0);
const bArea = (b.width ?? 0) * (b.height ?? 0);
return bArea - aArea;
}
function pickVisionImageIds(images: ParsedDocumentImage[], maxImages: number): string[] {
if (images.length === 0 || maxImages <= 0) return [];
const grouped = new Map<string, ParsedDocumentImage[]>();
for (const image of images) {
const key = image.sourceDocumentId || 'unknown';
const bucket = grouped.get(key) ?? [];
bucket.push(image);
grouped.set(key, bucket);
}
const groups = Array.from(grouped.entries())
.sort((a, b) => (a[1][0]?.sourceDocumentOrder ?? 0) - (b[1][0]?.sourceDocumentOrder ?? 0))
.map(([, group]) => [...group].sort(compareImagesForVision));
const selectedIds: string[] = [];
for (const group of groups) {
if (selectedIds.length >= maxImages) break;
const image = group.shift();
if (image) selectedIds.push(image.id);
}
while (selectedIds.length < maxImages) {
let added = false;
for (const group of groups) {
if (selectedIds.length >= maxImages) break;
const image = group.shift();
if (image) {
selectedIds.push(image.id);
added = true;
}
}
if (!added) break;
}
return selectedIds;
}
export function sortDocumentImagesForVision<
T extends Pick<PdfImage, 'visionPriority' | 'pageNumber' | 'id'>,
>(images: T[]): T[] {
return [...images].sort((a, b) => {
const priorityDiff = (b.visionPriority ?? 0) - (a.visionPriority ?? 0);
if (priorityDiff !== 0) return priorityDiff;
if (a.pageNumber !== b.pageNumber) return a.pageNumber - b.pageNumber;
const aNumericId = Number(a.id.match(/^img_(\d+)$/)?.[1] ?? Number.NaN);
const bNumericId = Number(b.id.match(/^img_(\d+)$/)?.[1] ?? Number.NaN);
if (Number.isFinite(aNumericId) && Number.isFinite(bNumericId)) {
return aNumericId - bNumericId;
}
return a.id.localeCompare(b.id);
});
}
export function buildDocumentBundle(
parts: ParsedDocumentPart[],
options?: { maxChars?: number; maxVisionImages?: number },
): DocumentBundleResult {
const maxChars = options?.maxChars ?? MAX_PDF_CONTENT_CHARS;
const maxVisionImages = options?.maxVisionImages ?? MAX_VISION_IMAGES;
const orderedParts = [...parts].sort((a, b) => a.source.order - b.source.order);
const stableParts = orderedParts.map((part) => {
const stableIdMap = new Map<string, string>();
const stableImages = part.images.map((image, index) => {
const stableId = `doc_${part.source.order}_img_${index + 1}`;
stableIdMap.set(image.id, stableId);
return {
...image,
id: stableId,
originalId: image.originalId ?? image.id,
sourceDocumentId: part.source.id,
sourceDocumentName: part.source.name,
sourceDocumentOrder: part.source.order,
};
});
return {
...part,
text: replaceImageIds(part.text, stableIdMap),
images: stableImages,
};
});
const headers = stableParts.map(buildSectionHeader);
const framingChars =
headers.reduce((sum, header) => sum + header.length, 0) +
Math.max(0, stableParts.length - 1) * SECTION_SEPARATOR.length;
const textContentBudget = Math.max(0, maxChars - framingChars);
const textBudgets = allocateDocumentTextBudgets(
stableParts.map((part) => part.text.length),
textContentBudget,
);
const flattenedImages = stableParts.flatMap((part) => part.images);
const finalIdMap = new Map<string, string>();
flattenedImages.forEach((image, index) => finalIdMap.set(image.id, `img_${index + 1}`));
const text = stableParts
.map((part, index) => {
const boundedText = replaceImageIds(
truncateTextAtBoundary(part.text, textBudgets[index]),
finalIdMap,
);
return `${headers[index]}${boundedText}`;
})
.join(SECTION_SEPARATOR);
const images = flattenedImages.map((image) => ({
...image,
id: finalIdMap.get(image.id) ?? image.id,
}));
const selectedVisionIds = pickVisionImageIds(images, maxVisionImages);
const visionPriority = new Map(
selectedVisionIds.map((id, index) => [id, selectedVisionIds.length - index]),
);
return {
text,
images: images.map((image) => ({
...image,
visionPriority: visionPriority.get(image.id) ?? 0,
})),
textContentBudget,
totalRawTextLength: stableParts.reduce((sum, part) => sum + part.rawTextLength, 0),
totalImageCount: images.length,
visionImageCount: selectedVisionIds.length,
};
}
+20
View File
@@ -0,0 +1,20 @@
import type { SelectedCourseMaterial } from '@/lib/types/generation';
type CourseMaterialFingerprintInput = Pick<File, 'name' | 'size' | 'lastModified'>;
export function courseMaterialFingerprint(file: CourseMaterialFingerprintInput): string {
return `${file.name}:${file.size}:${file.lastModified}`;
}
export function dedupeCourseMaterialFiles(
existing: SelectedCourseMaterial[],
incoming: File[],
): File[] {
const seen = new Set(existing.map(courseMaterialFingerprint));
return incoming.filter((file) => {
const fingerprint = courseMaterialFingerprint(file);
if (seen.has(fingerprint)) return false;
seen.add(fingerprint);
return true;
});
}
+10 -5
View File
@@ -50,12 +50,17 @@ function createPdfBackedDocumentExtractor(id: PDFProviderId): DocumentExtractorP
const parsed =
id === 'mineru-cloud'
? await parseWithMinerUCloud(config, input.buffer, input.fileName)
: input.mimeType === DOCUMENT_MIME_TYPES.pdf
? await parsePDF(config, input.buffer)
: await parseWithMinerUDocument(config, input.buffer, {
fileName: input.fileName || 'document',
: id === 'mineru'
? await parseWithMinerUDocument(config, input.buffer, {
fileName: input.fileName || 'document.pdf',
mimeType: input.mimeType,
});
})
: input.mimeType === DOCUMENT_MIME_TYPES.pdf
? await parsePDF(config, input.buffer)
: await parseWithMinerUDocument(config, input.buffer, {
fileName: input.fileName || 'document',
mimeType: input.mimeType,
});
return parsedPdfToDocumentArtifact(parsed, input);
},
+8
View File
@@ -12,6 +12,13 @@ export {
normalizeDocumentMimeType,
} from './mime';
export { documentArtifactToParsedPdfContent, parsedPdfToDocumentArtifact } from './pdf-compat';
export {
MAX_DOCUMENT_BUNDLE_FILES,
MAX_DOCUMENT_BUNDLE_TOTAL_SIZE_BYTES,
allocateDocumentTextBudgets,
buildDocumentBundle,
sortDocumentImagesForVision,
} from './bundle';
export type {
DocumentArtifact,
DocumentAsset,
@@ -24,3 +31,4 @@ export type {
DocumentExtractorProvider,
DocumentExtractorProviderId,
} from './types';
export type { DocumentBundleResult, ParsedDocumentImage, ParsedDocumentPart } from './bundle';
+4 -2
View File
@@ -13,6 +13,7 @@ import type {
} from '@/lib/types/generation';
import { buildPrompt, PROMPT_IDS } from '@/lib/prompts';
import { formatImageDescription, formatImagePlaceholder } from './prompt-formatters';
import { sortDocumentImagesForVision } from '@/lib/document/bundle';
import { parseJsonResponse } from './json-repair';
import { uniquifyMediaElementIds } from './scene-builder';
import type { AICallFn, GenerationResult } from './pipeline-types';
@@ -55,10 +56,11 @@ export async function generateSceneOutlinesFromRequirements(
if (pdfImages && pdfImages.length > 0) {
if (options?.visionEnabled && options?.imageMapping) {
// Vision mode: split into vision images (first N) and text-only (rest)
const allWithSrc = pdfImages.filter((img) => options.imageMapping![img.id]);
const sortedImages = sortDocumentImagesForVision(pdfImages);
const allWithSrc = sortedImages.filter((img) => options.imageMapping![img.id]);
const visionSlice = allWithSrc.slice(0, MAX_VISION_IMAGES);
const textOnlySlice = allWithSrc.slice(MAX_VISION_IMAGES);
const noSrcImages = pdfImages.filter((img) => !options.imageMapping![img.id]);
const noSrcImages = sortedImages.filter((img) => !options.imageMapping![img.id]);
const visionDescriptions = visionSlice.map((img) => formatImagePlaceholder(img));
const textDescriptions = [...textOnlySlice, ...noSrcImages].map((img) =>
+4 -2
View File
@@ -81,8 +81,9 @@ export function formatImageDescription(img: PdfImage): string {
const ratio = (img.width / img.height).toFixed(2);
dimInfo = ` | size: ${img.width}×${img.height} (aspect ratio ${ratio})`;
}
const sourceInfo = img.sourceDocumentName ? ` from ${img.sourceDocumentName}` : ' from PDF';
const desc = img.description ? ` | ${img.description}` : '';
return `- **${img.id}**: from PDF page ${img.pageNumber}${dimInfo}${desc}`;
return `- **${img.id}**:${sourceInfo} page ${img.pageNumber}${dimInfo}${desc}`;
}
/**
@@ -95,7 +96,8 @@ export function formatImagePlaceholder(img: PdfImage): string {
const ratio = (img.width / img.height).toFixed(2);
dimInfo = ` | size: ${img.width}×${img.height} (aspect ratio ${ratio})`;
}
return `- **${img.id}**: image from PDF page ${img.pageNumber}${dimInfo} [see attached]`;
const sourceInfo = img.sourceDocumentName ? ` from ${img.sourceDocumentName}` : ' from PDF';
return `- **${img.id}**: image${sourceInfo} page ${img.pageNumber}${dimInfo} [see attached]`;
}
/**
+7 -3
View File
@@ -8,6 +8,7 @@
import { nanoid } from 'nanoid';
import katex from 'katex';
import { MAX_VISION_IMAGES } from '@/lib/constants/generation';
import { sortDocumentImagesForVision } from '@/lib/document/bundle';
import type {
SceneOutline,
GeneratedSlideContent,
@@ -575,12 +576,13 @@ async function generateSlideContent(
let visionImages: Array<{ id: string; src: string }> | undefined;
if (assignedImages && assignedImages.length > 0) {
const sortedAssignedImages = sortDocumentImagesForVision(assignedImages);
if (visionEnabled && imageMapping) {
// Vision mode: split into vision images and text-only
const withSrc = assignedImages.filter((img) => imageMapping[img.id]);
const withSrc = sortedAssignedImages.filter((img) => imageMapping[img.id]);
const visionSlice = withSrc.slice(0, MAX_VISION_IMAGES);
const textOnlySlice = withSrc.slice(MAX_VISION_IMAGES);
const noSrcImages = assignedImages.filter((img) => !imageMapping[img.id]);
const noSrcImages = sortedAssignedImages.filter((img) => !imageMapping[img.id]);
const visionDescriptions = visionSlice.map((img) => formatImagePlaceholder(img));
const textDescriptions = [...textOnlySlice, ...noSrcImages].map((img) =>
@@ -595,7 +597,9 @@ async function generateSlideContent(
height: img.height,
}));
} else {
assignedImagesText = assignedImages.map((img) => formatImageDescription(img)).join('\n');
assignedImagesText = sortedAssignedImages
.map((img) => formatImageDescription(img))
.join('\n');
}
}
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "إزالة الملف",
"documentExtractor": "مستخرج المستندات",
"courseMaterialUpload": "رفع مادة المقرر",
"courseMaterialsSelected": "تم تحديد {{n}} مواد",
"courseMaterialMergeOrder": "سيتم دمج المواد بالترتيب المعروض أدناه.",
"removeCourseMaterial": "إزالة الملف",
"webSearchOn": "مُفعّل",
"webSearchOff": "انقر للتفعيل",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "أخبرني بأي شيء تريد تعلمه، مثلاً:\n\"علمني بايثون من الصفر في 30 دقيقة\"\n\"اشرح تحويل فورييه على السبورة\"\n\"كيف تلعب لعبة أفالون\"",
"requirementRequired": "يرجى إدخال متطلبات المقرر",
"fileTooLarge": "الملف كبير جدًا. يرجى اختيار ملف أصغر من 50 ميغابايت",
"courseMaterialCountLimit": "يمكنك رفع ما يصل إلى 5 ملفات لمواد المقرر",
"courseMaterialTotalSizeLimit": "يجب أن يكون الحجم الإجمالي لمواد المقرر أقل من 150 ميغابايت",
"unsupportedCourseMaterial": "المستخرج الحالي لا يدعم هذه الصيغة. غيّر المستخرج أو اختر ملفًا آخر."
},
"generation": {
"pdfLoadFailed": "فشل تحميل ملف PDF، يرجى المحاولة مرة أخرى",
"pdfParseFailed": "فشل تحليل PDF",
"analyzingCourseMaterial": "تحليل ملف {{type}}",
"analyzingCourseMaterial": "تحليل المستندات",
"analyzingCourseMaterialDesc": "جارٍ استخراج هيكل المستند ومحتواه...",
"courseMaterialLoadFailed": "فشل تحميل مادة المقرر، يرجى المحاولة مرة أخرى",
"courseMaterialParseFailed": "فشل تحليل مادة المقرر",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "Remove file",
"documentExtractor": "Extractor",
"courseMaterialUpload": "Upload course material",
"courseMaterialsSelected": "{{n}} materials selected",
"courseMaterialMergeOrder": "Materials are merged in the order shown below.",
"removeCourseMaterial": "Remove file",
"webSearchOn": "Enabled",
"webSearchOff": "Click to enable",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "Tell me anything you want to learn, e.g.\n\"Teach me Python from scratch in 30 minutes\"\n\"Explain Fourier Transform on the whiteboard\"\n\"How to play the board game Avalon\"",
"requirementRequired": "Please enter course requirements",
"fileTooLarge": "File too large. Please select a file smaller than 50MB",
"courseMaterialCountLimit": "You can upload up to 5 course material files",
"courseMaterialTotalSizeLimit": "Total course material size must be under 150MB",
"unsupportedCourseMaterial": "This format isn't supported by the current extractor. Switch the extractor or pick a different file."
},
"generation": {
"pdfLoadFailed": "Failed to load PDF file, please try again",
"pdfParseFailed": "PDF parsing failed",
"analyzingCourseMaterial": "Analyzing {{type}} file",
"analyzingCourseMaterial": "Analyzing documents",
"analyzingCourseMaterialDesc": "Extracting document structure and content...",
"courseMaterialLoadFailed": "Failed to load course material, please try again",
"courseMaterialParseFailed": "Course material parsing failed",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "ファイルを削除",
"documentExtractor": "抽出器",
"courseMaterialUpload": "教材をアップロード",
"courseMaterialsSelected": "{{n}}件の教材を選択済み",
"courseMaterialMergeOrder": "教材は下に表示された順序で結合されます。",
"removeCourseMaterial": "ファイルを削除",
"webSearchOn": "有効",
"webSearchOff": "クリックして有効化",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "学びたいことを自由に入力してください。例えば:\n「Pythonをゼロから30分で教えて」\n「フーリエ変換をホワイトボードで解説して」\n「ボードゲーム『アバロン』の遊び方」",
"requirementRequired": "コースの要件を入力してください",
"fileTooLarge": "ファイルが大きすぎます。50MB以下のファイルを選択してください",
"courseMaterialCountLimit": "教材ファイルは最大5件までアップロードできます",
"courseMaterialTotalSizeLimit": "教材の合計サイズは150MB未満にしてください",
"unsupportedCourseMaterial": "現在の解析器はこの形式に対応していません。解析器を切り替えるか、別のファイルを選択してください。"
},
"generation": {
"pdfLoadFailed": "PDFファイルの読み込みに失敗しました。もう一度お試しください",
"pdfParseFailed": "PDFの解析に失敗しました",
"analyzingCourseMaterial": "{{type}}ファイルを分析中",
"analyzingCourseMaterial": "ドキュメントを分析中",
"analyzingCourseMaterialDesc": "ドキュメントの構造と内容を抽出しています...",
"courseMaterialLoadFailed": "教材の読み込みに失敗しました。もう一度お試しください",
"courseMaterialParseFailed": "教材の解析に失敗しました",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "파일 제거",
"documentExtractor": "추출기",
"courseMaterialUpload": "강의 자료 업로드",
"courseMaterialsSelected": "강의 자료 {{n}}개 선택됨",
"courseMaterialMergeOrder": "강의 자료는 아래 표시된 순서대로 병합됩니다.",
"removeCourseMaterial": "파일 제거",
"webSearchOn": "사용 중",
"webSearchOff": "클릭하여 사용",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "학습하고 싶은 내용을 자유롭게 입력하세요, 예:\n\"파이썬을 30분 만에 처음부터 가르쳐주세요\"\n\"칠판에 푸리에 변환을 설명해주세요\"\n\"보드게임 Avalon 하는 방법\"",
"requirementRequired": "강좌 요구사항을 입력해주세요",
"fileTooLarge": "파일이 너무 큽니다. 50MB 미만의 파일을 선택해주세요",
"courseMaterialCountLimit": "강의 자료 파일은 최대 5개까지 업로드할 수 있습니다",
"courseMaterialTotalSizeLimit": "강의 자료 전체 크기는 150MB 미만이어야 합니다",
"unsupportedCourseMaterial": "현재 파서가 이 형식을 지원하지 않습니다. 파서를 변경하거나 다른 파일을 선택하세요."
},
"generation": {
"pdfLoadFailed": "PDF 파일 불러오기에 실패했습니다. 다시 시도해주세요",
"pdfParseFailed": "PDF 파싱에 실패했습니다",
"analyzingCourseMaterial": "{{type}} 파일 분석 중",
"analyzingCourseMaterial": "문서 분석 중",
"analyzingCourseMaterialDesc": "문서 구조와 내용을 추출하고 있어요...",
"courseMaterialLoadFailed": "강의 자료를 불러오지 못했습니다. 다시 시도해주세요",
"courseMaterialParseFailed": "강의 자료 파싱에 실패했습니다",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "Remover arquivo",
"documentExtractor": "Extrator",
"courseMaterialUpload": "Enviar material do curso",
"courseMaterialsSelected": "{{n}} materiais selecionados",
"courseMaterialMergeOrder": "Os materiais serão mesclados na ordem mostrada abaixo.",
"removeCourseMaterial": "Remover arquivo",
"webSearchOn": "Ativada",
"webSearchOff": "Clique para ativar",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "Conte-me o que você quer aprender, por exemplo:\n\"Me ensine Python do zero em 30 minutos\"\n\"Explique a Transformada de Fourier no quadro\"\n\"Como jogar o jogo de tabuleiro Avalon\"",
"requirementRequired": "Por favor, descreva os objetivos do curso",
"fileTooLarge": "Arquivo muito grande. Selecione um arquivo com menos de 50 MB",
"courseMaterialCountLimit": "Você pode enviar até 5 arquivos de material do curso",
"courseMaterialTotalSizeLimit": "O tamanho total dos materiais deve ser menor que 150 MB",
"unsupportedCourseMaterial": "O extrator atual não suporta este formato. Troque o extrator ou escolha outro arquivo."
},
"generation": {
"pdfLoadFailed": "Falha ao carregar o PDF, tente novamente",
"pdfParseFailed": "Falha ao analisar o PDF",
"analyzingCourseMaterial": "Analisando arquivo {{type}}",
"analyzingCourseMaterial": "Analisando documentos",
"analyzingCourseMaterialDesc": "Extraindo estrutura e conteúdo do documento...",
"courseMaterialLoadFailed": "Falha ao carregar o material do curso, tente novamente",
"courseMaterialParseFailed": "Falha ao analisar o material do curso",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "Удалить файл",
"documentExtractor": "Экстрактор",
"courseMaterialUpload": "Загрузить материал курса",
"courseMaterialsSelected": "Выбрано материалов: {{n}}",
"courseMaterialMergeOrder": "Материалы объединяются в порядке, показанном ниже.",
"removeCourseMaterial": "Удалить файл",
"webSearchOn": "Включено",
"webSearchOff": "Нажмите для включения",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "Расскажите, что вы хотите изучить, например:\n\"Научи меня Python с нуля за 30 минут\"\n\"Объясни преобразование Фурье на доске\"\n\"Как играть в настольную игру Авалон\"",
"requirementRequired": "Пожалуйста, укажите требования к курсу",
"fileTooLarge": "Файл слишком большой. Выберите файл до 50 МБ",
"courseMaterialCountLimit": "Можно загрузить до 5 файлов материалов курса",
"courseMaterialTotalSizeLimit": "Общий размер материалов курса должен быть меньше 150 МБ",
"unsupportedCourseMaterial": "Текущий парсер не поддерживает этот формат. Смените парсер или выберите другой файл."
},
"generation": {
"pdfLoadFailed": "Не удалось загрузить PDF, попробуйте снова",
"pdfParseFailed": "Ошибка обработки PDF",
"analyzingCourseMaterial": "Анализ файла {{type}}",
"analyzingCourseMaterial": "Анализ документов",
"analyzingCourseMaterialDesc": "Извлечение структуры и содержимого документа...",
"courseMaterialLoadFailed": "Не удалось загрузить материал курса, попробуйте снова",
"courseMaterialParseFailed": "Ошибка обработки материала курса",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "移除文件",
"documentExtractor": "文档解析器",
"courseMaterialUpload": "上传课程材料",
"courseMaterialsSelected": "已选择 {{n}} 个材料",
"courseMaterialMergeOrder": "材料会按下方显示顺序合并。",
"removeCourseMaterial": "移除文件",
"webSearchOn": "已开启",
"webSearchOff": "点击开启",
@@ -825,12 +827,14 @@
"requirementPlaceholder": "输入你想学的任何内容,例如:\n「从零学 Python,30 分钟写出第一个程序」\n「用白板给我讲解傅里叶变换」\n「阿瓦隆桌游怎么玩」",
"requirementRequired": "请输入课程需求",
"fileTooLarge": "文件过大,请选择小于 50MB 的文件",
"courseMaterialCountLimit": "最多可上传 5 个课程材料文件",
"courseMaterialTotalSizeLimit": "课程材料总大小需小于 150MB",
"unsupportedCourseMaterial": "当前解析器不支持该格式,请切换解析器或选择其他文件"
},
"generation": {
"pdfLoadFailed": "无法加载 PDF 文件,请重试",
"pdfParseFailed": "PDF 解析失败",
"analyzingCourseMaterial": "解析 {{type}} 文件",
"analyzingCourseMaterial": "解析文档",
"analyzingCourseMaterialDesc": "正在提取文档结构和内容...",
"courseMaterialLoadFailed": "无法加载课程材料,请重试",
"courseMaterialParseFailed": "课程材料解析失败",
+5 -1
View File
@@ -15,6 +15,8 @@
"removePdf": "移除檔案",
"documentExtractor": "文件解析器",
"courseMaterialUpload": "上傳課程材料",
"courseMaterialsSelected": "已選擇 {{n}} 個材料",
"courseMaterialMergeOrder": "材料會依下方顯示順序合併。",
"removeCourseMaterial": "移除檔案",
"webSearchOn": "已開啟",
"webSearchOff": "點擊開啟",
@@ -810,12 +812,14 @@
"requirementPlaceholder": "輸入你想學的任何內容,例如:\n「從零學 Python,30 分鐘寫出第一個程式」\n「用白板為我講解傅立葉轉換」\n「阿瓦隆桌遊怎麼玩」",
"requirementRequired": "請輸入課程需求",
"fileTooLarge": "檔案過大,請選擇小於 50MB 的檔案",
"courseMaterialCountLimit": "最多可上傳 5 個課程材料檔案",
"courseMaterialTotalSizeLimit": "課程材料總大小需小於 150MB",
"unsupportedCourseMaterial": "目前的解析器不支援此格式,請切換解析器或選擇其他檔案"
},
"generation": {
"pdfLoadFailed": "無法載入 PDF 檔案,請重試",
"pdfParseFailed": "PDF 解析失敗",
"analyzingCourseMaterial": "解析 {{type}} 檔案",
"analyzingCourseMaterial": "解析文件",
"analyzingCourseMaterialDesc": "正在擷取文件結構和內容...",
"courseMaterialLoadFailed": "無法載入課程材料,請重試",
"courseMaterialParseFailed": "課程材料解析失敗",
+9 -3
View File
@@ -147,6 +147,11 @@ import { extractMinerUResult } from './mineru-parser';
import { parseWithMinerUCloud } from './mineru-cloud';
const log = createLogger('PDFProviders');
const DEFAULT_MINERU_BACKEND = 'pipeline';
function getMinerUBackend(): string {
return process.env.PDF_MINERU_BACKEND?.trim() || DEFAULT_MINERU_BACKEND;
}
/**
* Turn a self-hosted MinerU error body into an actionable message.
@@ -355,9 +360,10 @@ export async function parseWithMinerUDocument(
// MinerU API form fields
// Defaults already: return_md=true, formula_enable=true, table_enable=true
formData.append('parse_method', 'auto');
// hybrid-auto-engine: best accuracy, uses VLM for layout understanding (requires GPU)
// pipeline: basic mode, no VLM, faster but lower quality image extraction
formData.append('backend', 'hybrid-auto-engine');
// `hybrid-auto-engine` may require a GPU/device configuration in the MinerU
// service. Default to the broadly compatible pipeline backend; operators can
// opt into hybrid/VLM mode with PDF_MINERU_BACKEND when their service is ready.
formData.append('backend', getMinerUBackend());
formData.append('return_content_list', 'true');
formData.append('return_images', 'true');
+26
View File
@@ -21,6 +21,11 @@ export interface PdfImage {
storageId?: string; // Reference to IndexedDB (session_xxx_img_1)
width?: number; // Image width (px or normalized)
height?: number; // Image height (px or normalized)
originalId?: string; // ID assigned by the extractor before bundle-level normalization
sourceDocumentId?: string; // DocumentBundle source ID
sourceDocumentName?: string; // Original source filename for citation back to material
sourceDocumentOrder?: number; // Upload order in the bundle
visionPriority?: number; // Higher values are attached first when vision budget is limited
}
/**
@@ -28,6 +33,27 @@ export interface PdfImage {
*/
export type ImageMapping = Record<string, string>;
export interface SelectedCourseMaterial {
id: string;
file: File;
name: string;
size: number;
lastModified: number;
type: string;
order: number;
}
export interface SessionDocumentSource {
id: string;
name: string;
size: number;
lastModified?: number;
mimeType?: string;
order: number;
storageKey: string;
providerId?: string;
}
// ==================== Stage 1 Input ====================
export interface UploadedDocument {
+22 -4
View File
@@ -51,9 +51,11 @@ export async function storeImages(
): Promise<string[]> {
const sessionId = nanoid(10);
const storedIds: string[] = [];
let currentImageId: string | undefined;
for (const img of images) {
try {
try {
for (const img of images) {
currentImageId = img.id;
const blob = base64ToBlob(img.src);
const mimeMatch = img.src.match(/data:(.*?);/);
const mimeType = mimeMatch ? mimeMatch[1] : 'image/png';
@@ -72,9 +74,14 @@ export async function storeImages(
await db.imageFiles.put(record);
storedIds.push(storageId);
} catch (error) {
log.error(`Failed to store image ${img.id}:`, error);
}
} catch (error) {
await Promise.allSettled(storedIds.map((id) => db.imageFiles.delete(id)));
const message = `Failed to store image bundle${
currentImageId ? ` at image ${currentImageId}` : ''
}`;
log.error(`${message}:`, error);
throw new Error(message, { cause: error });
}
return storedIds;
@@ -175,3 +182,14 @@ export async function loadPdfBlob(key: string): Promise<Blob | null> {
const record = await db.imageFiles.get(key);
return record?.blob ?? null;
}
/**
* Delete a stored PDF Blob from IndexedDB by its storage key.
*/
export async function deletePdfBlob(key: string): Promise<void> {
await db.imageFiles.delete(key);
}
export const storeDocumentBlob = storePdfBlob;
export const loadDocumentBlob = loadPdfBlob;
export const deleteDocumentBlob = deletePdfBlob;
+160
View File
@@ -0,0 +1,160 @@
import { describe, expect, it } from 'vitest';
import {
allocateDocumentTextBudgets,
buildDocumentBundle,
sortDocumentImagesForVision,
type ParsedDocumentPart,
} from '@/lib/document/bundle';
function part(order: number, overrides: Partial<ParsedDocumentPart> = {}): ParsedDocumentPart {
return {
source: {
id: `source-${order}`,
name: `Source ${order}.pdf`,
size: 1024,
lastModified: order,
mimeType: 'application/pdf',
order,
},
text: `Document ${order} references image_${order}.`,
rawTextLength: `Document ${order} references image_${order}.`.length,
pageCount: order + 1,
images: [
{
id: `image_${order}`,
src: `data:image/png;base64,${order}`,
pageNumber: order,
description: `figure ${order}`,
width: 100 + order,
height: 80 + order,
},
],
...overrides,
};
}
describe('document bundle', () => {
it('allocates a base text budget before proportional remainder', () => {
const budgets = allocateDocumentTextBudgets([100, 5000, 10000], 6000);
expect(budgets).toHaveLength(3);
expect(budgets[0]).toBe(100);
expect(budgets[1]).toBeGreaterThan(1500);
expect(budgets[2]).toBeGreaterThan(budgets[1]);
expect(budgets.reduce((sum, value) => sum + value, 0)).toBe(6000);
});
it('merges documents in source order and rewrites image IDs globally', () => {
const bundle = buildDocumentBundle([part(2), part(1)], {
maxChars: 2000,
maxVisionImages: 4,
});
expect(bundle.text.indexOf('## Source Document 1: Source 1.pdf')).toBeLessThan(
bundle.text.indexOf('## Source Document 2: Source 2.pdf'),
);
expect(bundle.text).toContain('Document 1 references img_1.');
expect(bundle.text).toContain('Document 2 references img_2.');
expect(bundle.images).toMatchObject([
{
id: 'img_1',
originalId: 'image_1',
sourceDocumentId: 'source-1',
sourceDocumentName: 'Source 1.pdf',
sourceDocumentOrder: 1,
},
{
id: 'img_2',
originalId: 'image_2',
sourceDocumentId: 'source-2',
sourceDocumentName: 'Source 2.pdf',
sourceDocumentOrder: 2,
},
]);
});
it('assigns vision priority round-robin across source documents', () => {
const bundle = buildDocumentBundle(
[
part(1, {
images: [
{
id: 'a1',
src: 'data:image/png;base64,a1',
pageNumber: 1,
description: 'first source primary',
width: 500,
height: 500,
},
{
id: 'a2',
src: 'data:image/png;base64,a2',
pageNumber: 2,
description: 'first source secondary',
width: 500,
height: 500,
},
],
}),
part(2, {
images: [
{
id: 'b1',
src: 'data:image/png;base64,b1',
pageNumber: 1,
description: 'second source primary',
width: 500,
height: 500,
},
],
}),
],
{ maxChars: 2000, maxVisionImages: 2 },
);
const priorities = new Map(
bundle.images.map((image) => [image.originalId, image.visionPriority]),
);
expect(priorities.get('a1')).toBe(2);
expect(priorities.get('b1')).toBe(1);
expect(priorities.get('a2')).toBe(0);
});
it('does not truncate into a rewritten image ID', () => {
const images = Array.from({ length: 11 }, (_, index) => ({
id: `image_${index + 1}`,
src: `data:image/png;base64,${index + 1}`,
pageNumber: 1,
}));
const header =
'## Source Document 1: Source 1.pdf\n' +
'- Order: 1\n' +
'- MIME type: application/pdf\n' +
'- Pages: 2\n\n';
const bundle = buildDocumentBundle(
[
part(1, {
text: 'prefix image_11 suffix',
rawTextLength: 'prefix image_11 suffix'.length,
images,
}),
],
{ maxChars: header.length + 'prefix img_1'.length, maxVisionImages: 4 },
);
expect(bundle.text).not.toContain('prefix img_1');
});
it('sorts same-priority img_N IDs numerically', () => {
const sorted = sortDocumentImagesForVision([
{ id: 'img_10', pageNumber: 1, visionPriority: 0 },
{ id: 'img_2', pageNumber: 1, visionPriority: 0 },
{ id: 'img_1', pageNumber: 1, visionPriority: 0 },
]);
expect(sorted.map((image) => image.id)).toEqual(['img_1', 'img_2', 'img_10']);
});
});
@@ -84,6 +84,25 @@ describe('POST /api/extract-document', () => {
});
});
it('returns 413 before extraction when the file exceeds the per-file size limit', async () => {
const res = await postExtractDocument({
file: new File([new Uint8Array(51 * 1024 * 1024)], 'large.pdf', {
type: 'application/pdf',
}),
providerId: 'mineru-cloud',
apiKey: 'cloud-key',
});
const json = await res.json();
expect(res.status).toBe(413);
expect(json).toMatchObject({
success: false,
errorCode: 'INVALID_REQUEST',
});
expect(json.error).toContain('Maximum size is 50MB');
expect(mocks.parseWithMinerUCloud).not.toHaveBeenCalled();
});
it('returns 400 for an unknown requested provider', async () => {
const res = await postExtractDocument({
file: new File(['hello'], 'notes.txt', { type: 'text/plain' }),