| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368 |
- /**
- * Scene Outlines Streaming API (SSE)
- *
- * Streams outline generation via Server-Sent Events.
- * Emits individual outline objects as they're parsed from the LLM response,
- * so the frontend can display them incrementally.
- *
- * SSE events:
- * { type: 'outline', data: SceneOutline, index: number }
- * { type: 'done', outlines: SceneOutline[] }
- * { type: 'error', error: string }
- */
- import { NextRequest } from 'next/server';
- import { streamLLM } from '@/lib/ai/llm';
- import { buildPrompt, PROMPT_IDS } from '@/lib/generation/prompts';
- import {
- formatImageDescription,
- formatImagePlaceholder,
- buildVisionUserContent,
- uniquifyMediaElementIds,
- formatTeacherPersonaForPrompt,
- } from '@/lib/generation/generation-pipeline';
- import type { AgentInfo } from '@/lib/generation/generation-pipeline';
- import { MAX_PDF_CONTENT_CHARS, MAX_VISION_IMAGES } from '@/lib/constants/generation';
- import { nanoid } from 'nanoid';
- import type {
- UserRequirements,
- PdfImage,
- SceneOutline,
- ImageMapping,
- } from '@/lib/types/generation';
- import { apiError } from '@/lib/server/api-response';
- import { createLogger } from '@/lib/logger';
- import { resolveModelFromHeaders } from '@/lib/server/resolve-model';
- const log = createLogger('Outlines Stream');
- export const maxDuration = 300;
- /**
- * Incremental JSON array parser.
- * Extracts complete top-level objects from a partially-streamed JSON array.
- * Returns newly found objects (skipping `alreadyParsed` count).
- */
- function extractNewOutlines(buffer: string, alreadyParsed: number): SceneOutline[] {
- const results: SceneOutline[] = [];
- // Find the start of the JSON array (skip any markdown fencing)
- const stripped = buffer.replace(/^[\s\S]*?(?=\[)/, '');
- const arrayStart = stripped.indexOf('[');
- if (arrayStart === -1) return results;
- let depth = 0;
- let objectStart = -1;
- let inString = false;
- let escaped = false;
- let objectCount = 0;
- for (let i = arrayStart + 1; i < stripped.length; i++) {
- const char = stripped[i];
- if (escaped) {
- escaped = false;
- continue;
- }
- if (char === '\\' && inString) {
- escaped = true;
- continue;
- }
- if (char === '"') {
- inString = !inString;
- continue;
- }
- if (inString) continue;
- if (char === '{') {
- if (depth === 0) objectStart = i;
- depth++;
- } else if (char === '}') {
- depth--;
- if (depth === 0 && objectStart >= 0) {
- objectCount++;
- if (objectCount > alreadyParsed) {
- try {
- const obj = JSON.parse(stripped.substring(objectStart, i + 1));
- results.push(obj);
- } catch {
- // Incomplete or invalid JSON — skip
- }
- }
- objectStart = -1;
- }
- }
- }
- return results;
- }
- export async function POST(req: NextRequest) {
- let requirementSnippet: string | undefined;
- let resolvedModelString: string | undefined;
- try {
- const body = await req.json();
- // Get API configuration from request headers
- const { model: languageModel, modelInfo, modelString } = await resolveModelFromHeaders(req);
- resolvedModelString = modelString;
- if (!body.requirements) {
- return apiError('MISSING_REQUIRED_FIELD', 400, 'Requirements are required');
- }
- const { requirements, pdfText, pdfImages, imageMapping, researchContext, agents } = body as {
- requirements: UserRequirements;
- pdfText?: string;
- pdfImages?: PdfImage[];
- imageMapping?: ImageMapping;
- researchContext?: string;
- agents?: AgentInfo[];
- };
- requirementSnippet = requirements?.requirement?.substring(0, 60);
- // Detect vision capability
- const hasVision = !!modelInfo?.capabilities?.vision;
- // Build prompt (same logic as generateSceneOutlinesFromRequirements)
- let availableImagesText =
- requirements.language === 'zh-CN' ? '无可用图片' : 'No images available';
- let visionImages: Array<{ id: string; src: string }> | undefined;
- if (pdfImages && pdfImages.length > 0) {
- if (hasVision && imageMapping) {
- // Vision mode: split into vision images (first N) and text-only (rest)
- const allWithSrc = pdfImages.filter((img) => imageMapping[img.id]);
- const visionSlice = allWithSrc.slice(0, MAX_VISION_IMAGES);
- const textOnlySlice = allWithSrc.slice(MAX_VISION_IMAGES);
- const noSrcImages = pdfImages.filter((img) => !imageMapping[img.id]);
- const visionDescriptions = visionSlice.map((img) =>
- formatImagePlaceholder(img, requirements.language),
- );
- const textDescriptions = [...textOnlySlice, ...noSrcImages].map((img) =>
- formatImageDescription(img, requirements.language),
- );
- availableImagesText = [...visionDescriptions, ...textDescriptions].join('\n');
- visionImages = visionSlice.map((img) => ({
- id: img.id,
- src: imageMapping[img.id],
- width: img.width,
- height: img.height,
- }));
- } else {
- // Text-only mode: full descriptions
- availableImagesText = pdfImages
- .map((img) => formatImageDescription(img, requirements.language))
- .join('\n');
- }
- }
- // Build media generation policy based on enabled flags
- const imageGenerationEnabled = req.headers.get('x-image-generation-enabled') === 'true';
- const videoGenerationEnabled = req.headers.get('x-video-generation-enabled') === 'true';
- let mediaGenerationPolicy = '';
- if (!imageGenerationEnabled && !videoGenerationEnabled) {
- mediaGenerationPolicy =
- '**IMPORTANT: Do NOT include any mediaGenerations in the outlines. Both image and video generation are disabled.**';
- } else if (!imageGenerationEnabled) {
- mediaGenerationPolicy =
- '**IMPORTANT: Do NOT include any image mediaGenerations (type: "image") in the outlines. Image generation is disabled. Video generation is allowed.**';
- } else if (!videoGenerationEnabled) {
- mediaGenerationPolicy =
- '**IMPORTANT: Do NOT include any video mediaGenerations (type: "video") in the outlines. Video generation is disabled. Image generation is allowed.**';
- }
- // Build teacher context from agents (if available)
- const teacherContext = formatTeacherPersonaForPrompt(agents);
- const prompts = buildPrompt(PROMPT_IDS.REQUIREMENTS_TO_OUTLINES, {
- requirement: requirements.requirement,
- language: requirements.language,
- pdfContent: pdfText
- ? pdfText.substring(0, MAX_PDF_CONTENT_CHARS)
- : requirements.language === 'zh-CN'
- ? '无'
- : 'None',
- availableImages: availableImagesText,
- researchContext: researchContext || (requirements.language === 'zh-CN' ? '无' : 'None'),
- mediaGenerationPolicy,
- teacherContext,
- });
- if (!prompts) {
- return apiError('INTERNAL_ERROR', 500, 'Prompt template not found');
- }
- log.info(
- `Generating outlines: "${requirements.requirement.substring(0, 50)}" [model=${modelString}]`,
- );
- // Create SSE stream with heartbeat to prevent connection timeout
- const encoder = new TextEncoder();
- const HEARTBEAT_INTERVAL_MS = 15_000;
- const stream = new ReadableStream({
- async start(controller) {
- // Heartbeat: periodically send SSE comments to keep the connection alive.
- let heartbeatTimer: ReturnType<typeof setInterval> | null = null;
- const startHeartbeat = () => {
- stopHeartbeat();
- heartbeatTimer = setInterval(() => {
- try {
- controller.enqueue(encoder.encode(`:heartbeat\n\n`));
- } catch {
- stopHeartbeat();
- }
- }, HEARTBEAT_INTERVAL_MS);
- };
- const stopHeartbeat = () => {
- if (heartbeatTimer) {
- clearInterval(heartbeatTimer);
- heartbeatTimer = null;
- }
- };
- const MAX_STREAM_RETRIES = 2;
- try {
- startHeartbeat();
- const streamParams = visionImages?.length
- ? {
- model: languageModel,
- system: prompts.system,
- messages: [
- {
- role: 'user' as const,
- content: buildVisionUserContent(prompts.user, visionImages),
- },
- ],
- maxOutputTokens: modelInfo?.outputWindow,
- }
- : {
- model: languageModel,
- system: prompts.system,
- prompt: prompts.user,
- maxOutputTokens: modelInfo?.outputWindow,
- };
- let parsedOutlines: SceneOutline[] = [];
- let lastError: string | undefined;
- for (let attempt = 1; attempt <= MAX_STREAM_RETRIES + 1; attempt++) {
- try {
- const result = streamLLM(streamParams, 'scene-outlines-stream');
- let fullText = '';
- parsedOutlines = [];
- for await (const chunk of result.textStream) {
- fullText += chunk;
- // Try to extract new outlines from the accumulated text
- const newOutlines = extractNewOutlines(fullText, parsedOutlines.length);
- for (const outline of newOutlines) {
- // Ensure ID and order
- const enriched = {
- ...outline,
- id: outline.id || nanoid(),
- order: parsedOutlines.length + 1,
- };
- parsedOutlines.push(enriched);
- const event = JSON.stringify({
- type: 'outline',
- data: enriched,
- index: parsedOutlines.length - 1,
- });
- controller.enqueue(encoder.encode(`data: ${event}\n\n`));
- }
- }
- // Validate: got outlines?
- if (parsedOutlines.length > 0) break;
- // Empty result — retry if we have attempts left
- lastError = fullText.trim()
- ? 'LLM response could not be parsed into outlines'
- : 'LLM returned empty response';
- if (attempt <= MAX_STREAM_RETRIES) {
- log.warn(
- `Empty outlines (attempt ${attempt}/${MAX_STREAM_RETRIES + 1}), retrying...`,
- );
- // Notify client a retry is happening
- const retryEvent = JSON.stringify({
- type: 'retry',
- attempt,
- maxAttempts: MAX_STREAM_RETRIES + 1,
- });
- controller.enqueue(encoder.encode(`data: ${retryEvent}\n\n`));
- }
- } catch (error) {
- lastError = error instanceof Error ? error.message : String(error);
- if (attempt <= MAX_STREAM_RETRIES) {
- log.warn(
- `Stream error (attempt ${attempt}/${MAX_STREAM_RETRIES + 1}), retrying...`,
- error,
- );
- const retryEvent = JSON.stringify({
- type: 'retry',
- attempt,
- maxAttempts: MAX_STREAM_RETRIES + 1,
- });
- controller.enqueue(encoder.encode(`data: ${retryEvent}\n\n`));
- continue;
- }
- }
- }
- if (parsedOutlines.length > 0) {
- // Replace sequential gen_img_N/gen_vid_N with globally unique IDs
- const uniquifiedOutlines = uniquifyMediaElementIds(parsedOutlines);
- // Send done event with all outlines
- const doneEvent = JSON.stringify({
- type: 'done',
- outlines: uniquifiedOutlines,
- });
- controller.enqueue(encoder.encode(`data: ${doneEvent}\n\n`));
- } else {
- // All retries exhausted, no outlines produced
- log.error(
- `Outline generation failed after ${MAX_STREAM_RETRIES + 1} attempts: ${lastError}`,
- );
- const errorEvent = JSON.stringify({
- type: 'error',
- error: lastError || 'Failed to generate outlines',
- });
- controller.enqueue(encoder.encode(`data: ${errorEvent}\n\n`));
- }
- } catch (error) {
- const errorEvent = JSON.stringify({
- type: 'error',
- error: error instanceof Error ? error.message : String(error),
- });
- controller.enqueue(encoder.encode(`data: ${errorEvent}\n\n`));
- } finally {
- stopHeartbeat();
- controller.close();
- }
- },
- });
- return new Response(stream, {
- headers: {
- 'Content-Type': 'text/event-stream',
- 'Cache-Control': 'no-cache',
- Connection: 'keep-alive',
- },
- });
- } catch (error) {
- log.error(
- `Outline streaming failed [requirement="${requirementSnippet ?? 'unknown'}...", model=${resolvedModelString ?? 'unknown'}]:`,
- error,
- );
- return apiError('INTERNAL_ERROR', 500, error instanceof Error ? error.message : String(error));
- }
- }
|