import path from 'path'; import fs from 'fs'; import { Plugin } from './plugin.js'; import { FFMpegUtils } from 'common-utils'; import { CaptionPlugin } from './generate-captions.js'; /** * Must come before the Captions plugin in the plugin list, since it modifies the transcript and bubbles before captions are generated. * The JoinTranscriptsPlugin joins the section media, bubbles, audios, and captions into a single section. * The output of this plugin is a modified transcript with a single section containing the concatenated video and the concatenated audio file, * and the bubbles across all sections are also moved to the single combined section with appropriate offsets. */ export class JoinTranscriptsPlugin extends Plugin { constructor(name, options) { super(name, options); } async applyPrerender(originalManuscript, jobId) { const transcript = originalManuscript.transcript || []; if (transcript.length <= 1) { this.log(`Skipping JoinTranscriptsPlugin because there is only ${transcript.length} section.`); return; } const originalManuscriptMeta = originalManuscript.meta || []; let combinedAudioPath = path.join('public', `combined-audio-${jobId}.mp3`); let combinedAudioCaptionsFilePath = path.join('public', `combined-audio-captions-${jobId}.json`); let combinedBubbles = []; let combinedMediaPath = path.join('public', `combined-media-${jobId}.mp4`); combinedBubbles = await this.combineAllTranscriptBubbles(transcript, { contagious: this.options.contagious || false, contagiousDelay: this.options.contagiousDelay ?? 0 }); let combinedCaptions = await this.combineAllAudioCaptions(transcript, combinedAudioCaptionsFilePath); await this.combineAllAudios(transcript, combinedAudioPath); // Combine all medias const combinedMediaMeta = await this.combineAllMedias(transcript, combinedAudioPath, combinedMediaPath); // Calculate total duration across all sections let totalDuration = 0; for (let item of transcript) { totalDuration += item.durationInSeconds || 0; } // Build the single combined section from the first section as a template const firstSection = transcript[0] || {}; const combinedSection = { ...firstSection, text: transcript.map(s => s.text || '').join(' '), index: 0, mediaAbsPaths: combinedMediaPath ? [{ path: combinedMediaPath, type: 'video', // Force type to video since we concatenate them into mp4 dimensions: combinedMediaMeta?.video ? { width: parseInt(combinedMediaMeta.video.width), height: parseInt(combinedMediaMeta.video.height) } : undefined, durationSec: totalDuration }] : [], bubbles: combinedBubbles, audioFullPath: combinedAudioPath, audioCaptionFile: combinedAudioCaptionsFilePath, audioCaption: combinedCaptions, durationInSeconds: totalDuration, duration: Math.ceil(totalDuration * (originalManuscriptMeta?.fps || 30)), offset: 0, // Remove transitions since everything is merged into one section transition_type: 'none', transition_file: undefined, transition_duration_sec: 0 }; // Replace the transcript with the single combined section originalManuscript.transcript = [combinedSection]; this.log(`Combined ${transcript.length} sections into 1 joined-transcripts section. Total duration: ${totalDuration.toFixed(2)}s, Bubbles: ${combinedBubbles.length}`); } async combineAllMedias(transcripts, audioPath, outputPath) { const listFile = outputPath + '.list.txt'; let listContent = ''; let hasMedia = false; let firstMediaMeta = null; for (let section of transcripts) { let sectionDuration = section.durationInSeconds || 0; if (section.mediaAbsPaths && section.mediaAbsPaths.length) { // Assume the first media is the primary one, similar to how remotion ui behaves let m = section.mediaAbsPaths[0]; let mediaPath = m.path; if (!mediaPath || !fs.existsSync(mediaPath)) { const flattenedPath = this.mediaPathFlatten(mediaPath); if (fs.existsSync(flattenedPath)) { mediaPath = flattenedPath; } else { continue; } } if (!firstMediaMeta) { try { firstMediaMeta = await FFMpegUtils.getMediaMetadata(mediaPath); } catch(e) { this.log(`Error getting metadata for ${mediaPath}: ${e}`); } } const ext = path.extname(mediaPath).toLowerCase(); let dur = m.durationSec || sectionDuration || 5; // Fix path for ffmpeg list file let safePath = mediaPath.replace(/\\/g, '/'); if (['.png', '.jpg', '.jpeg', '.webp'].includes(ext)) { listContent += `file '${safePath}'\n`; listContent += `duration ${dur}\n`; hasMedia = true; } else { listContent += `file '${safePath}'\n`; hasMedia = true; } } } if (!hasMedia) { this.log('No media found to combine.'); return null; } fs.writeFileSync(listFile, listContent); this.log(`Combining all media using list file ${listFile} into ${outputPath}`); try { let cmd = ''; if (audioPath && fs.existsSync(audioPath)) { cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -i "${audioPath}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p -c:a aac -map 0:v -map 1:a -shortest "${outputPath}" -y`; } else { cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p "${outputPath}" -y`; } await FFMpegUtils.execute(cmd); return firstMediaMeta; } catch (err) { this.log(`Error combining media: ${err}`); throw err; } } async combineAllTranscriptBubbles(transcripts, { contagious = false, contagiousDelay = 0 } = {}) { let combinedBubbles = []; let cumulativeOffset = 0; for (let section of transcripts) { const sectionDuration = section.durationInSeconds || 0; const bubbles = section.bubbles || []; for (let bubble of bubbles) { // Deep clone the bubble so we don't mutate the original const adjustedBubble = JSON.parse(JSON.stringify(bubble)); // Adjust timing with cumulative offset from previous sections if (adjustedBubble.fromSec !== undefined && adjustedBubble.fromSec !== null) { adjustedBubble.fromSec = adjustedBubble.fromSec + cumulativeOffset; } if (adjustedBubble.toSec !== undefined && adjustedBubble.toSec !== null) { adjustedBubble.toSec = adjustedBubble.toSec + cumulativeOffset; } combinedBubbles.push(adjustedBubble); } cumulativeOffset += sectionDuration; } // Sort bubbles by their start time combinedBubbles.sort((a, b) => (a.fromSec || 0) - (b.fromSec || 0)); // Apply contagious chaining: bubble[N+1].fromSec = bubble[N].toSec + contagiousDelay if (contagious && combinedBubbles.length > 1) { this.log(`Applying contagious chaining with delay=${contagiousDelay}s across ${combinedBubbles.length} bubbles.`); for (let i = 1; i < combinedBubbles.length; i++) { const prev = combinedBubbles[i - 1]; if (prev.toSec !== undefined && prev.toSec !== null) { const newFromSec = prev.toSec + contagiousDelay; if (combinedBubbles[i].fromSec !== undefined && combinedBubbles[i].toSec !== undefined) { // Preserve original duration, shift the window forward const originalDur = combinedBubbles[i].toSec - combinedBubbles[i].fromSec; combinedBubbles[i].fromSec = newFromSec; combinedBubbles[i].toSec = Math.round((newFromSec + originalDur) * 1000) / 1000; } else { combinedBubbles[i].fromSec = newFromSec; } // Mirror into mediaTextPrompts if present if (combinedBubbles[i].mediaTextPrompts?.length) { combinedBubbles[i].mediaTextPrompts[0].fromSec = combinedBubbles[i].fromSec; combinedBubbles[i].mediaTextPrompts[0].toSec = combinedBubbles[i].toSec; if (combinedBubbles[i].durationSec !== undefined) { combinedBubbles[i].durationSec = Math.round((combinedBubbles[i].toSec - combinedBubbles[i].fromSec) * 1000) / 1000; combinedBubbles[i].mediaTextPrompts[0].durationSec = combinedBubbles[i].durationSec; } } } } } // Overlap adjacent bubbles if they use slide animations to make them physically slide together! for (let i = 0; i < combinedBubbles.length - 1; i++) { let current = combinedBubbles[i]; let next = combinedBubbles[i + 1]; let isCurrentSlide = current.animExtra?.template?.includes('slide') || current.bubbleExtra?.animExtra?.template?.includes('slide') || current.transition_type?.includes('slide'); let isNextSlide = next.animExtra?.template?.includes('slide') || next.bubbleExtra?.animExtra?.template?.includes('slide') || next.transition_type?.includes('slide'); // If they are adjacent (or very close) if ((isCurrentSlide || isNextSlide) && Math.abs((next.fromSec || 0) - (current.toSec || 0)) < 0.1) { // Slide animations take 0.25s (capped in remotion) // Extend current by 0.125s and start next 0.125s early. current.toSec = (current.toSec || 0) + 0.125; next.fromSec = Math.max(0, (next.fromSec || 0) - 0.125); if (current.durationSec !== undefined) { current.durationSec = Math.round((current.toSec - current.fromSec) * 1000) / 1000; } if (next.durationSec !== undefined) { next.durationSec = Math.round((next.toSec - next.fromSec) * 1000) / 1000; } // Mirror into mediaTextPrompts if present if (current.mediaTextPrompts?.length) { current.mediaTextPrompts[0].fromSec = current.fromSec; current.mediaTextPrompts[0].toSec = current.toSec; current.mediaTextPrompts[0].durationSec = current.durationSec; } if (next.mediaTextPrompts?.length) { next.mediaTextPrompts[0].fromSec = next.fromSec; next.mediaTextPrompts[0].toSec = next.toSec; next.mediaTextPrompts[0].durationSec = next.durationSec; } } } this.log(`Combined ${combinedBubbles.length} bubbles from ${transcripts.length} sections`); return combinedBubbles; } // Combines all audio files from all transcripts into a single audio file. The combined audio file is saved to the specified output path. async combineAllAudios(transcripts, outAudioPath) { // flatten path first for each audio before processing this.mediaPathFlatten(..); const audioPaths = []; for (let section of transcripts) { let audioPath = section.audioFullPath; if (!audioPath) continue; // Try the original path first, then fall back to flattened if (!fs.existsSync(audioPath)) { const flattenedPath = this.mediaPathFlatten(audioPath); if (fs.existsSync(flattenedPath)) { audioPath = flattenedPath; this.log(`Using flattened audio path: ${flattenedPath}`); } else { this.log(`Audio path does not exist: ${audioPath}. Flattened path ${flattenedPath} also not found. Skipping.`); continue; } } audioPaths.push(audioPath); } if (audioPaths.length === 0) { this.log('No audio files found to combine.'); return; } if (audioPaths.length === 1) { // Only one audio file, just copy it fs.copyFileSync(audioPaths[0], outAudioPath); this.log(`Single audio file copied to ${outAudioPath}`); return; } this.log(`Joining ${audioPaths.length} audio files into ${outAudioPath}`); await FFMpegUtils.joinAudios(audioPaths, outAudioPath); this.log(`Combined audio saved to ${outAudioPath}`); } // Combines all audio captions for all transcripts into a single audio caption file taking care of offset. The combined audio caption file is saved to the specified output path. async combineAllAudioCaptions(transcripts, outAudioCaptionPath) { // flatten path first for each audio caption file before processing this.mediaPathFlatten(..); let combinedTranscriptText = ''; let combinedWords = []; let cumulativeOffset = 0; for (let section of transcripts) { let captionFilePath = section.audioCaptionFile; if (!captionFilePath) { // No caption file for this section, just advance the offset cumulativeOffset += section.durationInSeconds || 0; continue; } // If the caption file is an .ass file, check for the original JSON source if (path.extname(captionFilePath) === '.ass') { // Try the _audioCaptionFile (original JSON before ASS conversion) if available if (section._audioCaptionFile) { captionFilePath = section._audioCaptionFile; } else { // Try converting .ass path back to .json captionFilePath = captionFilePath.replace('.ass', '.json'); } } // Resolve the caption file path let resolvedPath = captionFilePath; if (!fs.existsSync(resolvedPath)) { // Try with absolute path from cwd resolvedPath = path.join(process.cwd(), captionFilePath); } if (!fs.existsSync(resolvedPath)) { // Try flattened path const flattenedPath = this.mediaPathFlatten(captionFilePath); if (fs.existsSync(flattenedPath)) { resolvedPath = flattenedPath; this.log(`Using flattened caption path: ${flattenedPath}`); } else { const flattenedAbsPath = path.join(process.cwd(), flattenedPath); if (fs.existsSync(flattenedAbsPath)) { resolvedPath = flattenedAbsPath; this.log(`Using flattened absolute caption path: ${flattenedAbsPath}`); } else { this.log(`Caption file not found: ${captionFilePath}. Skipping.`); cumulativeOffset += section.durationInSeconds || 0; continue; } } } try { const captionData = JSON.parse(fs.readFileSync(resolvedPath, 'utf-8')); const sectionTranscript = captionData.transcript || ''; const sectionWords = captionData.words || []; // Append transcript text with a space separator if (combinedTranscriptText.length > 0 && sectionTranscript.length > 0) { combinedTranscriptText += ' '; } combinedTranscriptText += sectionTranscript; // Adjust word timings with cumulative offset and add to combined list for (let word of sectionWords) { const adjustedWord = { ...word }; if (adjustedWord.start !== undefined && adjustedWord.start !== null) { adjustedWord.start = adjustedWord.start + cumulativeOffset; } if (adjustedWord.end !== undefined && adjustedWord.end !== null) { adjustedWord.end = adjustedWord.end + cumulativeOffset; } combinedWords.push(adjustedWord); } } catch (err) { this.log(`Error reading caption file ${resolvedPath}: ${err}`); } cumulativeOffset += section.durationInSeconds || 0; } // Write the combined caption file const combinedCaptions = { transcript: combinedTranscriptText, words: combinedWords, start: 0, end: cumulativeOffset }; fs.writeFileSync(outAudioCaptionPath, JSON.stringify(combinedCaptions, null, 2)); this.log(`Combined ${combinedWords.length} caption words from ${transcripts.length} sections into ${outAudioCaptionPath}`); return combinedCaptions; } }