remote-rdr / server-plugins /join-transcripts.js
shiveshnavin's picture
Add bubbles
29ad7b4
Raw History Blame Contribute Delete
16 kB
import path from 'path';
import fs from 'fs';
import { Plugin } from './plugin.js';
import { FFMpegUtils } from 'common-utils';
import { CaptionPlugin } from './generate-captions.js';
/**
* Must come before the Captions plugin in the plugin list, since it modifies the transcript and bubbles before captions are generated.
* The JoinTranscriptsPlugin joins the section media, bubbles, audios, and captions into a single section.
* The output of this plugin is a modified transcript with a single section containing the concatenated video and the concatenated audio file,
* and the bubbles across all sections are also moved to the single combined section with appropriate offsets.
*/
export class JoinTranscriptsPlugin extends Plugin {
constructor(name, options) {
super(name, options);
}
async applyPrerender(originalManuscript, jobId) {
const transcript = originalManuscript.transcript || [];
if (transcript.length <= 1) {
this.log(`Skipping JoinTranscriptsPlugin because there is only ${transcript.length} section.`);
return;
}
const originalManuscriptMeta = originalManuscript.meta || [];
let combinedAudioPath = path.join('public', `combined-audio-${jobId}.mp3`);
let combinedAudioCaptionsFilePath = path.join('public', `combined-audio-captions-${jobId}.json`);
let combinedBubbles = [];
let combinedMediaPath = path.join('public', `combined-media-${jobId}.mp4`);
combinedBubbles = await this.combineAllTranscriptBubbles(transcript, {
contagious: this.options.contagious || false,
contagiousDelay: this.options.contagiousDelay ?? 0
});
let combinedCaptions = await this.combineAllAudioCaptions(transcript, combinedAudioCaptionsFilePath);
await this.combineAllAudios(transcript, combinedAudioPath);
// Combine all medias
const combinedMediaMeta = await this.combineAllMedias(transcript, combinedAudioPath, combinedMediaPath);
// Calculate total duration across all sections
let totalDuration = 0;
for (let item of transcript) {
totalDuration += item.durationInSeconds || 0;
}
// Build the single combined section from the first section as a template
const firstSection = transcript[0] || {};
const combinedSection = {
...firstSection,
text: transcript.map(s => s.text || '').join(' '),
index: 0,
mediaAbsPaths: combinedMediaPath ? [{
path: combinedMediaPath,
type: 'video', // Force type to video since we concatenate them into mp4
dimensions: combinedMediaMeta?.video ? {
width: parseInt(combinedMediaMeta.video.width),
height: parseInt(combinedMediaMeta.video.height)
} : undefined,
durationSec: totalDuration
}] : [],
bubbles: combinedBubbles,
audioFullPath: combinedAudioPath,
audioCaptionFile: combinedAudioCaptionsFilePath,
audioCaption: combinedCaptions,
durationInSeconds: totalDuration,
duration: Math.ceil(totalDuration * (originalManuscriptMeta?.fps || 30)),
offset: 0,
// Remove transitions since everything is merged into one section
transition_type: 'none',
transition_file: undefined,
transition_duration_sec: 0
};
// Replace the transcript with the single combined section
originalManuscript.transcript = [combinedSection];
this.log(`Combined ${transcript.length} sections into 1 joined-transcripts section. Total duration: ${totalDuration.toFixed(2)}s, Bubbles: ${combinedBubbles.length}`);
}
async combineAllMedias(transcripts, audioPath, outputPath) {
const listFile = outputPath + '.list.txt';
let listContent = '';
let hasMedia = false;
let firstMediaMeta = null;
for (let section of transcripts) {
let sectionDuration = section.durationInSeconds || 0;
if (section.mediaAbsPaths && section.mediaAbsPaths.length) {
// Assume the first media is the primary one, similar to how remotion ui behaves
let m = section.mediaAbsPaths[0];
let mediaPath = m.path;
if (!mediaPath || !fs.existsSync(mediaPath)) {
const flattenedPath = this.mediaPathFlatten(mediaPath);
if (fs.existsSync(flattenedPath)) {
mediaPath = flattenedPath;
} else {
continue;
}
}
if (!firstMediaMeta) {
try {
firstMediaMeta = await FFMpegUtils.getMediaMetadata(mediaPath);
} catch(e) {
this.log(`Error getting metadata for ${mediaPath}: ${e}`);
}
}
const ext = path.extname(mediaPath).toLowerCase();
let dur = m.durationSec || sectionDuration || 5;
// Fix path for ffmpeg list file
let safePath = mediaPath.replace(/\\/g, '/');
if (['.png', '.jpg', '.jpeg', '.webp'].includes(ext)) {
listContent += `file '${safePath}'\n`;
listContent += `duration ${dur}\n`;
hasMedia = true;
} else {
listContent += `file '${safePath}'\n`;
hasMedia = true;
}
}
}
if (!hasMedia) {
this.log('No media found to combine.');
return null;
}
fs.writeFileSync(listFile, listContent);
this.log(`Combining all media using list file ${listFile} into ${outputPath}`);
try {
let cmd = '';
if (audioPath && fs.existsSync(audioPath)) {
cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -i "${audioPath}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p -c:a aac -map 0:v -map 1:a -shortest "${outputPath}" -y`;
} else {
cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p "${outputPath}" -y`;
}
await FFMpegUtils.execute(cmd);
return firstMediaMeta;
} catch (err) {
this.log(`Error combining media: ${err}`);
throw err;
}
}
async combineAllTranscriptBubbles(transcripts, { contagious = false, contagiousDelay = 0 } = {}) {
let combinedBubbles = [];
let cumulativeOffset = 0;
for (let section of transcripts) {
const sectionDuration = section.durationInSeconds || 0;
const bubbles = section.bubbles || [];
for (let bubble of bubbles) {
// Deep clone the bubble so we don't mutate the original
const adjustedBubble = JSON.parse(JSON.stringify(bubble));
// Adjust timing with cumulative offset from previous sections
if (adjustedBubble.fromSec !== undefined && adjustedBubble.fromSec !== null) {
adjustedBubble.fromSec = adjustedBubble.fromSec + cumulativeOffset;
}
if (adjustedBubble.toSec !== undefined && adjustedBubble.toSec !== null) {
adjustedBubble.toSec = adjustedBubble.toSec + cumulativeOffset;
}
combinedBubbles.push(adjustedBubble);
}
cumulativeOffset += sectionDuration;
}
// Sort bubbles by their start time
combinedBubbles.sort((a, b) => (a.fromSec || 0) - (b.fromSec || 0));
// Apply contagious chaining: bubble[N+1].fromSec = bubble[N].toSec + contagiousDelay
if (contagious && combinedBubbles.length > 1) {
this.log(`Applying contagious chaining with delay=${contagiousDelay}s across ${combinedBubbles.length} bubbles.`);
for (let i = 1; i < combinedBubbles.length; i++) {
const prev = combinedBubbles[i - 1];
if (prev.toSec !== undefined && prev.toSec !== null) {
const newFromSec = prev.toSec + contagiousDelay;
if (combinedBubbles[i].fromSec !== undefined && combinedBubbles[i].toSec !== undefined) {
// Preserve original duration, shift the window forward
const originalDur = combinedBubbles[i].toSec - combinedBubbles[i].fromSec;
combinedBubbles[i].fromSec = newFromSec;
combinedBubbles[i].toSec = Math.round((newFromSec + originalDur) * 1000) / 1000;
} else {
combinedBubbles[i].fromSec = newFromSec;
}
// Mirror into mediaTextPrompts if present
if (combinedBubbles[i].mediaTextPrompts?.length) {
combinedBubbles[i].mediaTextPrompts[0].fromSec = combinedBubbles[i].fromSec;
combinedBubbles[i].mediaTextPrompts[0].toSec = combinedBubbles[i].toSec;
if (combinedBubbles[i].durationSec !== undefined) {
combinedBubbles[i].durationSec = Math.round((combinedBubbles[i].toSec - combinedBubbles[i].fromSec) * 1000) / 1000;
combinedBubbles[i].mediaTextPrompts[0].durationSec = combinedBubbles[i].durationSec;
}
}
}
}
}
// Overlap adjacent bubbles if they use slide animations to make them physically slide together!
for (let i = 0; i < combinedBubbles.length - 1; i++) {
let current = combinedBubbles[i];
let next = combinedBubbles[i + 1];
let isCurrentSlide = current.animExtra?.template?.includes('slide') || current.bubbleExtra?.animExtra?.template?.includes('slide') || current.transition_type?.includes('slide');
let isNextSlide = next.animExtra?.template?.includes('slide') || next.bubbleExtra?.animExtra?.template?.includes('slide') || next.transition_type?.includes('slide');
// If they are adjacent (or very close)
if ((isCurrentSlide || isNextSlide) && Math.abs((next.fromSec || 0) - (current.toSec || 0)) < 0.1) {
// Slide animations take 0.25s (capped in remotion)
// Extend current by 0.125s and start next 0.125s early.
current.toSec = (current.toSec || 0) + 0.125;
next.fromSec = Math.max(0, (next.fromSec || 0) - 0.125);
if (current.durationSec !== undefined) {
current.durationSec = Math.round((current.toSec - current.fromSec) * 1000) / 1000;
}
if (next.durationSec !== undefined) {
next.durationSec = Math.round((next.toSec - next.fromSec) * 1000) / 1000;
}
// Mirror into mediaTextPrompts if present
if (current.mediaTextPrompts?.length) {
current.mediaTextPrompts[0].fromSec = current.fromSec;
current.mediaTextPrompts[0].toSec = current.toSec;
current.mediaTextPrompts[0].durationSec = current.durationSec;
}
if (next.mediaTextPrompts?.length) {
next.mediaTextPrompts[0].fromSec = next.fromSec;
next.mediaTextPrompts[0].toSec = next.toSec;
next.mediaTextPrompts[0].durationSec = next.durationSec;
}
}
}
this.log(`Combined ${combinedBubbles.length} bubbles from ${transcripts.length} sections`);
return combinedBubbles;
}
// Combines all audio files from all transcripts into a single audio file. The combined audio file is saved to the specified output path.
async combineAllAudios(transcripts, outAudioPath) {
// flatten path first for each audio before processing this.mediaPathFlatten(..);
const audioPaths = [];
for (let section of transcripts) {
let audioPath = section.audioFullPath;
if (!audioPath) continue;
// Try the original path first, then fall back to flattened
if (!fs.existsSync(audioPath)) {
const flattenedPath = this.mediaPathFlatten(audioPath);
if (fs.existsSync(flattenedPath)) {
audioPath = flattenedPath;
this.log(`Using flattened audio path: ${flattenedPath}`);
} else {
this.log(`Audio path does not exist: ${audioPath}. Flattened path ${flattenedPath} also not found. Skipping.`);
continue;
}
}
audioPaths.push(audioPath);
}
if (audioPaths.length === 0) {
this.log('No audio files found to combine.');
return;
}
if (audioPaths.length === 1) {
// Only one audio file, just copy it
fs.copyFileSync(audioPaths[0], outAudioPath);
this.log(`Single audio file copied to ${outAudioPath}`);
return;
}
this.log(`Joining ${audioPaths.length} audio files into ${outAudioPath}`);
await FFMpegUtils.joinAudios(audioPaths, outAudioPath);
this.log(`Combined audio saved to ${outAudioPath}`);
}
// Combines all audio captions for all transcripts into a single audio caption file taking care of offset. The combined audio caption file is saved to the specified output path.
async combineAllAudioCaptions(transcripts, outAudioCaptionPath) {
// flatten path first for each audio caption file before processing this.mediaPathFlatten(..);
let combinedTranscriptText = '';
let combinedWords = [];
let cumulativeOffset = 0;
for (let section of transcripts) {
let captionFilePath = section.audioCaptionFile;
if (!captionFilePath) {
// No caption file for this section, just advance the offset
cumulativeOffset += section.durationInSeconds || 0;
continue;
}
// If the caption file is an .ass file, check for the original JSON source
if (path.extname(captionFilePath) === '.ass') {
// Try the _audioCaptionFile (original JSON before ASS conversion) if available
if (section._audioCaptionFile) {
captionFilePath = section._audioCaptionFile;
} else {
// Try converting .ass path back to .json
captionFilePath = captionFilePath.replace('.ass', '.json');
}
}
// Resolve the caption file path
let resolvedPath = captionFilePath;
if (!fs.existsSync(resolvedPath)) {
// Try with absolute path from cwd
resolvedPath = path.join(process.cwd(), captionFilePath);
}
if (!fs.existsSync(resolvedPath)) {
// Try flattened path
const flattenedPath = this.mediaPathFlatten(captionFilePath);
if (fs.existsSync(flattenedPath)) {
resolvedPath = flattenedPath;
this.log(`Using flattened caption path: ${flattenedPath}`);
} else {
const flattenedAbsPath = path.join(process.cwd(), flattenedPath);
if (fs.existsSync(flattenedAbsPath)) {
resolvedPath = flattenedAbsPath;
this.log(`Using flattened absolute caption path: ${flattenedAbsPath}`);
} else {
this.log(`Caption file not found: ${captionFilePath}. Skipping.`);
cumulativeOffset += section.durationInSeconds || 0;
continue;
}
}
}
try {
const captionData = JSON.parse(fs.readFileSync(resolvedPath, 'utf-8'));
const sectionTranscript = captionData.transcript || '';
const sectionWords = captionData.words || [];
// Append transcript text with a space separator
if (combinedTranscriptText.length > 0 && sectionTranscript.length > 0) {
combinedTranscriptText += ' ';
}
combinedTranscriptText += sectionTranscript;
// Adjust word timings with cumulative offset and add to combined list
for (let word of sectionWords) {
const adjustedWord = { ...word };
if (adjustedWord.start !== undefined && adjustedWord.start !== null) {
adjustedWord.start = adjustedWord.start + cumulativeOffset;
}
if (adjustedWord.end !== undefined && adjustedWord.end !== null) {
adjustedWord.end = adjustedWord.end + cumulativeOffset;
}
combinedWords.push(adjustedWord);
}
} catch (err) {
this.log(`Error reading caption file ${resolvedPath}: ${err}`);
}
cumulativeOffset += section.durationInSeconds || 0;
}
// Write the combined caption file
const combinedCaptions = {
transcript: combinedTranscriptText,
words: combinedWords,
start: 0,
end: cumulativeOffset
};
fs.writeFileSync(outAudioCaptionPath, JSON.stringify(combinedCaptions, null, 2));
this.log(`Combined ${combinedWords.length} caption words from ${transcripts.length} sections into ${outAudioCaptionPath}`);
return combinedCaptions;
}
}