Spaces:
Running
Running
Download server-plugins/join-transcripts.js from Semibit/remote-rdr: direct link, hf CLI and curl.
- Browser
- Download file 16 kB
-
https://huggingface.co/spaces/Semibit/remote-rdr/resolve/main/server-plugins/join-transcripts.js
- Command line
-
hf download hf://spaces/Semibit/remote-rdr/server-plugins/join-transcripts.js
-
curl -L -o join-transcripts.js https://huggingface.co/spaces/Semibit/remote-rdr/resolve/main/server-plugins/join-transcripts.js
16 kB
| import path from 'path'; | |
| import fs from 'fs'; | |
| import { Plugin } from './plugin.js'; | |
| import { FFMpegUtils } from 'common-utils'; | |
| import { CaptionPlugin } from './generate-captions.js'; | |
| /** | |
| * Must come before the Captions plugin in the plugin list, since it modifies the transcript and bubbles before captions are generated. | |
| * The JoinTranscriptsPlugin joins the section media, bubbles, audios, and captions into a single section. | |
| * The output of this plugin is a modified transcript with a single section containing the concatenated video and the concatenated audio file, | |
| * and the bubbles across all sections are also moved to the single combined section with appropriate offsets. | |
| */ | |
| export class JoinTranscriptsPlugin extends Plugin { | |
| constructor(name, options) { | |
| super(name, options); | |
| } | |
| async applyPrerender(originalManuscript, jobId) { | |
| const transcript = originalManuscript.transcript || []; | |
| if (transcript.length <= 1) { | |
| this.log(`Skipping JoinTranscriptsPlugin because there is only ${transcript.length} section.`); | |
| return; | |
| } | |
| const originalManuscriptMeta = originalManuscript.meta || []; | |
| let combinedAudioPath = path.join('public', `combined-audio-${jobId}.mp3`); | |
| let combinedAudioCaptionsFilePath = path.join('public', `combined-audio-captions-${jobId}.json`); | |
| let combinedBubbles = []; | |
| let combinedMediaPath = path.join('public', `combined-media-${jobId}.mp4`); | |
| combinedBubbles = await this.combineAllTranscriptBubbles(transcript, { | |
| contagious: this.options.contagious || false, | |
| contagiousDelay: this.options.contagiousDelay ?? 0 | |
| }); | |
| let combinedCaptions = await this.combineAllAudioCaptions(transcript, combinedAudioCaptionsFilePath); | |
| await this.combineAllAudios(transcript, combinedAudioPath); | |
| // Combine all medias | |
| const combinedMediaMeta = await this.combineAllMedias(transcript, combinedAudioPath, combinedMediaPath); | |
| // Calculate total duration across all sections | |
| let totalDuration = 0; | |
| for (let item of transcript) { | |
| totalDuration += item.durationInSeconds || 0; | |
| } | |
| // Build the single combined section from the first section as a template | |
| const firstSection = transcript[0] || {}; | |
| const combinedSection = { | |
| ...firstSection, | |
| text: transcript.map(s => s.text || '').join(' '), | |
| index: 0, | |
| mediaAbsPaths: combinedMediaPath ? [{ | |
| path: combinedMediaPath, | |
| type: 'video', // Force type to video since we concatenate them into mp4 | |
| dimensions: combinedMediaMeta?.video ? { | |
| width: parseInt(combinedMediaMeta.video.width), | |
| height: parseInt(combinedMediaMeta.video.height) | |
| } : undefined, | |
| durationSec: totalDuration | |
| }] : [], | |
| bubbles: combinedBubbles, | |
| audioFullPath: combinedAudioPath, | |
| audioCaptionFile: combinedAudioCaptionsFilePath, | |
| audioCaption: combinedCaptions, | |
| durationInSeconds: totalDuration, | |
| duration: Math.ceil(totalDuration * (originalManuscriptMeta?.fps || 30)), | |
| offset: 0, | |
| // Remove transitions since everything is merged into one section | |
| transition_type: 'none', | |
| transition_file: undefined, | |
| transition_duration_sec: 0 | |
| }; | |
| // Replace the transcript with the single combined section | |
| originalManuscript.transcript = [combinedSection]; | |
| this.log(`Combined ${transcript.length} sections into 1 joined-transcripts section. Total duration: ${totalDuration.toFixed(2)}s, Bubbles: ${combinedBubbles.length}`); | |
| } | |
| async combineAllMedias(transcripts, audioPath, outputPath) { | |
| const listFile = outputPath + '.list.txt'; | |
| let listContent = ''; | |
| let hasMedia = false; | |
| let firstMediaMeta = null; | |
| for (let section of transcripts) { | |
| let sectionDuration = section.durationInSeconds || 0; | |
| if (section.mediaAbsPaths && section.mediaAbsPaths.length) { | |
| // Assume the first media is the primary one, similar to how remotion ui behaves | |
| let m = section.mediaAbsPaths[0]; | |
| let mediaPath = m.path; | |
| if (!mediaPath || !fs.existsSync(mediaPath)) { | |
| const flattenedPath = this.mediaPathFlatten(mediaPath); | |
| if (fs.existsSync(flattenedPath)) { | |
| mediaPath = flattenedPath; | |
| } else { | |
| continue; | |
| } | |
| } | |
| if (!firstMediaMeta) { | |
| try { | |
| firstMediaMeta = await FFMpegUtils.getMediaMetadata(mediaPath); | |
| } catch(e) { | |
| this.log(`Error getting metadata for ${mediaPath}: ${e}`); | |
| } | |
| } | |
| const ext = path.extname(mediaPath).toLowerCase(); | |
| let dur = m.durationSec || sectionDuration || 5; | |
| // Fix path for ffmpeg list file | |
| let safePath = mediaPath.replace(/\\/g, '/'); | |
| if (['.png', '.jpg', '.jpeg', '.webp'].includes(ext)) { | |
| listContent += `file '${safePath}'\n`; | |
| listContent += `duration ${dur}\n`; | |
| hasMedia = true; | |
| } else { | |
| listContent += `file '${safePath}'\n`; | |
| hasMedia = true; | |
| } | |
| } | |
| } | |
| if (!hasMedia) { | |
| this.log('No media found to combine.'); | |
| return null; | |
| } | |
| fs.writeFileSync(listFile, listContent); | |
| this.log(`Combining all media using list file ${listFile} into ${outputPath}`); | |
| try { | |
| let cmd = ''; | |
| if (audioPath && fs.existsSync(audioPath)) { | |
| cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -i "${audioPath}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p -c:a aac -map 0:v -map 1:a -shortest "${outputPath}" -y`; | |
| } else { | |
| cmd = `ffmpeg -f concat -safe 0 -i "${listFile}" -c:v libx264 -preset veryfast -crf 23 -pix_fmt yuv420p "${outputPath}" -y`; | |
| } | |
| await FFMpegUtils.execute(cmd); | |
| return firstMediaMeta; | |
| } catch (err) { | |
| this.log(`Error combining media: ${err}`); | |
| throw err; | |
| } | |
| } | |
| async combineAllTranscriptBubbles(transcripts, { contagious = false, contagiousDelay = 0 } = {}) { | |
| let combinedBubbles = []; | |
| let cumulativeOffset = 0; | |
| for (let section of transcripts) { | |
| const sectionDuration = section.durationInSeconds || 0; | |
| const bubbles = section.bubbles || []; | |
| for (let bubble of bubbles) { | |
| // Deep clone the bubble so we don't mutate the original | |
| const adjustedBubble = JSON.parse(JSON.stringify(bubble)); | |
| // Adjust timing with cumulative offset from previous sections | |
| if (adjustedBubble.fromSec !== undefined && adjustedBubble.fromSec !== null) { | |
| adjustedBubble.fromSec = adjustedBubble.fromSec + cumulativeOffset; | |
| } | |
| if (adjustedBubble.toSec !== undefined && adjustedBubble.toSec !== null) { | |
| adjustedBubble.toSec = adjustedBubble.toSec + cumulativeOffset; | |
| } | |
| combinedBubbles.push(adjustedBubble); | |
| } | |
| cumulativeOffset += sectionDuration; | |
| } | |
| // Sort bubbles by their start time | |
| combinedBubbles.sort((a, b) => (a.fromSec || 0) - (b.fromSec || 0)); | |
| // Apply contagious chaining: bubble[N+1].fromSec = bubble[N].toSec + contagiousDelay | |
| if (contagious && combinedBubbles.length > 1) { | |
| this.log(`Applying contagious chaining with delay=${contagiousDelay}s across ${combinedBubbles.length} bubbles.`); | |
| for (let i = 1; i < combinedBubbles.length; i++) { | |
| const prev = combinedBubbles[i - 1]; | |
| if (prev.toSec !== undefined && prev.toSec !== null) { | |
| const newFromSec = prev.toSec + contagiousDelay; | |
| if (combinedBubbles[i].fromSec !== undefined && combinedBubbles[i].toSec !== undefined) { | |
| // Preserve original duration, shift the window forward | |
| const originalDur = combinedBubbles[i].toSec - combinedBubbles[i].fromSec; | |
| combinedBubbles[i].fromSec = newFromSec; | |
| combinedBubbles[i].toSec = Math.round((newFromSec + originalDur) * 1000) / 1000; | |
| } else { | |
| combinedBubbles[i].fromSec = newFromSec; | |
| } | |
| // Mirror into mediaTextPrompts if present | |
| if (combinedBubbles[i].mediaTextPrompts?.length) { | |
| combinedBubbles[i].mediaTextPrompts[0].fromSec = combinedBubbles[i].fromSec; | |
| combinedBubbles[i].mediaTextPrompts[0].toSec = combinedBubbles[i].toSec; | |
| if (combinedBubbles[i].durationSec !== undefined) { | |
| combinedBubbles[i].durationSec = Math.round((combinedBubbles[i].toSec - combinedBubbles[i].fromSec) * 1000) / 1000; | |
| combinedBubbles[i].mediaTextPrompts[0].durationSec = combinedBubbles[i].durationSec; | |
| } | |
| } | |
| } | |
| } | |
| } | |
| // Overlap adjacent bubbles if they use slide animations to make them physically slide together! | |
| for (let i = 0; i < combinedBubbles.length - 1; i++) { | |
| let current = combinedBubbles[i]; | |
| let next = combinedBubbles[i + 1]; | |
| let isCurrentSlide = current.animExtra?.template?.includes('slide') || current.bubbleExtra?.animExtra?.template?.includes('slide') || current.transition_type?.includes('slide'); | |
| let isNextSlide = next.animExtra?.template?.includes('slide') || next.bubbleExtra?.animExtra?.template?.includes('slide') || next.transition_type?.includes('slide'); | |
| // If they are adjacent (or very close) | |
| if ((isCurrentSlide || isNextSlide) && Math.abs((next.fromSec || 0) - (current.toSec || 0)) < 0.1) { | |
| // Slide animations take 0.25s (capped in remotion) | |
| // Extend current by 0.125s and start next 0.125s early. | |
| current.toSec = (current.toSec || 0) + 0.125; | |
| next.fromSec = Math.max(0, (next.fromSec || 0) - 0.125); | |
| if (current.durationSec !== undefined) { | |
| current.durationSec = Math.round((current.toSec - current.fromSec) * 1000) / 1000; | |
| } | |
| if (next.durationSec !== undefined) { | |
| next.durationSec = Math.round((next.toSec - next.fromSec) * 1000) / 1000; | |
| } | |
| // Mirror into mediaTextPrompts if present | |
| if (current.mediaTextPrompts?.length) { | |
| current.mediaTextPrompts[0].fromSec = current.fromSec; | |
| current.mediaTextPrompts[0].toSec = current.toSec; | |
| current.mediaTextPrompts[0].durationSec = current.durationSec; | |
| } | |
| if (next.mediaTextPrompts?.length) { | |
| next.mediaTextPrompts[0].fromSec = next.fromSec; | |
| next.mediaTextPrompts[0].toSec = next.toSec; | |
| next.mediaTextPrompts[0].durationSec = next.durationSec; | |
| } | |
| } | |
| } | |
| this.log(`Combined ${combinedBubbles.length} bubbles from ${transcripts.length} sections`); | |
| return combinedBubbles; | |
| } | |
| // Combines all audio files from all transcripts into a single audio file. The combined audio file is saved to the specified output path. | |
| async combineAllAudios(transcripts, outAudioPath) { | |
| // flatten path first for each audio before processing this.mediaPathFlatten(..); | |
| const audioPaths = []; | |
| for (let section of transcripts) { | |
| let audioPath = section.audioFullPath; | |
| if (!audioPath) continue; | |
| // Try the original path first, then fall back to flattened | |
| if (!fs.existsSync(audioPath)) { | |
| const flattenedPath = this.mediaPathFlatten(audioPath); | |
| if (fs.existsSync(flattenedPath)) { | |
| audioPath = flattenedPath; | |
| this.log(`Using flattened audio path: ${flattenedPath}`); | |
| } else { | |
| this.log(`Audio path does not exist: ${audioPath}. Flattened path ${flattenedPath} also not found. Skipping.`); | |
| continue; | |
| } | |
| } | |
| audioPaths.push(audioPath); | |
| } | |
| if (audioPaths.length === 0) { | |
| this.log('No audio files found to combine.'); | |
| return; | |
| } | |
| if (audioPaths.length === 1) { | |
| // Only one audio file, just copy it | |
| fs.copyFileSync(audioPaths[0], outAudioPath); | |
| this.log(`Single audio file copied to ${outAudioPath}`); | |
| return; | |
| } | |
| this.log(`Joining ${audioPaths.length} audio files into ${outAudioPath}`); | |
| await FFMpegUtils.joinAudios(audioPaths, outAudioPath); | |
| this.log(`Combined audio saved to ${outAudioPath}`); | |
| } | |
| // Combines all audio captions for all transcripts into a single audio caption file taking care of offset. The combined audio caption file is saved to the specified output path. | |
| async combineAllAudioCaptions(transcripts, outAudioCaptionPath) { | |
| // flatten path first for each audio caption file before processing this.mediaPathFlatten(..); | |
| let combinedTranscriptText = ''; | |
| let combinedWords = []; | |
| let cumulativeOffset = 0; | |
| for (let section of transcripts) { | |
| let captionFilePath = section.audioCaptionFile; | |
| if (!captionFilePath) { | |
| // No caption file for this section, just advance the offset | |
| cumulativeOffset += section.durationInSeconds || 0; | |
| continue; | |
| } | |
| // If the caption file is an .ass file, check for the original JSON source | |
| if (path.extname(captionFilePath) === '.ass') { | |
| // Try the _audioCaptionFile (original JSON before ASS conversion) if available | |
| if (section._audioCaptionFile) { | |
| captionFilePath = section._audioCaptionFile; | |
| } else { | |
| // Try converting .ass path back to .json | |
| captionFilePath = captionFilePath.replace('.ass', '.json'); | |
| } | |
| } | |
| // Resolve the caption file path | |
| let resolvedPath = captionFilePath; | |
| if (!fs.existsSync(resolvedPath)) { | |
| // Try with absolute path from cwd | |
| resolvedPath = path.join(process.cwd(), captionFilePath); | |
| } | |
| if (!fs.existsSync(resolvedPath)) { | |
| // Try flattened path | |
| const flattenedPath = this.mediaPathFlatten(captionFilePath); | |
| if (fs.existsSync(flattenedPath)) { | |
| resolvedPath = flattenedPath; | |
| this.log(`Using flattened caption path: ${flattenedPath}`); | |
| } else { | |
| const flattenedAbsPath = path.join(process.cwd(), flattenedPath); | |
| if (fs.existsSync(flattenedAbsPath)) { | |
| resolvedPath = flattenedAbsPath; | |
| this.log(`Using flattened absolute caption path: ${flattenedAbsPath}`); | |
| } else { | |
| this.log(`Caption file not found: ${captionFilePath}. Skipping.`); | |
| cumulativeOffset += section.durationInSeconds || 0; | |
| continue; | |
| } | |
| } | |
| } | |
| try { | |
| const captionData = JSON.parse(fs.readFileSync(resolvedPath, 'utf-8')); | |
| const sectionTranscript = captionData.transcript || ''; | |
| const sectionWords = captionData.words || []; | |
| // Append transcript text with a space separator | |
| if (combinedTranscriptText.length > 0 && sectionTranscript.length > 0) { | |
| combinedTranscriptText += ' '; | |
| } | |
| combinedTranscriptText += sectionTranscript; | |
| // Adjust word timings with cumulative offset and add to combined list | |
| for (let word of sectionWords) { | |
| const adjustedWord = { ...word }; | |
| if (adjustedWord.start !== undefined && adjustedWord.start !== null) { | |
| adjustedWord.start = adjustedWord.start + cumulativeOffset; | |
| } | |
| if (adjustedWord.end !== undefined && adjustedWord.end !== null) { | |
| adjustedWord.end = adjustedWord.end + cumulativeOffset; | |
| } | |
| combinedWords.push(adjustedWord); | |
| } | |
| } catch (err) { | |
| this.log(`Error reading caption file ${resolvedPath}: ${err}`); | |
| } | |
| cumulativeOffset += section.durationInSeconds || 0; | |
| } | |
| // Write the combined caption file | |
| const combinedCaptions = { | |
| transcript: combinedTranscriptText, | |
| words: combinedWords, | |
| start: 0, | |
| end: cumulativeOffset | |
| }; | |
| fs.writeFileSync(outAudioCaptionPath, JSON.stringify(combinedCaptions, null, 2)); | |
| this.log(`Combined ${combinedWords.length} caption words from ${transcripts.length} sections into ${outAudioCaptionPath}`); | |
| return combinedCaptions; | |
| } | |
| } | |