Skip to content

Warp

mg_warp_audiovisual_beats

mg_warp_audiovisual_beats(self, audio_file, speed=(0.5, 2), data=None, filtertype='Adaptative', threshold=0.05, kernel_size=5, target_name=None, overwrite=True)

Warp audio beats with visual beats (patterns of motion that can be shifted in time to control visual rhythm). Visual beats are warped after computing a directogram which factors the magnitude of motion in the video into different angles.

Source: Abe Davis -- Visual Rhythm and Beat (section 5)

Parameters:

Name Type Description Default
audio_file str

Path to the audio file.

required
speed tuple

Speed's change between the audiovisual beats which can be adjusted to slow down or speed up the visual rhythms. Defaults to (0.5,2).

(0.5, 2)
data array_like

Computed directogram data can be added separately to avoid the directogram processing time (which can be quite long). Defaults to None.

None
filtertype str

'Regular' turns all values below threshold to 0. 'Binary' turns all values below threshold to 0, above threshold to 1. 'Blob' removes individual pixels with erosion method. 'Adaptative' perform adaptative threshold as the weighted sum of 11 neighborhood pixels where weights are a Gaussian window. Defaults to 'Adaptative'.

'Adaptative'
threshold float

Eliminates pixel values less than given threshold. Ranges from 0 to 1. Defaults to 0.05.

0.05
kernel_size int

Size of structuring element. Defaults to 5.

5
target_name str

Target output name for the directogram. Defaults to None (which assumes that the input filename with the suffix "_dg" should be used).

None
overwrite bool

Whether to allow overwriting existing files or to automatically increment target filenames to avoid overwriting. Defaults to True.

True

Returns:

Name Type Description
MgVideo 'musicalgestures.MgVideo'

A MgVideo as warp_audiovisual_beats for parent MgVideo

Source code in musicalgestures/_warp.py
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
def mg_warp_audiovisual_beats(self, audio_file: str, speed: tuple = (0.5, 2), data=None, filtertype: str = 'Adaptative', threshold: float = 0.05, kernel_size: int = 5, target_name: str | None = None, overwrite: bool = True) -> "musicalgestures.MgVideo":
    """
    Warp audio beats with visual beats (patterns of motion that can be shifted in time to control visual rhythm).
    Visual beats are warped after computing a directogram which factors the magnitude of motion in the video into different angles.

    Source: Abe Davis -- [Visual Rhythm and Beat](http://www.abedavis.com/files/papers/VisualRhythm_Davis18.pdf) (section 5)

    Args:
        audio_file (str): Path to the audio file.
        speed (tuple, optional): Speed's change between the audiovisual beats which can be adjusted to slow down or speed up the visual rhythms. Defaults to (0.5,2).
        data (array_like, optional): Computed directogram data can be added separately to avoid the directogram processing time (which can be quite long). Defaults to None.
        filtertype (str, optional): 'Regular' turns all values below `threshold` to 0. 'Binary' turns all values below `threshold` to 0, above `threshold` to 1. 'Blob' removes individual pixels with erosion method. 'Adaptative' perform adaptative threshold as the weighted sum of 11 neighborhood pixels where weights are a Gaussian window. Defaults to 'Adaptative'.
        threshold (float, optional): Eliminates pixel values less than given threshold. Ranges from 0 to 1. Defaults to 0.05.
        kernel_size (int, optional): Size of structuring element. Defaults to 5.
        target_name (str, optional): Target output name for the directogram. Defaults to None (which assumes that the input filename with the suffix "_dg" should be used).
        overwrite (bool, optional): Whether to allow overwriting existing files or to automatically increment target filenames to avoid overwriting. Defaults to True.

    Returns:
        MgVideo: A MgVideo as warp_audiovisual_beats for parent MgVideo
    """

    # COMPUTE DIRECTOGRAMS ------------------------------------------------------------------------------------------------------

    if data is None:
        directogram = mg_directograms(self, title=None, filtertype=filtertype, threshold=threshold, kernel_size=kernel_size, target_name=target_name, overwrite=overwrite)
        directograms = directogram.data['directogram']
        fps = directogram.data['FPS']

    else:
        directograms = data
        vidcap = cv2.VideoCapture(self.filename)
        fps = int(vidcap.get(cv2.CAP_PROP_FPS))

    # COMPUTE AUDIO AND VISUAL BEATS --------------------------------------------------------------------------------------------

    pb = MgProgressbar(total=130, prefix='Warping audiovisual beats:')
    pb.progress(0)
    # audio_source is a no-op for a real audio file and extracts the track if a video
    # is passed instead, which librosa 1.0 can no longer open by itself.
    signal, sr = librosa.load(audio_source(audio_file), mono=True)
    pb.progress(5)

    # Compute onset and impact envelopes
    onset_envelopes = librosa.onset.onset_strength(signal, sr=sr)
    pb.progress(10)
    impact_envelopes = impact_envelope(directograms)
    pb.progress(15)

    # Compute beats with librosa
    pb.progress(20)
    audio_beats = librosa.beat.beat_track(onset_envelope=onset_envelopes,sr=sr, hop_length=512, trim=False, units='samples')
    pb.progress(25)
    visual_beats = librosa.beat.beat_track(onset_envelope=impact_envelopes,sr=fps, hop_length=1.0, trim=False, units='samples')

    # WARP AUDIO AND VISUAL BEATS -----------------------------------------------------------------------------------------------

    _ensure_numba()  # JIT-compile beats_diff on first use
    audio_differences = beats_diff(audio_beats[1], signal)
    pb.progress(30)
    visual_differences = beats_diff(visual_beats[1], np.ndarray.flatten(directograms))
    pb.progress(35)

    # Asserting if the arrays have equal shape and elements
    assert np.array_equal(audio_beats[1], np.cumsum(audio_differences[:-1]))
    assert np.array_equal(visual_beats[1], np.cumsum(visual_differences[:-1]))

    pb.progress(40)
    audio_diff_size = audio_differences.size
    audio_differences_sync = []
    visual_differences_sync = []

    # Loop through each audio and visual beat difference index
    pb.progress(45)
    audio_index, visual_index = 0, 0
    audio_diff = audio_differences[audio_index]
    visual_diff = visual_differences[visual_index]

    pb.progress(50)
    while True:

        # Convert beat differences to time
        audio_time = audio_diff / sr
        visual_time = visual_diff / fps

        speed_change = visual_time / audio_time

        # If the visual beat difference is too short, we check the next index
        if speed_change < speed[0]:
            visual_index += 1
            if visual_index == visual_differences.shape[0]:
                break
            visual_diff = visual_differences[visual_index]

        # If the audio beat difference is too short, we check the next index
        elif speed[1] < speed_change:   
            audio_index += 1
            # Iterate continuously over the audio indexes until visual indexes reach the size of the visual differences array
            audio_diff += audio_differences[audio_index % audio_diff_size]

        else:
            audio_index += 1
            visual_index += 1
            audio_differences_sync.append(audio_diff)
            visual_differences_sync.append(visual_diff)
            if visual_index == visual_differences.shape[0]:
                break
            audio_diff = audio_differences[audio_index % audio_diff_size]
            visual_diff = visual_differences[visual_index]

    pb.progress(55)
    audio_beats_sync = np.cumsum(audio_differences_sync[:-1])
    pb.progress(60)
    visual_beats_sync = np.cumsum(visual_differences_sync[:-1])

    # RENDER AUDIOVISUAL BEATS --------------------------------------------------------------------------------------------------
    pb.progress(65)
    of, fex = os.path.splitext(self.filename)

    if target_name is None:
        target_name = of + '_warped.avi'
    else:
        # enforce avi
        target_name = os.path.splitext(target_name)[0] + '.avi'
    if not overwrite:
        target_name = generate_outfilename(target_name)

    pb.progress(70)
    new_length = audio_beats_sync[-1] + 1
    extended_file_name = f'{audio_file[:-4]}_{new_length}.wav'

    pb.progress(75)
    if not os.path.isfile(extended_file_name):
        data, sample_rate = librosa.load(audio_source(audio_file), mono=True)
        old_length = data.shape[0]
        tail = new_length - old_length * (new_length // old_length)
        extended_data = np.hstack(tuple([data] * (new_length // old_length) + [data[:tail]]))
        sf.write(extended_file_name, extended_data, sample_rate)

    pb.progress(80)
    if os.path.isfile(target_name):
        os.remove(target_name)

    pb.progress(85)        
    temp_file_name = of + '_temp.avi'
    filename = of + '.avi'

    pb.progress(90)
    vidcap = cv2.VideoCapture(filename)
    ret, frame = vidcap.read()
    pb.progress(95)
    output_stream = cv2.VideoWriter(temp_file_name, cv2.VideoWriter_fourcc(*'mp4v'), fps, (frame.shape[1], frame.shape[0]))

    pb.progress(100)
    if ret == True:

        # Iterate through each output frame until the final beat is reached
        last_audio_beat, last_visual_beat = 0, 0
        input_frame_index = 0

        for audio_beat, visual_beat in zip(audio_beats_sync, visual_beats_sync):
            # Output time range
            audio_start_time = last_audio_beat / sr
            audio_end_time = audio_beat / sr
            # Input time range
            visual_start_time = last_visual_beat / fps
            visual_end_time = visual_beat / fps
            # Conversion multiplier
            multiplier = (visual_end_time - visual_start_time) / (audio_end_time - audio_start_time)

            # Iterate through every output frame in the current beat range
            output_start_index = int(np.ceil(audio_start_time * fps))
            output_end_index = int(np.floor(audio_end_time * fps))

            for output_index in range(output_start_index, output_end_index + 1):

                output_time = output_index / fps
                input_time = (output_time - audio_start_time) * multiplier + visual_start_time
                input_index = round(input_time * fps)

                while input_frame_index < input_index:
                    ret, frame = vidcap.read()
                    input_frame_index += 1
                output_stream.write(frame)

            last_audio_beat, last_visual_beat = audio_beat, visual_beat

    # Close visual stream
    pb.progress(105)
    output_stream.release()
    pb.progress(110)
    vidcap.release()

    audio_file = extended_file_name

    pb.progress(115)
    cmd = f'ffmpeg -i {temp_file_name} -i {audio_file} -c:v copy -c:a aac -strict experimental -t {visual_beats_sync[-1] / fps} {wrap_str(target_name)}'

    pb.progress(120)
    subprocess.check_call(cmd, shell=True) 
    pb.progress(125)   
    os.remove(temp_file_name)
    pb.progress(130)

    # Save the warped video on the parent MgVideo. Use a distinct attribute name (not the method
    # name) so it doesn't shadow the warp_audiovisual_beats() method on the instance.
    self.warp_video = musicalgestures.MgVideo(target_name, color=self.color, returned_by_process=True)

    return self.warp_video