Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 25 additions & 0 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
name: Tests

on:
push:
pull_request:

permissions:
contents: read

jobs:
test:
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest, windows-latest]
python-version: ["3.10", "3.13"]
steps:
- uses: actions/checkout@v4
- name: Set up uv
uses: astral-sh/setup-uv@v6
with:
python-version: ${{ matrix.python-version }}
- name: Run tests
run: uv run pytest
142 changes: 127 additions & 15 deletions pymusiclooper/analysis.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@

import librosa
import numpy as np
import scipy.signal
from numba import njit

from pymusiclooper.audio import MLAudio
Expand Down Expand Up @@ -172,13 +173,11 @@ def find_best_loop_points(
mlaudio, chroma, bpm, candidate_pairs, disable_pruning
)

# prefer longer loops for highly similar sequences
if len(filtered_candidate_pairs) > 1:
_prioritize_duration(filtered_candidate_pairs)

# Set the exact loop start and end in samples and adjust them
# to the nearest zero crossing. Avoids audio popping/clicking while looping
# as much as possible.
# Set the exact loop start and end in samples. The frame-level points are refined by lining up
# the waveforms at the loop points, then choosing the seam where they differ the least;
# if the waveforms do not match well enough for that, each point is moved to its nearest zero crossing.
# Avoids audio popping/clicking while looping as much as possible.
mono_playback_audio = mlaudio.playback_audio.mean(axis=1)
for pair in filtered_candidate_pairs:
if mlaudio.trim_offset > 0:
pair._loop_start_frame_idx = int(
Expand All @@ -187,22 +186,24 @@ def find_best_loop_points(
pair._loop_end_frame_idx = int(
mlaudio.apply_trim_offset(pair._loop_end_frame_idx)
)
pair.loop_start = nearest_zero_crossing(
mlaudio.playback_audio,
mlaudio.rate,
mlaudio.frames_to_samples(pair._loop_start_frame_idx)
)
pair.loop_end = nearest_zero_crossing(
pair.loop_start, pair.loop_end = _refine_loop_points(
mlaudio.playback_audio,
mono_playback_audio,
mlaudio.rate,
mlaudio.frames_to_samples(pair._loop_end_frame_idx)
int(mlaudio.frames_to_samples(pair._loop_start_frame_idx)),
int(mlaudio.frames_to_samples(pair._loop_end_frame_idx)),
)

if not filtered_candidate_pairs:
raise LoopNotFoundError(
f"No loop points found for {mlaudio.filename} with current parameters."
)

# prefer longer loops for highly similar sequences
# (must run after the loop positions in samples are set, since it compares loop durations)
if len(filtered_candidate_pairs) > 1:
_prioritize_duration(filtered_candidate_pairs)

logging.info(
f"Filtered to {len(filtered_candidate_pairs)} best candidate loop points"
)
Expand Down Expand Up @@ -522,8 +523,12 @@ def _calculate_subseq_beat_similarity(
cosine_sim = dot_prod / (np.maximum(b1_norm * b2_norm, 1e-10))

if max_offset < test_length:
# Pad the missing frames on the side farthest from the loop point:
# after the tested frames when looking ahead, before them when looking behind
missing_frames = test_length - max_offset
pad_width = (missing_frames, 0) if test_end_offset < 0 else (0, missing_frames)
return np.average(
np.pad(cosine_sim, pad_width=(0, test_length - max_offset), mode="constant", constant_values=0),
np.pad(cosine_sim, pad_width=pad_width, mode="constant", constant_values=0),
weights=weights,
)
else:
Expand All @@ -534,6 +539,113 @@ def _weights(length: int, start: int = 100, stop: int = 1):
return np.geomspace(start, stop, num=length)


# STFT hop length used by the analysis (librosa's default); frame-level loop points are only this precise
_HOP_LENGTH = 512
# How far (in samples) the loop end may be moved to line up the waveforms
_ALIGNMENT_SEARCH_RADIUS = 2 * _HOP_LENGTH
# Length of the audio compared on each side of the loop points when aligning them (in seconds)
_ALIGNMENT_HALF_WINDOW = 0.075
# Below this correlation, the audio around the loop points does not match well enough
# for lining up the waveforms to be reliable (found by comparing both methods on real tracks)
_MIN_ALIGNMENT_CORRELATION = 0.3


def _refine_loop_points(
audio: np.ndarray, mono_audio: np.ndarray, rate: int, loop_start: int, loop_end: int
) -> Tuple[int, int]:
"""Refines frame-level loop points to exact sample positions.

Lines up the waveforms at the loop points, then picks the seam where they differ the least.
If the waveforms do not match well enough, each point is moved to its nearest zero crossing instead.

Args:
audio (np.ndarray): Playback audio, in the shape `(samples, n_channels)`
mono_audio (np.ndarray): The playback audio mixed down to mono, in the shape `(samples,)`
rate (int): Sample rate of the audio
loop_start (int): Approximate loop start in samples
loop_end (int): Approximate loop end in samples

Returns:
Tuple[int, int]: The refined (loop_start, loop_end)
"""
aligned_loop_end, correlation = _align_loop_end(mono_audio, rate, loop_start, loop_end)
if correlation >= _MIN_ALIGNMENT_CORRELATION:
return _best_seam(audio, rate, loop_start, aligned_loop_end)
return (
nearest_zero_crossing(audio, rate, loop_start),
nearest_zero_crossing(audio, rate, loop_end),
)


def _align_loop_end(mono_audio: np.ndarray, rate: int, loop_start: int, loop_end: int) -> Tuple[int, float]:
"""Moves the loop end (within `_ALIGNMENT_SEARCH_RADIUS` samples) to where the waveform around it best matches
the waveform around the loop start, using normalized cross-correlation.

Args:
mono_audio (np.ndarray): Mono playback audio, in the shape `(samples,)`
rate (int): Sample rate of the audio
loop_start (int): Loop start in samples
loop_end (int): Approximate loop end in samples

Returns:
Tuple[int, float]: The aligned loop end, and the correlation of the waveforms at that point (-1 to 1).
The correlation is 0 if there was not enough audio around the loop points to compare.
"""
# The loop end must stay after the loop start
radius = min(_ALIGNMENT_SEARCH_RADIUS, loop_end - loop_start - 1)
half_window = int(_ALIGNMENT_HALF_WINDOW * rate)
n_samples = mono_audio.shape[0]

# Compare as much audio as available on each side, up to half_window
before = min(half_window, loop_start, loop_end - radius)
after = min(half_window, n_samples - loop_start, n_samples - loop_end - radius)
if radius < 0 or before + after < _HOP_LENGTH:
return loop_end, 0.0

reference = mono_audio[loop_start - before:loop_start + after].astype(np.float64)
search = mono_audio[loop_end - before - radius:loop_end + after + radius].astype(np.float64)

# correlation[i] compares the reference with the search window shifted by (i - radius) samples
correlation = scipy.signal.correlate(search, reference, mode="valid", method="fft")
cumulative_energy = np.concatenate(([0.0], np.cumsum(search**2)))
window_energy = cumulative_energy[reference.size:] - cumulative_energy[:-reference.size]
normalized = correlation / np.sqrt(np.maximum(window_energy * np.dot(reference, reference), 1e-20))

best = int(np.argmax(normalized))
return loop_end + best - radius, float(normalized[best])


def _best_seam(audio: np.ndarray, rate: int, loop_start: int, loop_end: int) -> Tuple[int, int]:
"""Shifts both loop points by the same amount (keeping the loop length) to where the waveforms at the
loop start and loop end differ the least, within +/-5ms. The difference at the jump is what causes clicks.

Args:
audio (np.ndarray): Playback audio, in the shape `(samples, n_channels)`
rate (int): Sample rate of the audio
loop_start (int): Loop start in samples
loop_end (int): Loop end in samples

Returns:
Tuple[int, int]: The shifted (loop_start, loop_end)
"""
radius = max(1, rate // 200)
# Compare the two samples before and after each point
offsets = np.arange(-2, 2)
lowest_shift = max(-radius, -(loop_start + offsets[0]))
highest_shift = min(radius, audio.shape[0] - 1 - (loop_end + offsets[-1]))
if lowest_shift > highest_shift:
return loop_start, loop_end

shifts = np.arange(lowest_shift, highest_shift + 1)
idx = shifts[:, np.newaxis] + offsets[np.newaxis, :]
difference = np.abs(audio[loop_start + idx] - audio[loop_end + idx]).sum(axis=(1, 2))
# Prefer the smallest shift among equally good ones
difference += 1e-6 * np.abs(shifts)

shift = int(shifts[np.argmin(difference)])
return loop_start + shift, loop_end + shift


@njit(cache=True)
def nearest_zero_crossing(audio: np.ndarray, rate: int, sample_idx: int) -> int:
"""Implementation of Audacity's `At Zero Crossings` feature. https://manual.audacityteam.org/man/select_menu_at_zero_crossings.html
Expand Down
3 changes: 2 additions & 1 deletion pymusiclooper/audio.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,7 +48,8 @@ def __init__(self, filepath: str) -> None:
raise AudioLoadError(f"\"{filepath}\" only contains silence and cannot be analyzed.")

# Normalize audio channels to between -1.0 and +1.0 before analysis
mono_signal /= np.max(np.abs(mono_signal))
# (not in-place: for mono input, to_mono returns raw_audio itself, which is also used for playback/export)
mono_signal = mono_signal / np.max(np.abs(mono_signal))

self.audio, self.trim_offset = librosa.effects.trim(mono_signal, top_db=40)
self.trim_offset = self.trim_offset[0]
Expand Down
19 changes: 12 additions & 7 deletions pymusiclooper/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -206,10 +206,12 @@ def extend(
samples_to_fade = min(
self.mlaudio.seconds_to_samples(fade_length), final_loop.shape[0]
)
final_loop[-samples_to_fade:] = (
final_loop[-samples_to_fade:]
* np.linspace(1, 0, samples_to_fade)[:, np.newaxis]
)
# Guard against x[-0:], which would select the whole section
if samples_to_fade > 0:
final_loop[-samples_to_fade:] = (
final_loop[-samples_to_fade:]
* np.linspace(1, 0, samples_to_fade)[:, np.newaxis]
)

# Format extended file name with its duration suffixed
extended_loop_length = final_loop.shape[0] + (
Expand Down Expand Up @@ -272,9 +274,10 @@ def export_txt(
loop_end: Union[str, int, float, str],
txt_name: str = "loops",
output_dir: Optional[str] = None
):
"""Exports the given loop points to a text file named `loop.txt` in append mode with the format:
) -> str:
"""Exports the given loop points to a text file named `loops.txt` in append mode with the format:
`{loop_start} {loop_end} {filename}`
Returns the path to the text file.

Args:
loop_start (Union[int, float, str]): Loop start in samples, seconds or ftime.
Expand All @@ -290,6 +293,8 @@ def export_txt(
with open(out_path, "a") as file:
file.write(f"{loop_start} {loop_end} {self.mlaudio.filename}\n")

return out_path


def _find_start_tag(
self,
Expand Down Expand Up @@ -377,7 +382,7 @@ def export_tags(
import taglib

if output_dir is None:
output_dir = os.path.abspath(self.mlaudio.filepath)
output_dir = os.path.dirname(os.path.abspath(self.mlaudio.filepath))

track_name, file_extension = os.path.splitext(self.mlaudio.filename)

Expand Down
9 changes: 4 additions & 5 deletions pymusiclooper/handler.py
Original file line number Diff line number Diff line change
Expand Up @@ -132,11 +132,11 @@ def get_user_input():
preview = False

if num_input == "more":
self.interactive_handler(show_top=show_top * 2)
return self.interactive_handler(show_top=show_top * 2)
if num_input == "all":
self.interactive_handler(show_top=total_candidates)
return self.interactive_handler(show_top=total_candidates)
if num_input == "reset":
self.interactive_handler()
return self.interactive_handler()

if num_input[-1] == "p":
idx = int(num_input[:-1])
Expand Down Expand Up @@ -322,12 +322,11 @@ def txt_export_runner(self, loop_start: int, loop_end: int):
if self.alt_export_top != 0:
self.alt_export_runner(mode="TXT")
else:
self.musiclooper.export_txt(
out_path = self.musiclooper.export_txt(
self._fmt(loop_start),
self._fmt(loop_end),
output_dir=self.output_directory,
)
out_path = os.path.join(self.output_directory, "loop.txt")
message = f'Successfully added "{self.musiclooper.filename}" loop points to "{out_path}"'
if self.batch_mode:
logging.info(message)
Expand Down
9 changes: 9 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,7 @@ dependencies = [
"click-params>=0.5.0,<0.6",
"click-option-group>=0.5.6,<0.6",
"lazy-loader>=0.3",
"scipy>=1.6.0",
]

[project.urls]
Expand Down Expand Up @@ -64,3 +65,11 @@ select = [
# isort
"I",
]

[dependency-groups]
dev = [
"pytest>=9.1.1",
]

[tool.pytest.ini_options]
testpaths = ["tests"]
Loading