worship / aligner.py
yancolasino's picture
Upload 8 files
fc64f7d verified
Raw
History Blame Contribute Delete
5.29 kB
import re
def align_chords_and_lyrics(lyrics_lines: list, chords: list) -> str:
"""
Aligns detected chords with lyrics lines using word-level timestamps.
Returns: A complete ChordPro string with inline chords (e.g., [C]Hosanna [G]in [Am]the [F]highest).
"""
if not lyrics_lines:
return ""
# If no chords were detected, just return the lyrics
if not chords:
return "\n".join([line["text"] for line in lyrics_lines])
chordpro_lines = []
chord_idx = 0
num_chords = len(chords)
# We want to keep track of chords that have already been placed
# to avoid placing duplicate chords or skipping chords
last_placed_chord_time = -1
for line_idx, line in enumerate(lyrics_lines):
line_text = line["text"]
words = line.get("words", [])
# If the line is empty or has no words, just keep it as is
if not words:
chordpro_lines.append(line_text)
continue
line_start = line["start"]
line_end = line["end"]
# 1. Check if there are any chords that occurred BEFORE this line started,
# but AFTER the previous line ended. These represent intro/bridge chords.
intro_chords = []
while chord_idx < num_chords and chords[chord_idx]["time"] < line_start:
c_info = chords[chord_idx]
# Avoid duplicate placements
if c_info["time"] > last_placed_chord_time and c_info["chord"] != "N":
intro_chords.append(f"[{c_info['chord']}]")
last_placed_chord_time = c_info["time"]
chord_idx += 1
if intro_chords:
# Place intro/instrumental chords on their own line
chordpro_lines.append(" ".join(intro_chords))
# 2. Align chords with words inside the current line
line_buffer = ""
for w_idx, word_info in enumerate(words):
word = word_info["word"]
w_start = word_info["start"]
w_end = word_info["end"]
# Find all chords that fall in this word's timeframe
# Or just before this word (but after the previous word)
word_chords = []
# Lookahead for chords within this word's window
temp_idx = chord_idx
while temp_idx < num_chords:
c_info = chords[temp_idx]
c_time = c_info["time"]
# If chord falls before the end of this word
if c_time <= w_end:
# And it occurred after the previous word ended (or since last placed)
if c_time > last_placed_chord_time:
if c_info["chord"] != "N":
word_chords.append(c_info["chord"])
last_placed_chord_time = c_time
temp_idx += 1
else:
break
# Update the main chord index pointer
chord_idx = max(chord_idx, temp_idx)
# Construct the word with its inline chords
if word_chords:
# Add the chords at the front of the word
chord_prefix = "".join([f"[{c}]" for c in word_chords])
line_buffer += f"{chord_prefix}{word} "
else:
line_buffer += f"{word} "
chordpro_lines.append(line_buffer.strip())
# 3. Add any remaining chords that occur after the last line of lyrics
outro_chords = []
while chord_idx < num_chords:
c_info = chords[chord_idx]
if c_info["time"] > last_placed_chord_time and c_info["chord"] != "N":
outro_chords.append(f"[{c_info['chord']}]")
last_placed_chord_time = c_info["time"]
chord_idx += 1
if outro_chords:
chordpro_lines.append("")
chordpro_lines.append(" ".join(outro_chords))
# Clean up multiple spaces and return the joined lines
final_output = "\n".join(chordpro_lines)
return final_output
if __name__ == "__main__":
# Test alignment logic
mock_lyrics = [
{
"text": "Hosanna, Hosanna",
"start": 1.0,
"end": 3.0,
"words": [
{"word": "Hosanna,", "start": 1.0, "end": 1.9},
{"word": "Hosanna", "start": 2.0, "end": 3.0}
]
},
{
"text": "Hosanna in the Highest",
"start": 3.5,
"end": 6.0,
"words": [
{"word": "Hosanna", "start": 3.5, "end": 4.2},
{"word": "in", "start": 4.3, "end": 4.6},
{"word": "the", "start": 4.7, "end": 5.0},
{"word": "Highest", "start": 5.1, "end": 6.0}
]
}
]
mock_chords = [
{"chord": "E/G#", "time": 0.8},
{"chord": "A", "time": 1.5},
{"chord": "B", "time": 2.1},
{"chord": "C#m", "time": 2.8},
{"chord": "A", "time": 3.6},
{"chord": "C#m", "time": 4.5},
{"chord": "B", "time": 5.3}
]
print(align_chords_and_lyrics(mock_lyrics, mock_chords))