Skip to content

Fix and reverify karaoke sync upgrade #1

Fix and reverify karaoke sync upgrade

Fix and reverify karaoke sync upgrade #1

name: Fix and verify karaoke sync upgrade once
on:
push:
branches: [fix/karaoke-preserve-word-timing]
paths: [".github/workflows/karaoke-sync-fixup-once.yml"]
permissions:
contents: write
jobs:
apply-fix-and-verify:
runs-on: ubuntu-latest
timeout-minutes: 35
steps:
- name: Check out PR branch
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
with:
ref: fix/karaoke-preserve-word-timing
persist-credentials: true
- name: Set up Java 17
uses: actions/setup-java@cf277c60eb25467037889841efdb72551f06f6c3 # v4
with:
distribution: temurin
java-version: "17"
- name: Set up Android SDK
uses: android-actions/setup-android@9fc6c4e9069bf8d3d10b2204b1fb8f6ef7065407 # v3
- name: Install Android build SDK
run: sdkmanager "platforms;android-36" "build-tools;36.0.0"
- name: Apply original options 1-4 upgrade
shell: bash
run: |
python - <<'PY'
from pathlib import Path
import base64, gzip, re, subprocess
workflow = Path('.github/workflows/karaoke-sync-upgrade-once.yml').read_text()
match = re.search(r"printf '%s' '([A-Za-z0-9+/=]+)' \\| base64", workflow)
if not match:
raise SystemExit('Could not recover the original upgrade payload')
script = gzip.decompress(base64.b64decode(match.group(1)))
path = Path('/tmp/apply-karaoke-sync.py')
path.write_bytes(script)
subprocess.run(['python', str(path)], check=True)
PY
- name: Fix timing model and estimator
shell: bash
run: |
cat > app/src/main/java/com/kienhoang/dualsubreplay/data/SubtitleModels.kt <<'EOF'
package com.kienhoang.dualsubreplay.data
data class RawCaptionCue(
val startMs: Long,
val endMs: Long,
val text: String,
val words: List<SubtitleWord> = emptyList(),
)
data class SubtitleSegment(
val id: Long,
val startMs: Long,
val endMs: Long,
val originalText: String,
val translatedText: String? = null,
val words: List<SubtitleWord> = emptyList(),
)
enum class SubtitleTimingSource {
YOUTUBE_EXACT,
YOUTUBE_DOM_OBSERVED,
ESTIMATED,
}
/** A single spoken word/chunk with its absolute timing inside the video. */
data class SubtitleWord(
val text: String,
val startMs: Long,
val endMs: Long,
val timingSource: SubtitleTimingSource = SubtitleTimingSource.ESTIMATED,
)
data class CaptionTrackResult(
val languageCode: String,
val isGenerated: Boolean,
val cues: List<RawCaptionCue>,
val availableLanguages: List<CaptionLanguage> = emptyList(),
)
data class CaptionLanguage(
val code: String,
val name: String,
)
EOF
cat > app/src/main/java/com/kienhoang/dualsubreplay/data/WordTiming.kt <<'EOF'
package com.kienhoang.dualsubreplay.data
/**
* Word-level timing helpers shared by the caption parser, the merger, and the
* real-time spoken-word highlight in the UI.
*/
/**
* With a frame-synchronised WebView clock most polling latency disappears.
* Keep only a tiny render lead so Compose can paint the next word without
* visibly jumping ahead of the speaker.
*/
internal const val KARAOKE_HIGHLIGHT_LEAD_MS = 20L
private const val MIN_ESTIMATED_WORD_MS = 60L
private val sentencePause = Regex("[.!?。!?…]+[\\\"'’”)]*$")
private val clausePause = Regex("[,;:,;:]+[\\\"'’”)]*$")
private val latinVowelGroups = Regex("(?i)[aeiouy]+")
/**
* A speech-oriented fallback weight. It avoids giving very long written words
* an unrealistically huge share of a cue and reserves a little time for
* punctuation pauses. This is still explicitly ESTIMATED timing.
*/
private fun estimatedTimingWeight(token: String): Long {
val spokenCharacters = token.count(Char::isLetterOrDigit).coerceAtLeast(1)
val vowelGroups = latinVowelGroups.findAll(token).count()
val roughSyllables = maxOf(vowelGroups, (spokenCharacters + 2) / 3)
.coerceIn(1, 6)
val pauseWeight = when {
sentencePause.containsMatchIn(token) -> 3
clausePause.containsMatchIn(token) -> 2
else -> 0
}
return (roughSyllables + pauseWeight).toLong()
}
/**
* Splits text into words and estimates their boundaries from speech-oriented
* weights. When the cue is long enough, every word receives a small minimum
* slice before the remaining duration is distributed by the weights.
*/
internal fun estimateWordTimings(text: String, startMs: Long, endMs: Long): List<SubtitleWord> {
val tokens = text.split(Regex("\\s+")).filter(String::isNotBlank)
if (tokens.isEmpty() || endMs <= startMs) return emptyList()
val duration = endMs - startMs
val minimumPerWord = if (duration >= tokens.size * MIN_ESTIMATED_WORD_MS) {
MIN_ESTIMATED_WORD_MS
} else {
0L
}
val reserved = minimumPerWord * tokens.size
val distributable = (duration - reserved).coerceAtLeast(0L)
val weights = tokens.map(::estimatedTimingWeight)
val totalWeight = weights.sum().coerceAtLeast(1L)
var consumedWeight = 0L
var cursor = startMs
return tokens.mapIndexed { index, token ->
consumedWeight += weights[index]
val proportionalEnd = if (index == tokens.lastIndex) {
endMs
} else {
startMs +
minimumPerWord * (index + 1L) +
distributable * consumedWeight / totalWeight
}
val safeEnd = proportionalEnd.coerceIn(cursor, endMs)
val word = SubtitleWord(
text = token,
startMs = cursor,
endMs = safeEnd,
timingSource = SubtitleTimingSource.ESTIMATED,
)
cursor = safeEnd
word
}
}
/**
* Index of the word being spoken at [timeMs]. Between words the previously
* started word stays highlighted so short gaps do not flicker.
*/
internal fun activeWordIndex(words: List<SubtitleWord>, timeMs: Long): Int {
val firstWord = words.firstOrNull() ?: return -1
if (timeMs < firstWord.startMs) return -1
val syncTimeMs = timeMs + KARAOKE_HIGHLIGHT_LEAD_MS
var result = 0
for (index in 1 until words.size) {
if (words[index].startMs <= syncTimeMs) result = index else break
}
return result
}
EOF
- name: Test, lint, and build
run: |
set -euo pipefail
bash ./gradlew testDebugUnitTest lintDebug assembleDebug assembleDebugAndroidTest --stacktrace
- name: Commit verified upgrade
shell: bash
run: |
set -euo pipefail
git config user.name "github-actions[bot]"
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
git add app
git commit -m "Upgrade karaoke sync with frame clock and DOM hints"
git push origin HEAD:fix/karaoke-preserve-word-timing