Skip to content

Commit 75491cc

Browse files
committed
Fix long-video caption preparation failure
1 parent d86bb0a commit 75491cc

6 files changed

Lines changed: 186 additions & 21 deletions

File tree

‎app/src/main/java/com/kienhoang/dualsubreplay/data/SubtitleStore.kt‎

Lines changed: 38 additions & 16 deletions
Original file line numberDiff line numberDiff line change
@@ -47,7 +47,9 @@ internal class SubtitleStore private constructor(
4747
/** Call on an IO dispatcher. Each read owns its handle so cancellation cannot leak it. */
4848
suspend fun read(indices: IntRange): List<SubtitleSegment> {
4949
if (indices.isEmpty()) return emptyList()
50-
require(indices.first >= 0 && indices.last < size)
50+
require(indices.first >= 0 && indices.last < size) {
51+
"Subtitle window ${indices.first}..${indices.last} is outside 0..${size - 1}."
52+
}
5153
return FileInputStream(file).use { source ->
5254
source.channel.position(offsets[indices.first])
5355
DataInputStream(BufferedInputStream(source)).use { input ->
@@ -71,24 +73,44 @@ internal class SubtitleStore private constructor(
7173
suspend fun create(
7274
directory: File,
7375
segments: List<SubtitleSegment>,
76+
): SubtitleStore = create(directory) { append -> append(segments) }
77+
78+
/**
79+
* Writes caption batches directly to disk. This keeps formatting a very long transcript
80+
* bounded too; lazy translation is not useful if preparation first materializes the whole
81+
* track a second time.
82+
*/
83+
suspend fun create(
84+
directory: File,
85+
writeBatches: suspend (append: suspend (List<SubtitleSegment>) -> Unit) -> Unit,
7486
): SubtitleStore {
75-
require(segments.size <= MAX_SEGMENTS) { "Caption track contains too many entries." }
7687
check(directory.isDirectory || directory.mkdirs()) { "Cannot create subtitle storage." }
7788
val file = File.createTempFile("transcript-", ".bin", directory)
7889
try {
79-
val starts = LongArray(segments.size)
80-
val offsets = LongArray(segments.size)
90+
val starts = mutableListOf<Long>()
91+
val offsets = mutableListOf<Long>()
8192
DataOutputStream(BufferedOutputStream(FileOutputStream(file))).use { output ->
82-
segments.forEachIndexed { index, segment ->
83-
currentCoroutineContext().ensureActive()
84-
require(index == 0 || segment.startMs >= starts[index - 1])
85-
starts[index] = segment.startMs
86-
offsets[index] = output.size().toLong()
87-
output.writeSegment(segment)
88-
check(output.size().toLong() <= MAX_STORE_BYTES) { "Caption storage limit exceeded." }
93+
writeBatches { batch ->
94+
require(starts.size + batch.size <= MAX_SEGMENTS) {
95+
"Caption track contains too many entries."
96+
}
97+
batch.forEach { sourceSegment ->
98+
currentCoroutineContext().ensureActive()
99+
val segment = sourceSegment
100+
// Overlapping YouTube ASR cues can make a split chunk start after the
101+
// following cue. Keep the display order, but clamp only the lookup key
102+
// so binary search remains valid instead of crashing with an unnamed
103+
// IllegalArgumentException ("Failed requirement").
104+
starts += maxOf(segment.startMs, starts.lastOrNull() ?: Long.MIN_VALUE)
105+
offsets += output.size().toLong()
106+
output.writeSegment(segment)
107+
check(output.size().toLong() <= MAX_STORE_BYTES) {
108+
"Caption storage limit exceeded."
109+
}
110+
}
89111
}
90112
}
91-
return SubtitleStore(file, starts, offsets)
113+
return SubtitleStore(file, starts.toLongArray(), offsets.toLongArray())
92114
} catch (error: Throwable) {
93115
file.delete()
94116
throw error
@@ -102,7 +124,7 @@ private fun DataOutput.writeSegment(segment: SubtitleSegment) {
102124
writeLong(segment.startMs)
103125
writeLong(segment.endMs)
104126
writeText(segment.originalText)
105-
require(segment.words.size <= 16_384)
127+
require(segment.words.size <= 16_384) { "A caption contains too many timed words." }
106128
writeInt(segment.words.size)
107129
segment.words.forEach { word ->
108130
writeText(word.text)
@@ -117,21 +139,21 @@ private fun DataInput.readSegment(): SubtitleSegment {
117139
val end = readLong()
118140
val text = readText()
119141
val count = readInt()
120-
require(count in 0..16_384)
142+
require(count in 0..16_384) { "Stored caption word count is invalid." }
121143
val words = List(count) { SubtitleWord(readText(), readLong(), readLong()) }
122144
return SubtitleSegment(id, start, end, text, words = words)
123145
}
124146

125147
private fun DataOutput.writeText(text: String) {
126148
val bytes = text.toByteArray(Charsets.UTF_8)
127-
require(bytes.size <= 1024 * 1024)
149+
require(bytes.size <= 1024 * 1024) { "A caption text entry is larger than 1 MiB." }
128150
writeInt(bytes.size)
129151
write(bytes)
130152
}
131153

132154
private fun DataInput.readText(): String {
133155
val count = readInt()
134-
require(count in 0..1024 * 1024)
156+
require(count in 0..1024 * 1024) { "Stored caption text length is invalid." }
135157
val bytes = ByteArray(count)
136158
readFully(bytes)
137159
return bytes.toString(Charsets.UTF_8)

‎app/src/main/java/com/kienhoang/dualsubreplay/ui/AppViewModel.kt‎

Lines changed: 7 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -1068,9 +1068,13 @@ class AppViewModel internal constructor(
10681068
}
10691069
try {
10701070
withContext(Dispatchers.IO) {
1071-
val raw = rawStore.read(0 until rawStore.size)
1072-
val display = captionDisplaySegments(raw, format, natural)
1073-
displayStore = SubtitleStore.create(subtitleDirectory, display)
1071+
displayStore =
1072+
prepareCaptionDisplayStore(
1073+
source = rawStore,
1074+
directory = subtitleDirectory,
1075+
format = format,
1076+
natural = natural,
1077+
)
10741078
}
10751079
translator.withSession(sourceLanguage, targetLanguage, onDownloadingChange = { downloading ->
10761080
_state.update { current ->

‎app/src/main/java/com/kienhoang/dualsubreplay/ui/CaptionFormat.kt‎

Lines changed: 60 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -2,6 +2,8 @@ package com.kienhoang.dualsubreplay.ui
22

33
import com.kienhoang.dualsubreplay.data.SubtitleMerger
44
import com.kienhoang.dualsubreplay.data.SubtitleSegment
5+
import com.kienhoang.dualsubreplay.data.SubtitleStore
6+
import java.io.File
57

68
internal const val CAPTION_FORMAT_PREFERENCE = "caption_format"
79

@@ -25,9 +27,20 @@ internal fun captionDisplaySegments(
2527
source: List<SubtitleSegment>,
2628
format: CaptionFormat,
2729
natural: Boolean,
30+
): List<SubtitleSegment> =
31+
captionDisplaySegments(
32+
source = source,
33+
units = sentenceCaptionUnits(source, source, natural),
34+
format = format,
35+
)
36+
37+
private fun captionDisplaySegments(
38+
source: List<SubtitleSegment>,
39+
units: List<CaptionTranslationUnit>,
40+
format: CaptionFormat,
2841
): List<SubtitleSegment> {
2942
val sentences =
30-
sentenceCaptionUnits(source, source, natural).mapIndexed { index, unit ->
43+
units.mapIndexed { index, unit ->
3144
SubtitleSegment(
3245
id = index.toLong(),
3346
startMs = source[unit.indices.first()].startMs,
@@ -38,3 +51,49 @@ internal fun captionDisplaySegments(
3851
}
3952
return if (format == CaptionFormat.SHORT_PHRASES) SubtitleMerger.splitLongSegments(sentences) else sentences
4053
}
54+
55+
internal const val CAPTION_PREPARATION_BATCH_SIZE = 256
56+
57+
/**
58+
* Builds the presentation track without loading the complete source track into memory. For natural
59+
* captions only the last open sentence unit crosses a batch boundary, so retaining that unit makes
60+
* the batched result identical to formatting one continuous list.
61+
*/
62+
internal suspend fun prepareCaptionDisplayStore(
63+
source: SubtitleStore,
64+
directory: File,
65+
format: CaptionFormat,
66+
natural: Boolean,
67+
batchSize: Int = CAPTION_PREPARATION_BATCH_SIZE,
68+
): SubtitleStore {
69+
require(batchSize > 0) { "Caption preparation batch size must be positive." }
70+
return SubtitleStore.create(directory) { append ->
71+
var cursor = 0
72+
var nextId = 0L
73+
var carry = emptyList<SubtitleSegment>()
74+
while (cursor < source.size) {
75+
val end = minOf(cursor + batchSize, source.size)
76+
val buffered = carry + source.read(cursor until end)
77+
val finalBatch = end == source.size
78+
val units = sentenceCaptionUnits(buffered, buffered, natural)
79+
val readyUnits =
80+
if (natural && !finalBatch && units.isNotEmpty()) {
81+
units.dropLast(1)
82+
} else {
83+
units
84+
}
85+
val ready =
86+
captionDisplaySegments(buffered, readyUnits, format).map { segment ->
87+
segment.copy(id = nextId++)
88+
}
89+
append(ready)
90+
carry =
91+
if (natural && !finalBatch && units.isNotEmpty()) {
92+
units.last().indices.map(buffered::get)
93+
} else {
94+
emptyList()
95+
}
96+
cursor = end
97+
}
98+
}
99+
}

‎app/src/test/java/com/kienhoang/dualsubreplay/data/SubtitleStoreTest.kt‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -34,6 +34,21 @@ class SubtitleStoreTest {
3434
withStore(rows) { store -> check(store.read(store.windowIndices(0)) == rows) }
3535
}
3636

37+
@Test
38+
fun overlappingSplitCueOrderDoesNotCrashTheLookupIndex() =
39+
runBlocking {
40+
val rows =
41+
listOf(
42+
SubtitleSegment(0, 0, 1_000, "first"),
43+
SubtitleSegment(1, 4_000, 5_000, "later chunk from the first cue"),
44+
SubtitleSegment(2, 1_000, 2_000, "overlapping next cue"),
45+
)
46+
withStore(rows) { store ->
47+
check(store.read(0..2) == rows)
48+
check(store.read(store.windowIndices(1_500)).any { it.originalText == "overlapping next cue" })
49+
}
50+
}
51+
3752
@Test
3853
fun closingAStoreDeletesItsTranscriptFile() =
3954
runBlocking {

‎app/src/test/java/com/kienhoang/dualsubreplay/ui/CaptionFormatTest.kt‎

Lines changed: 61 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,9 +1,12 @@
11
package com.kienhoang.dualsubreplay.ui
22

33
import com.kienhoang.dualsubreplay.data.SubtitleSegment
4+
import com.kienhoang.dualsubreplay.data.SubtitleStore
45
import com.kienhoang.dualsubreplay.data.SubtitleWord
6+
import kotlinx.coroutines.runBlocking
57
import org.junit.Assert.*
68
import org.junit.Test
9+
import java.nio.file.Files
710

811
class CaptionFormatTest {
912
@Test fun defaultAndLegacyPreferencesAreStable() {
@@ -41,4 +44,62 @@ class CaptionFormatTest {
4144
assertTrue(whole.all { it.endMs - it.startMs <= 12000 && it.originalText.length <= 240 })
4245
assertEquals(source.joinToString(" ") { it.originalText }, whole.joinToString(" ") { it.originalText })
4346
}
47+
48+
@Test fun batchedPreparationMatchesContinuousFormatting() =
49+
runBlocking {
50+
val rows =
51+
List(80) { index ->
52+
SubtitleSegment(
53+
index.toLong(),
54+
index * 1_000L,
55+
(index + 1) * 1_000L,
56+
if (index % 9 == 8) "fragment $index." else "fragment $index",
57+
)
58+
}
59+
val expected = captionDisplaySegments(rows, CaptionFormat.SHORT_PHRASES, true)
60+
val directory = Files.createTempDirectory("caption-format-batches").toFile()
61+
try {
62+
SubtitleStore.create(directory, rows).use { source ->
63+
prepareCaptionDisplayStore(
64+
source = source,
65+
directory = directory,
66+
format = CaptionFormat.SHORT_PHRASES,
67+
natural = true,
68+
batchSize = 7,
69+
).use { display ->
70+
assertEquals(expected, display.read(0 until display.size))
71+
}
72+
}
73+
} finally {
74+
directory.deleteRecursively()
75+
}
76+
}
77+
78+
@Test fun overlappingAutoCaptionsCanBeSplitWithoutFailedRequirement() =
79+
runBlocking {
80+
val longText = "one two three four five six seven eight nine ten eleven twelve thirteen fourteen"
81+
val rows =
82+
listOf(
83+
SubtitleSegment(0, 0, 6_000, longText),
84+
SubtitleSegment(1, 1_000, 2_000, "overlapping next caption"),
85+
)
86+
val directory = Files.createTempDirectory("caption-overlap-regression").toFile()
87+
try {
88+
SubtitleStore.create(directory, rows).use { source ->
89+
prepareCaptionDisplayStore(
90+
source = source,
91+
directory = directory,
92+
format = CaptionFormat.SHORT_PHRASES,
93+
natural = false,
94+
batchSize = 1,
95+
).use { display ->
96+
val stored = display.read(0 until display.size)
97+
assertTrue(stored.size > rows.size)
98+
assertTrue(stored.any { it.originalText == "overlapping next caption" })
99+
}
100+
}
101+
} finally {
102+
directory.deleteRecursively()
103+
}
104+
}
44105
}

‎docs/qa/long-video-subtitles.md‎

Lines changed: 5 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,9 @@ Subtitles follow the player's actual position. No translation starts until a pla
66

77
Seeking replaces the pending window. At most the current ML Kit request is allowed to finish; its result cannot publish at an obsolete seek position. Pausing or backgrounding stops additional work. Leaving a video cancels its session and its in-flight caption HTTP request. Changing caption format or target language replaces the translation session; returning to a previously translated sentence uses the cache.
88

9-
Source and display transcripts are held in temporary indexed disk files, with only timestamps/offsets and the current window retained in memory. IDs and original word timestamps survive window reads, preserving replay and karaoke timing. The UI receives a snapshot of at most 96 entries, instead of a copy of the entire transcript after every translation.
9+
Source and display transcripts are held in temporary indexed disk files, with only timestamps/offsets and the current window retained in memory. Display formatting reads 256 source entries at a time and carries only the final open sentence across batches. IDs and original word timestamps survive window reads, preserving replay and karaoke timing. The UI receives a snapshot of at most 96 entries, instead of a copy of the entire transcript after every translation.
10+
11+
Auto-generated captions may overlap. Splitting one long cue can therefore produce display order such as `0s, 4s, 1s`. The disk index clamps only its binary-search key to a monotonic value while retaining each cue's real timestamp and display order, so overlapping speech no longer trips a sorting assertion or loses karaoke timing.
1012

1113
Translation uses one lazily opened ML Kit client per session, plus the existing bounded memory cache and a persistent LRU cache limited to 2,048 entries / 4 MiB of UTF-8 translated text. Cache keys include exact source text and both languages. Transcript files are removed on session exit and leftovers are cleared when a new ViewModel first opens its transcript storage.
1214

@@ -22,6 +24,8 @@ The existing YouTube caption endpoint returns a complete timed-text document. Th
2224
- Cancellation even when a provider finishes after cancellation.
2325
- Unknown playback position and initially paused playback do not start translation.
2426
- Dense 100,000-entry transcript: windows remain bounded and contain the current cue.
27+
- Batched display preparation is identical to continuous formatting across sentence boundaries.
28+
- Overlapping split captions remain readable and do not fail the disk-index requirement.
2529
- Unicode, cue IDs, and word-timing round trips, empty tracks, and transcript cleanup.
2630
- Persistent cache reuse, language/text isolation, LRU entry eviction, UTF-8 byte caps, missing files, and interrupted writes.
2731

0 commit comments

Comments
 (0)