fix(stats): stop counting duplicate typeset subtitle lines (#191)

This commit is contained in:
2026-08-14 00:12:56 -07:00
committed by GitHub
parent 046e74ea91
commit b98d4d65c7
34 changed files with 2644 additions and 35 deletions
@@ -0,0 +1,467 @@
import assert from 'node:assert/strict';
import fs from 'node:fs';
import os from 'node:os';
import path from 'node:path';
import test from 'node:test';
import { Database } from '../sqlite.js';
import type { DatabaseSync } from '../sqlite.js';
import { ensureSchema } from '../storage.js';
import { cleanupDuplicateSubtitleLines } from '../duplicate-line-cleanup.js';
const DAY_MS = 86_400_000;
const BASE_MS = 1_700_000_000_000;
const WORD_ID = 1;
interface SeedLine {
session: number;
text: string;
startMs: number;
endMs: number;
/** Recording wall-clock, i.e. what the lookback window filters on. */
createdMs?: number;
}
function makeDbPath(): string {
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'subminer-duplicate-line-test-'));
return path.join(dir, 'immersion.sqlite');
}
function cleanupDbPath(dbPath: string): void {
const dir = path.dirname(dbPath);
if (!fs.existsSync(dir)) return;
fs.rmSync(dir, { recursive: true, force: true });
}
/** One episode, two sessions of it, and one word occurrence per seeded line. */
function seed(db: DatabaseSync, lines: SeedLine[]): void {
db.exec(`
INSERT INTO imm_anime(anime_id, normalized_title_key, canonical_title, CREATED_DATE, LAST_UPDATE_DATE)
VALUES (1, 'show', 'Show', ${BASE_MS}, ${BASE_MS});
INSERT INTO imm_videos(video_id, video_key, anime_id, canonical_title, source_type, watched, duration_ms, CREATED_DATE, LAST_UPDATE_DATE)
VALUES (1, 'v1', 1, 'Ep 1', 1, 1, 1440000, ${BASE_MS}, ${BASE_MS});
INSERT INTO imm_sessions(session_id, session_uuid, video_id, started_at_ms, ended_at_ms, status, CREATED_DATE, LAST_UPDATE_DATE)
VALUES (1, 's1', 1, '${BASE_MS}', '${BASE_MS + 1000}', 2, ${BASE_MS}, ${BASE_MS}),
(2, 's2', 1, '${BASE_MS + DAY_MS}', '${BASE_MS + DAY_MS + 1000}', 2, ${BASE_MS}, ${BASE_MS});
INSERT INTO imm_words(id, headword, word, reading, part_of_speech, pos1, first_seen, last_seen, frequency)
VALUES (${WORD_ID}, '飛び上がる', '飛び上がる', '', 'verb', '動詞', ${Math.floor(BASE_MS / 1000)}, ${Math.floor(BASE_MS / 1000)}, 0);
`);
const insertLine = db.prepare(
`INSERT INTO imm_subtitle_lines(
line_id, session_id, video_id, anime_id, line_index,
segment_start_ms, segment_end_ms, text, CREATED_DATE, LAST_UPDATE_DATE)
VALUES (?, ?, 1, 1, ?, ?, ?, ?, ?, ?)`,
);
const insertOccurrence = db.prepare(
`INSERT INTO imm_word_line_occurrences(line_id, word_id, occurrence_count, seen_ms)
VALUES (?, ?, 1, ?)`,
);
lines.forEach((line, index) => {
const lineId = index + 1;
const lineIndex = index + 1;
const createdMs = line.createdMs ?? BASE_MS;
insertLine.run(
lineId,
line.session,
lineIndex,
line.startMs,
line.endMs,
line.text,
createdMs,
createdMs,
);
insertOccurrence.run(lineId, WORD_ID, createdMs);
});
db.exec(`
UPDATE imm_words SET frequency = (
SELECT COALESCE(SUM(o.occurrence_count), 0)
FROM imm_word_line_occurrences o WHERE o.word_id = imm_words.id
)
`);
}
function createDb(lines: SeedLine[]): { db: DatabaseSync; dbPath: string } {
const dbPath = makeDbPath();
const db = new Database(dbPath);
ensureSchema(db);
seed(db, lines);
return { db, dbPath };
}
/** A typeset line mpv reported once per animation frame. */
function karaokeFrames(
session: number,
text: string,
startMs: number,
frames: number,
frameMs: number,
): SeedLine[] {
return Array.from({ length: frames }, (_, index) => ({
session,
text,
startMs: startMs + index * frameMs,
endMs: startMs + (index + 1) * frameMs,
}));
}
function countLines(db: DatabaseSync): number {
return (db.prepare('SELECT COUNT(*) AS total FROM imm_subtitle_lines').get() as { total: number })
.total;
}
function wordFrequency(db: DatabaseSync): number {
const row = db.prepare('SELECT frequency FROM imm_words WHERE id = ?').get(WORD_ID) as {
frequency: number;
} | null;
return row?.frequency ?? 0;
}
test('a karaoke burst collapses to one line and gives back its word counts', () => {
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 40, 40),
{ session: 1, text: 'おはよう', startMs: 20_000, endMs: 22_000 },
]);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 1);
assert.equal(summary.removedLines, 39);
assert.equal(summary.removedWordOccurrences, 39);
assert.equal(countLines(db), 2);
assert.equal(wordFrequency(db), 2);
// The surviving line covers the whole run, the way the parsed cue would.
const kept = db
.prepare(
'SELECT segment_start_ms AS startMs, segment_end_ms AS endMs FROM imm_subtitle_lines WHERE line_id = 1',
)
.get() as { startMs: number; endMs: number };
assert.equal(kept.startMs, 10_000);
assert.equal(kept.endMs, 10_000 + 40 * 40);
assert.equal(summary.samples.length, 1);
assert.equal(summary.samples[0]!.text, '飛び上がる');
assert.equal(summary.samples[0]!.frames, 40);
assert.equal(summary.samples[0]!.videoTitle, 'Ep 1');
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('ordinary repeated dialogue survives', () => {
// Six contiguous `飛び上がる`, each held for a normal beat rather than a frame.
const lines = Array.from({ length: 6 }, (_, index) => ({
session: 1,
text: '飛び上がる',
startMs: 5_000 + index * 800,
endMs: 5_000 + (index + 1) * 800,
}));
const { db, dbPath } = createDb(lines);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 0);
assert.equal(summary.removedLines, 0);
assert.equal(countLines(db), 6);
assert.equal(wordFrequency(db), 6);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a long run of quarter-second frames is still a burst', () => {
// Between the timing-only bound (0.1s) and the animation-frame bound (0.3s): heavier
// typesetting lands here, and the run length is what makes it conclusive.
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 250));
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 1);
assert.equal(summary.removedLines, 5);
assert.equal(countLines(db), 1);
assert.equal(wordFrequency(db), 1);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a qualifying short-frame burst may end with one long hold frame', () => {
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 8, 40),
{ session: 1, text: '飛び上がる', startMs: 10_320, endMs: 12_320 },
]);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 1);
assert.equal(summary.removedLines, 8);
assert.equal(countLines(db), 1);
assert.equal(wordFrequency(db), 1);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a long event before the final frame prevents burst cleanup', () => {
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 5, 40),
{ session: 1, text: '飛び上がる', startMs: 10_200, endMs: 12_200 },
{ session: 1, text: '飛び上がる', startMs: 12_200, endMs: 12_240 },
]);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 7);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a run of frames longer than the animation bound survives', () => {
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 400));
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 6);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('the four-frame residue the live gate stores is cleaned up', () => {
// The streaming gate records the first four frames of a burst before the run is long
// enough to recognise. Four contiguous identical events under the strict timing-only
// bound are that residue, and no real dialogue.
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 4, 40));
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 1);
assert.equal(summary.removedLines, 3);
assert.equal(countLines(db), 1);
assert.equal(wordFrequency(db), 1);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a four-frame run above the strict frame bound survives', () => {
// Long enough per event to be plausible dialogue; only a five-event run may use the
// looser animation-frame bound.
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 4, 250));
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 4);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('an explicit minRunLength raises the bar', () => {
// Five quarter-second frames qualify under the defaults; a cautious run asking for six
// leaves them alone. Above the strict bound, so the residue rule stays out of it.
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 5, 250));
try {
const preview = cleanupDuplicateSubtitleLines(db, { dryRun: true });
assert.equal(preview.burstGroups, 1);
const summary = cleanupDuplicateSubtitleLines(db, { minRunLength: 6 });
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 5);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('an explicit maxFrameSeconds tightens the frame bound', () => {
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 250));
try {
const summary = cleanupDuplicateSubtitleLines(db, { maxFrameSeconds: 0.2 });
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 6);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a non-finite maxFrameSeconds falls back to the default bound', () => {
// Six normal-beat lines: Infinity must not turn every event into a "short frame".
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 800));
try {
const summary = cleanupDuplicateSubtitleLines(db, { maxFrameSeconds: Infinity });
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 6);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('sampleLimit zero removes bursts but reports no samples', () => {
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 40, 40));
try {
const summary = cleanupDuplicateSubtitleLines(db, { sampleLimit: 0 });
assert.equal(summary.removedLines, 39);
assert.deepEqual(summary.samples, []);
assert.equal(countLines(db), 1);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a short run below every threshold survives', () => {
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 3, 40));
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 0);
assert.equal(countLines(db), 3);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('interleaved dual-line karaoke collapses each line to one row', () => {
// Kanji and romaji lines frame-flipped together, the way fansub OPs are typeset. The
// rows arrive interleaved in time order; each text must still chain into its own run.
const kanji = karaokeFrames(1, '飛び上がる', 10_000, 20, 60);
const romaji = karaokeFrames(1, 'tobiagaru', 10_001, 20, 60);
const interleaved = [...kanji, ...romaji].sort((a, b) => a.startMs - b.startMs);
const { db, dbPath } = createDb(interleaved);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 2);
assert.equal(summary.removedLines, 38);
assert.equal(countLines(db), 2);
assert.equal(wordFrequency(db), 2);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('the same line in a rewatch session is never merged into the first watch', () => {
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40),
...karaokeFrames(2, '飛び上がる', 10_000, 6, 40),
]);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 2);
assert.equal(summary.removedLines, 10);
// One surviving line per session, not one across both.
assert.equal(countLines(db), 2);
assert.equal(wordFrequency(db), 2);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a gap between runs splits them', () => {
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40),
...karaokeFrames(1, '飛び上がる', 60_000, 6, 40),
]);
try {
const summary = cleanupDuplicateSubtitleLines(db);
assert.equal(summary.burstGroups, 2);
assert.equal(countLines(db), 2);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('a dry run reports what an apply would do and writes nothing', () => {
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 40, 40));
try {
const preview = cleanupDuplicateSubtitleLines(db, { dryRun: true });
assert.equal(preview.dryRun, true);
assert.equal(preview.removedLines, 39);
assert.equal(countLines(db), 40);
assert.equal(wordFrequency(db), 40);
const applied = cleanupDuplicateSubtitleLines(db);
assert.equal(applied.removedLines, preview.removedLines);
assert.equal(applied.removedWordOccurrences, preview.removedWordOccurrences);
assert.equal(countLines(db), 1);
} finally {
db.close();
cleanupDbPath(dbPath);
}
});
test('the lookback window leaves older bursts alone', () => {
const recentMs = BASE_MS;
const oldMs = BASE_MS - 40 * DAY_MS;
const { db, dbPath } = createDb([
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40).map((line) => ({
...line,
createdMs: oldMs,
})),
...karaokeFrames(2, '飛び上がる', 10_000, 6, 40).map((line) => ({
...line,
createdMs: recentMs,
})),
]);
globalThis.__subminerTestNowMs = BASE_MS;
try {
const summary = cleanupDuplicateSubtitleLines(db, { lookbackDays: 30 });
assert.equal(summary.lookbackDays, 30);
assert.equal(summary.scannedLines, 6);
assert.equal(summary.burstGroups, 1);
assert.equal(summary.removedLines, 5);
// Six untouched old frames plus the one surviving recent line.
assert.equal(countLines(db), 7);
assert.equal(wordFrequency(db), 7);
} finally {
globalThis.__subminerTestNowMs = undefined;
db.close();
cleanupDbPath(dbPath);
}
});
@@ -0,0 +1,423 @@
/*
* Retroactive removal of animation-burst subtitle lines from the stats database.
*
* Before the live ingest gate existed, a karaoke OP recorded one line -- and one count
* for every word in it -- per animation frame, which is enough to put an OP lyric at the
* top of "Top Repeated Words" for good. This module finds those runs in what is already
* stored and takes them back down to one line.
*
* Only timing is available here: the stored text has been stripped of ASS markup, so the
* authoring evidence the file-level parser uses (`\t`, `\move`, karaoke timing, a
* changing override signature) is long gone. What is left is a run of identical,
* contiguous, short-lived lines inside a single session.
*
* The run has to be as long as the timing-only rule in `subtitle-cue-dedup` demands, but
* its short frames may be as long as the animation-frame bound rather than the much
* tighter timing-only one. A qualifying run may end with one longer hold, which is a
* common karaoke shape. Five or more repeats of the same text, each ending where the next
* begins, is already conclusive on its own -- no dialogue does that -- and the tighter
* bound would walk straight past the heavier typesetting that motivated this, where
* frames sit nearer a quarter of a second. Both bounds are options, so a cautious run can
* ask for more, and a dry run always reports before anything is removed.
*
* Scope: subtitle lines, their word/kanji occurrences, and the `imm_words`/`imm_kanji`
* aggregates those occurrences feed. Session telemetry (`lines_seen`, `tokens_seen`) and
* the rollups derived from it are left alone; they are cumulative samples taken at record
* time, and for sessions whose raw rows have since been pruned they cannot be recomputed.
*/
import type { DatabaseSync } from './sqlite';
import {
ANIMATION_FRAME_MAX_SECONDS,
DUPLICATE_CUE_GAP_TOLERANCE_SECONDS,
MIN_STREAM_RESIDUE_FRAMES,
MIN_TIMING_ONLY_FRAMES,
TIMING_ONLY_FRAME_MAX_SECONDS,
} from '../subtitle-burst-constants';
import {
applyLexicalRemovals,
makePlaceholders,
planLexicalRemovalsForLines,
toDbTimestamp,
} from './query-shared';
import { nowMs } from './time';
const MS_PER_DAY = 86_400_000;
/** SQLite caps bound parameters per statement; stay well under it. */
const ID_BATCH_SIZE = 400;
const DEFAULT_SAMPLE_LIMIT = 20;
export interface DuplicateSubtitleLineCleanupOptions {
/** Only consider lines recorded within this many days. Null or omitted = all history. */
lookbackDays?: number | null;
/** Measure without writing. */
dryRun?: boolean;
/** Identical contiguous lines needed before a run counts as an animation. */
minRunLength?: number;
/** Longest a single event may last and still look like an animation frame. */
maxFrameSeconds?: number;
/** How many of the largest runs to describe in the summary. */
sampleLimit?: number;
}
export interface DuplicateSubtitleLineBurst {
sessionId: number;
videoId: number;
text: string;
/** Kept line, extended to cover the whole run. */
keptLineId: number;
removedLineIds: number[];
startMs: number;
endMs: number;
}
export interface DuplicateSubtitleLineSample {
videoId: number;
videoTitle: string | null;
text: string;
frames: number;
removedLines: number;
startMs: number;
endMs: number;
}
export interface DuplicateSubtitleLineCleanupSummary {
dryRun: boolean;
lookbackDays: number | null;
scannedLines: number;
burstGroups: number;
removedLines: number;
removedWordOccurrences: number;
removedKanjiOccurrences: number;
samples: DuplicateSubtitleLineSample[];
}
export interface StoredSubtitleLineRow {
lineId: number;
sessionId: number;
videoId: number;
text: string;
startMs: number;
endMs: number;
}
interface ResolvedBounds {
lookbackDays: number | null;
minRunLength: number;
maxFrameMs: number;
/** Shorter runs qualify only when every event sits under this much stricter bound. */
residueMinRunLength: number;
strictFrameMs: number;
gapToleranceMs: number;
sampleLimit: number;
}
function resolveBounds(options: DuplicateSubtitleLineCleanupOptions): ResolvedBounds {
const lookbackDays =
typeof options.lookbackDays === 'number' && Number.isFinite(options.lookbackDays)
? Math.max(1, Math.floor(options.lookbackDays))
: null;
const minRunLength =
typeof options.minRunLength === 'number' && Number.isFinite(options.minRunLength)
? Math.max(2, Math.floor(options.minRunLength))
: MIN_TIMING_ONLY_FRAMES;
const maxFrameSeconds =
typeof options.maxFrameSeconds === 'number' &&
Number.isFinite(options.maxFrameSeconds) &&
options.maxFrameSeconds > 0
? options.maxFrameSeconds
: ANIMATION_FRAME_MAX_SECONDS;
const sampleLimit =
typeof options.sampleLimit === 'number' && options.sampleLimit >= 0
? Math.floor(options.sampleLimit)
: DEFAULT_SAMPLE_LIMIT;
return {
lookbackDays,
minRunLength,
maxFrameMs: Math.round(maxFrameSeconds * 1000),
residueMinRunLength: Math.max(MIN_STREAM_RESIDUE_FRAMES, minRunLength - 1),
strictFrameMs: Math.round(TIMING_ONLY_FRAME_MAX_SECONDS * 1000),
gapToleranceMs: Math.round(DUPLICATE_CUE_GAP_TOLERANCE_SECONDS * 1000),
sampleLimit,
};
}
/**
* `CREATED_DATE` holds epoch milliseconds on rows this app wrote, but older and synced
* rows can carry seconds, so normalize before comparing against the cutoff.
*/
const CREATED_MS_SQL = `
CASE
WHEN sl.CREATED_DATE < 10000000000 THEN sl.CREATED_DATE * 1000
ELSE sl.CREATED_DATE
END`;
function readCandidateLines(db: DatabaseSync, bounds: ResolvedBounds): StoredSubtitleLineRow[] {
const scope =
bounds.lookbackDays === null
? ''
: `AND sl.CREATED_DATE IS NOT NULL AND ${CREATED_MS_SQL} >= ?`;
const params = bounds.lookbackDays === null ? [] : [nowMs() - bounds.lookbackDays * MS_PER_DAY];
return db
.prepare(
`SELECT
sl.line_id AS lineId,
sl.session_id AS sessionId,
sl.video_id AS videoId,
sl.text AS text,
sl.segment_start_ms AS startMs,
sl.segment_end_ms AS endMs
FROM imm_subtitle_lines sl
WHERE sl.segment_start_ms IS NOT NULL
AND sl.segment_end_ms IS NOT NULL
${scope}
ORDER BY sl.session_id, sl.video_id, sl.segment_start_ms, sl.line_id`,
)
.all(...params) as StoredSubtitleLineRow[];
}
function isBurst(run: StoredSubtitleLineRow[], bounds: ResolvedBounds): boolean {
const isShortFrame = (row: StoredSubtitleLineRow): boolean =>
row.endMs - row.startMs <= bounds.maxFrameMs;
// The residue the live gate leaves behind: it records the first frames of a burst
// before the run is long enough to recognise, so one frame fewer than the timing-only
// minimum, every one under the strict timing-only bound. No dialogue holds identical
// sub-tenth-second lines back to back that many times.
if (
run.length >= bounds.residueMinRunLength &&
run.every((row) => row.endMs - row.startMs <= bounds.strictFrameMs)
) {
return true;
}
if (run.length < bounds.minRunLength) {
return false;
}
if (run.every(isShortFrame)) {
return true;
}
// Karaoke commonly finishes its short animation frames with one long hold. Only the
// final event may exceed the frame bound, and the short frames before it must already
// meet the minimum run length on their own.
return (
run.length - 1 >= bounds.minRunLength &&
run.slice(0, -1).every(isShortFrame) &&
!isShortFrame(run[run.length - 1]!)
);
}
function toBurst(run: StoredSubtitleLineRow[]): DuplicateSubtitleLineBurst {
const [first] = run;
return {
sessionId: first!.sessionId,
videoId: first!.videoId,
text: first!.text,
keptLineId: first!.lineId,
removedLineIds: run.slice(1).map((row) => row.lineId),
startMs: first!.startMs,
endMs: run.reduce((latest, row) => Math.max(latest, row.endMs), first!.endMs),
};
}
/**
* Group stored lines into animation runs.
*
* Rows are bucketed per (session, video, text) before chaining, the way the file-level
* dedup buckets cues: dual-line karaoke interleaves two texts frame by frame, and
* chaining across the interleave would break every run at length one.
*
* Runs never cross a session, which is what keeps a rewatch intact: the same episode
* watched twice stores the same line twice, and those two belong to different sessions.
*/
export function findDuplicateSubtitleLineBursts(
rows: readonly StoredSubtitleLineRow[],
options: DuplicateSubtitleLineCleanupOptions = {},
): DuplicateSubtitleLineBurst[] {
const bounds = resolveBounds(options);
// Insertion order preserves the query's startMs ordering within each bucket.
const rowsByKey = new Map<string, StoredSubtitleLineRow[]>();
for (const row of rows) {
const key = `${row.sessionId}|${row.videoId}|${row.text}`;
const bucket = rowsByKey.get(key);
if (bucket) {
bucket.push(row);
} else {
rowsByKey.set(key, [row]);
}
}
const bursts: DuplicateSubtitleLineBurst[] = [];
for (const bucket of rowsByKey.values()) {
if (bucket.length < 2) {
continue;
}
let run: StoredSubtitleLineRow[] = [];
let chainEndMs = 0;
const closeRun = (): void => {
if (run.length > 1 && isBurst(run, bounds)) {
bursts.push(toBurst(run));
}
run = [];
};
for (const row of bucket) {
if (run.length > 0 && row.startMs <= chainEndMs + bounds.gapToleranceMs) {
run.push(row);
chainEndMs = Math.max(chainEndMs, row.endMs);
continue;
}
closeRun();
run = [row];
chainEndMs = row.endMs;
}
closeRun();
}
return bursts;
}
function chunk<T>(values: T[], size: number): T[][] {
const chunks: T[][] = [];
for (let i = 0; i < values.length; i += size) {
chunks.push(values.slice(i, i + size));
}
return chunks;
}
function buildSamples(
db: DatabaseSync,
bursts: DuplicateSubtitleLineBurst[],
sampleLimit: number,
): DuplicateSubtitleLineSample[] {
if (sampleLimit === 0 || bursts.length === 0) {
return [];
}
const largest = [...bursts]
.sort((a, b) => b.removedLineIds.length - a.removedLineIds.length)
.slice(0, sampleLimit);
const videoIds = [...new Set(largest.map((burst) => burst.videoId))];
const titles = new Map<number, string>();
for (const batch of chunk(videoIds, ID_BATCH_SIZE)) {
const rows = db
.prepare(
`SELECT video_id AS videoId, canonical_title AS title
FROM imm_videos
WHERE video_id IN (${makePlaceholders(batch)})`,
)
.all(...batch) as Array<{ videoId: number; title: string | null }>;
for (const row of rows) {
if (row.title) titles.set(row.videoId, row.title);
}
}
return largest.map((burst) => ({
videoId: burst.videoId,
videoTitle: titles.get(burst.videoId) ?? null,
text: burst.text,
frames: burst.removedLineIds.length + 1,
removedLines: burst.removedLineIds.length,
startMs: burst.startMs,
endMs: burst.endMs,
}));
}
function sumRemovedOccurrences(
db: DatabaseSync,
table: 'imm_word_line_occurrences' | 'imm_kanji_line_occurrences',
lineIds: number[],
): number {
let total = 0;
for (const batch of chunk(lineIds, ID_BATCH_SIZE)) {
const row = db
.prepare(
`SELECT COALESCE(SUM(occurrence_count), 0) AS total
FROM ${table}
WHERE line_id IN (${makePlaceholders(batch)})`,
)
.get(...batch) as { total: number } | null;
total += row?.total ?? 0;
}
return total;
}
function applyBursts(db: DatabaseSync, bursts: DuplicateSubtitleLineBurst[]): void {
const removedLineIds = bursts.flatMap((burst) => burst.removedLineIds);
const currentMs = toDbTimestamp(nowMs());
db.exec('BEGIN IMMEDIATE');
try {
for (const batch of chunk(removedLineIds, ID_BATCH_SIZE)) {
const placeholders = makePlaceholders(batch);
// Measured before the delete, applied after it: `applyLexicalRemovals` checks the
// surviving occurrences to decide whether a zeroed count really means the word is
// gone, so the rows it inspects have to be the post-delete ones.
const plan = planLexicalRemovalsForLines(db, batch);
db.prepare(`DELETE FROM imm_word_line_occurrences WHERE line_id IN (${placeholders})`).run(
...batch,
);
db.prepare(`DELETE FROM imm_kanji_line_occurrences WHERE line_id IN (${placeholders})`).run(
...batch,
);
db.prepare(`DELETE FROM imm_subtitle_lines WHERE line_id IN (${placeholders})`).run(...batch);
applyLexicalRemovals(db, plan);
}
const extendStmt = db.prepare(
`UPDATE imm_subtitle_lines
SET segment_end_ms = ?, LAST_UPDATE_DATE = ?
WHERE line_id = ? AND (segment_end_ms IS NULL OR segment_end_ms < ?)`,
);
for (const burst of bursts) {
extendStmt.run(burst.endMs, currentMs, burst.keptLineId, burst.endMs);
}
db.exec('COMMIT');
} catch (error) {
try {
db.exec('ROLLBACK');
} catch {
// Surface the transaction failure, not the rollback's.
}
throw error;
}
}
/**
* Collapse stored animation bursts down to one line each.
*
* A dry run measures exactly what an apply would remove, using the same scan, so the
* numbers shown in a confirmation prompt are the numbers that will happen.
*/
export function cleanupDuplicateSubtitleLines(
db: DatabaseSync,
options: DuplicateSubtitleLineCleanupOptions = {},
): DuplicateSubtitleLineCleanupSummary {
const bounds = resolveBounds(options);
const dryRun = options.dryRun === true;
const rows = readCandidateLines(db, bounds);
const bursts = findDuplicateSubtitleLineBursts(rows, options);
const removedLineIds = bursts.flatMap((burst) => burst.removedLineIds);
const summary: DuplicateSubtitleLineCleanupSummary = {
dryRun,
lookbackDays: bounds.lookbackDays,
scannedLines: rows.length,
burstGroups: bursts.length,
removedLines: removedLineIds.length,
removedWordOccurrences: sumRemovedOccurrences(db, 'imm_word_line_occurrences', removedLineIds),
removedKanjiOccurrences: sumRemovedOccurrences(
db,
'imm_kanji_line_occurrences',
removedLineIds,
),
samples: buildSamples(db, bursts, bounds.sampleLimit),
};
if (dryRun || removedLineIds.length === 0) {
return summary;
}
applyBursts(db, bursts);
return summary;
}
@@ -276,6 +276,19 @@ export function planLexicalRemovalsForSessions(
return planLexicalRemovals(db, `sl.session_id IN (${makePlaceholders(sessionIds)})`, sessionIds);
}
/**
* Measure what deleting these individual subtitle lines removes from the vocabulary
* tables. Used by the duplicate-line cleanup, which drops animation frames out of the
* middle of sessions that otherwise stay intact.
*/
export function planLexicalRemovalsForLines(
db: DatabaseSync,
lineIds: number[],
): LexicalRemovalPlan {
if (lineIds.length === 0) return EMPTY_LEXICAL_REMOVAL_PLAN;
return planLexicalRemovals(db, `sl.line_id IN (${makePlaceholders(lineIds)})`, lineIds);
}
/** Measure what deleting these videos removes from the vocabulary tables. */
export function planLexicalRemovalsForVideos(
db: DatabaseSync,