mirror of
https://github.com/ksyasuda/SubMiner.git
synced 2026-08-14 13:55:55 -07:00
fix(stats): stop counting duplicate typeset subtitle lines (#191)
This commit is contained in:
@@ -0,0 +1,467 @@
|
||||
import assert from 'node:assert/strict';
|
||||
import fs from 'node:fs';
|
||||
import os from 'node:os';
|
||||
import path from 'node:path';
|
||||
import test from 'node:test';
|
||||
import { Database } from '../sqlite.js';
|
||||
import type { DatabaseSync } from '../sqlite.js';
|
||||
import { ensureSchema } from '../storage.js';
|
||||
import { cleanupDuplicateSubtitleLines } from '../duplicate-line-cleanup.js';
|
||||
|
||||
const DAY_MS = 86_400_000;
|
||||
const BASE_MS = 1_700_000_000_000;
|
||||
const WORD_ID = 1;
|
||||
|
||||
interface SeedLine {
|
||||
session: number;
|
||||
text: string;
|
||||
startMs: number;
|
||||
endMs: number;
|
||||
/** Recording wall-clock, i.e. what the lookback window filters on. */
|
||||
createdMs?: number;
|
||||
}
|
||||
|
||||
function makeDbPath(): string {
|
||||
const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'subminer-duplicate-line-test-'));
|
||||
return path.join(dir, 'immersion.sqlite');
|
||||
}
|
||||
|
||||
function cleanupDbPath(dbPath: string): void {
|
||||
const dir = path.dirname(dbPath);
|
||||
if (!fs.existsSync(dir)) return;
|
||||
fs.rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
/** One episode, two sessions of it, and one word occurrence per seeded line. */
|
||||
function seed(db: DatabaseSync, lines: SeedLine[]): void {
|
||||
db.exec(`
|
||||
INSERT INTO imm_anime(anime_id, normalized_title_key, canonical_title, CREATED_DATE, LAST_UPDATE_DATE)
|
||||
VALUES (1, 'show', 'Show', ${BASE_MS}, ${BASE_MS});
|
||||
INSERT INTO imm_videos(video_id, video_key, anime_id, canonical_title, source_type, watched, duration_ms, CREATED_DATE, LAST_UPDATE_DATE)
|
||||
VALUES (1, 'v1', 1, 'Ep 1', 1, 1, 1440000, ${BASE_MS}, ${BASE_MS});
|
||||
INSERT INTO imm_sessions(session_id, session_uuid, video_id, started_at_ms, ended_at_ms, status, CREATED_DATE, LAST_UPDATE_DATE)
|
||||
VALUES (1, 's1', 1, '${BASE_MS}', '${BASE_MS + 1000}', 2, ${BASE_MS}, ${BASE_MS}),
|
||||
(2, 's2', 1, '${BASE_MS + DAY_MS}', '${BASE_MS + DAY_MS + 1000}', 2, ${BASE_MS}, ${BASE_MS});
|
||||
INSERT INTO imm_words(id, headword, word, reading, part_of_speech, pos1, first_seen, last_seen, frequency)
|
||||
VALUES (${WORD_ID}, '飛び上がる', '飛び上がる', '', 'verb', '動詞', ${Math.floor(BASE_MS / 1000)}, ${Math.floor(BASE_MS / 1000)}, 0);
|
||||
`);
|
||||
|
||||
const insertLine = db.prepare(
|
||||
`INSERT INTO imm_subtitle_lines(
|
||||
line_id, session_id, video_id, anime_id, line_index,
|
||||
segment_start_ms, segment_end_ms, text, CREATED_DATE, LAST_UPDATE_DATE)
|
||||
VALUES (?, ?, 1, 1, ?, ?, ?, ?, ?, ?)`,
|
||||
);
|
||||
const insertOccurrence = db.prepare(
|
||||
`INSERT INTO imm_word_line_occurrences(line_id, word_id, occurrence_count, seen_ms)
|
||||
VALUES (?, ?, 1, ?)`,
|
||||
);
|
||||
|
||||
lines.forEach((line, index) => {
|
||||
const lineId = index + 1;
|
||||
const lineIndex = index + 1;
|
||||
const createdMs = line.createdMs ?? BASE_MS;
|
||||
insertLine.run(
|
||||
lineId,
|
||||
line.session,
|
||||
lineIndex,
|
||||
line.startMs,
|
||||
line.endMs,
|
||||
line.text,
|
||||
createdMs,
|
||||
createdMs,
|
||||
);
|
||||
insertOccurrence.run(lineId, WORD_ID, createdMs);
|
||||
});
|
||||
|
||||
db.exec(`
|
||||
UPDATE imm_words SET frequency = (
|
||||
SELECT COALESCE(SUM(o.occurrence_count), 0)
|
||||
FROM imm_word_line_occurrences o WHERE o.word_id = imm_words.id
|
||||
)
|
||||
`);
|
||||
}
|
||||
|
||||
function createDb(lines: SeedLine[]): { db: DatabaseSync; dbPath: string } {
|
||||
const dbPath = makeDbPath();
|
||||
const db = new Database(dbPath);
|
||||
ensureSchema(db);
|
||||
seed(db, lines);
|
||||
return { db, dbPath };
|
||||
}
|
||||
|
||||
/** A typeset line mpv reported once per animation frame. */
|
||||
function karaokeFrames(
|
||||
session: number,
|
||||
text: string,
|
||||
startMs: number,
|
||||
frames: number,
|
||||
frameMs: number,
|
||||
): SeedLine[] {
|
||||
return Array.from({ length: frames }, (_, index) => ({
|
||||
session,
|
||||
text,
|
||||
startMs: startMs + index * frameMs,
|
||||
endMs: startMs + (index + 1) * frameMs,
|
||||
}));
|
||||
}
|
||||
|
||||
function countLines(db: DatabaseSync): number {
|
||||
return (db.prepare('SELECT COUNT(*) AS total FROM imm_subtitle_lines').get() as { total: number })
|
||||
.total;
|
||||
}
|
||||
|
||||
function wordFrequency(db: DatabaseSync): number {
|
||||
const row = db.prepare('SELECT frequency FROM imm_words WHERE id = ?').get(WORD_ID) as {
|
||||
frequency: number;
|
||||
} | null;
|
||||
return row?.frequency ?? 0;
|
||||
}
|
||||
|
||||
test('a karaoke burst collapses to one line and gives back its word counts', () => {
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 40, 40),
|
||||
{ session: 1, text: 'おはよう', startMs: 20_000, endMs: 22_000 },
|
||||
]);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 1);
|
||||
assert.equal(summary.removedLines, 39);
|
||||
assert.equal(summary.removedWordOccurrences, 39);
|
||||
assert.equal(countLines(db), 2);
|
||||
assert.equal(wordFrequency(db), 2);
|
||||
|
||||
// The surviving line covers the whole run, the way the parsed cue would.
|
||||
const kept = db
|
||||
.prepare(
|
||||
'SELECT segment_start_ms AS startMs, segment_end_ms AS endMs FROM imm_subtitle_lines WHERE line_id = 1',
|
||||
)
|
||||
.get() as { startMs: number; endMs: number };
|
||||
assert.equal(kept.startMs, 10_000);
|
||||
assert.equal(kept.endMs, 10_000 + 40 * 40);
|
||||
|
||||
assert.equal(summary.samples.length, 1);
|
||||
assert.equal(summary.samples[0]!.text, '飛び上がる');
|
||||
assert.equal(summary.samples[0]!.frames, 40);
|
||||
assert.equal(summary.samples[0]!.videoTitle, 'Ep 1');
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('ordinary repeated dialogue survives', () => {
|
||||
// Six contiguous `飛び上がる`, each held for a normal beat rather than a frame.
|
||||
const lines = Array.from({ length: 6 }, (_, index) => ({
|
||||
session: 1,
|
||||
text: '飛び上がる',
|
||||
startMs: 5_000 + index * 800,
|
||||
endMs: 5_000 + (index + 1) * 800,
|
||||
}));
|
||||
const { db, dbPath } = createDb(lines);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(summary.removedLines, 0);
|
||||
assert.equal(countLines(db), 6);
|
||||
assert.equal(wordFrequency(db), 6);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a long run of quarter-second frames is still a burst', () => {
|
||||
// Between the timing-only bound (0.1s) and the animation-frame bound (0.3s): heavier
|
||||
// typesetting lands here, and the run length is what makes it conclusive.
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 250));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 1);
|
||||
assert.equal(summary.removedLines, 5);
|
||||
assert.equal(countLines(db), 1);
|
||||
assert.equal(wordFrequency(db), 1);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a qualifying short-frame burst may end with one long hold frame', () => {
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 8, 40),
|
||||
{ session: 1, text: '飛び上がる', startMs: 10_320, endMs: 12_320 },
|
||||
]);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 1);
|
||||
assert.equal(summary.removedLines, 8);
|
||||
assert.equal(countLines(db), 1);
|
||||
assert.equal(wordFrequency(db), 1);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a long event before the final frame prevents burst cleanup', () => {
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 5, 40),
|
||||
{ session: 1, text: '飛び上がる', startMs: 10_200, endMs: 12_200 },
|
||||
{ session: 1, text: '飛び上がる', startMs: 12_200, endMs: 12_240 },
|
||||
]);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 7);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a run of frames longer than the animation bound survives', () => {
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 400));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 6);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('the four-frame residue the live gate stores is cleaned up', () => {
|
||||
// The streaming gate records the first four frames of a burst before the run is long
|
||||
// enough to recognise. Four contiguous identical events under the strict timing-only
|
||||
// bound are that residue, and no real dialogue.
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 4, 40));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 1);
|
||||
assert.equal(summary.removedLines, 3);
|
||||
assert.equal(countLines(db), 1);
|
||||
assert.equal(wordFrequency(db), 1);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a four-frame run above the strict frame bound survives', () => {
|
||||
// Long enough per event to be plausible dialogue; only a five-event run may use the
|
||||
// looser animation-frame bound.
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 4, 250));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 4);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('an explicit minRunLength raises the bar', () => {
|
||||
// Five quarter-second frames qualify under the defaults; a cautious run asking for six
|
||||
// leaves them alone. Above the strict bound, so the residue rule stays out of it.
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 5, 250));
|
||||
|
||||
try {
|
||||
const preview = cleanupDuplicateSubtitleLines(db, { dryRun: true });
|
||||
assert.equal(preview.burstGroups, 1);
|
||||
|
||||
const summary = cleanupDuplicateSubtitleLines(db, { minRunLength: 6 });
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 5);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('an explicit maxFrameSeconds tightens the frame bound', () => {
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 250));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db, { maxFrameSeconds: 0.2 });
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 6);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a non-finite maxFrameSeconds falls back to the default bound', () => {
|
||||
// Six normal-beat lines: Infinity must not turn every event into a "short frame".
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 6, 800));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db, { maxFrameSeconds: Infinity });
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 6);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('sampleLimit zero removes bursts but reports no samples', () => {
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 40, 40));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db, { sampleLimit: 0 });
|
||||
|
||||
assert.equal(summary.removedLines, 39);
|
||||
assert.deepEqual(summary.samples, []);
|
||||
assert.equal(countLines(db), 1);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a short run below every threshold survives', () => {
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 1_000, 3, 40));
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 0);
|
||||
assert.equal(countLines(db), 3);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('interleaved dual-line karaoke collapses each line to one row', () => {
|
||||
// Kanji and romaji lines frame-flipped together, the way fansub OPs are typeset. The
|
||||
// rows arrive interleaved in time order; each text must still chain into its own run.
|
||||
const kanji = karaokeFrames(1, '飛び上がる', 10_000, 20, 60);
|
||||
const romaji = karaokeFrames(1, 'tobiagaru', 10_001, 20, 60);
|
||||
const interleaved = [...kanji, ...romaji].sort((a, b) => a.startMs - b.startMs);
|
||||
const { db, dbPath } = createDb(interleaved);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 2);
|
||||
assert.equal(summary.removedLines, 38);
|
||||
assert.equal(countLines(db), 2);
|
||||
assert.equal(wordFrequency(db), 2);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('the same line in a rewatch session is never merged into the first watch', () => {
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40),
|
||||
...karaokeFrames(2, '飛び上がる', 10_000, 6, 40),
|
||||
]);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 2);
|
||||
assert.equal(summary.removedLines, 10);
|
||||
// One surviving line per session, not one across both.
|
||||
assert.equal(countLines(db), 2);
|
||||
assert.equal(wordFrequency(db), 2);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a gap between runs splits them', () => {
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40),
|
||||
...karaokeFrames(1, '飛び上がる', 60_000, 6, 40),
|
||||
]);
|
||||
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db);
|
||||
|
||||
assert.equal(summary.burstGroups, 2);
|
||||
assert.equal(countLines(db), 2);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('a dry run reports what an apply would do and writes nothing', () => {
|
||||
const { db, dbPath } = createDb(karaokeFrames(1, '飛び上がる', 10_000, 40, 40));
|
||||
|
||||
try {
|
||||
const preview = cleanupDuplicateSubtitleLines(db, { dryRun: true });
|
||||
|
||||
assert.equal(preview.dryRun, true);
|
||||
assert.equal(preview.removedLines, 39);
|
||||
assert.equal(countLines(db), 40);
|
||||
assert.equal(wordFrequency(db), 40);
|
||||
|
||||
const applied = cleanupDuplicateSubtitleLines(db);
|
||||
assert.equal(applied.removedLines, preview.removedLines);
|
||||
assert.equal(applied.removedWordOccurrences, preview.removedWordOccurrences);
|
||||
assert.equal(countLines(db), 1);
|
||||
} finally {
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
|
||||
test('the lookback window leaves older bursts alone', () => {
|
||||
const recentMs = BASE_MS;
|
||||
const oldMs = BASE_MS - 40 * DAY_MS;
|
||||
const { db, dbPath } = createDb([
|
||||
...karaokeFrames(1, '飛び上がる', 10_000, 6, 40).map((line) => ({
|
||||
...line,
|
||||
createdMs: oldMs,
|
||||
})),
|
||||
...karaokeFrames(2, '飛び上がる', 10_000, 6, 40).map((line) => ({
|
||||
...line,
|
||||
createdMs: recentMs,
|
||||
})),
|
||||
]);
|
||||
|
||||
globalThis.__subminerTestNowMs = BASE_MS;
|
||||
try {
|
||||
const summary = cleanupDuplicateSubtitleLines(db, { lookbackDays: 30 });
|
||||
|
||||
assert.equal(summary.lookbackDays, 30);
|
||||
assert.equal(summary.scannedLines, 6);
|
||||
assert.equal(summary.burstGroups, 1);
|
||||
assert.equal(summary.removedLines, 5);
|
||||
// Six untouched old frames plus the one surviving recent line.
|
||||
assert.equal(countLines(db), 7);
|
||||
assert.equal(wordFrequency(db), 7);
|
||||
} finally {
|
||||
globalThis.__subminerTestNowMs = undefined;
|
||||
db.close();
|
||||
cleanupDbPath(dbPath);
|
||||
}
|
||||
});
|
||||
@@ -0,0 +1,423 @@
|
||||
/*
|
||||
* Retroactive removal of animation-burst subtitle lines from the stats database.
|
||||
*
|
||||
* Before the live ingest gate existed, a karaoke OP recorded one line -- and one count
|
||||
* for every word in it -- per animation frame, which is enough to put an OP lyric at the
|
||||
* top of "Top Repeated Words" for good. This module finds those runs in what is already
|
||||
* stored and takes them back down to one line.
|
||||
*
|
||||
* Only timing is available here: the stored text has been stripped of ASS markup, so the
|
||||
* authoring evidence the file-level parser uses (`\t`, `\move`, karaoke timing, a
|
||||
* changing override signature) is long gone. What is left is a run of identical,
|
||||
* contiguous, short-lived lines inside a single session.
|
||||
*
|
||||
* The run has to be as long as the timing-only rule in `subtitle-cue-dedup` demands, but
|
||||
* its short frames may be as long as the animation-frame bound rather than the much
|
||||
* tighter timing-only one. A qualifying run may end with one longer hold, which is a
|
||||
* common karaoke shape. Five or more repeats of the same text, each ending where the next
|
||||
* begins, is already conclusive on its own -- no dialogue does that -- and the tighter
|
||||
* bound would walk straight past the heavier typesetting that motivated this, where
|
||||
* frames sit nearer a quarter of a second. Both bounds are options, so a cautious run can
|
||||
* ask for more, and a dry run always reports before anything is removed.
|
||||
*
|
||||
* Scope: subtitle lines, their word/kanji occurrences, and the `imm_words`/`imm_kanji`
|
||||
* aggregates those occurrences feed. Session telemetry (`lines_seen`, `tokens_seen`) and
|
||||
* the rollups derived from it are left alone; they are cumulative samples taken at record
|
||||
* time, and for sessions whose raw rows have since been pruned they cannot be recomputed.
|
||||
*/
|
||||
|
||||
import type { DatabaseSync } from './sqlite';
|
||||
import {
|
||||
ANIMATION_FRAME_MAX_SECONDS,
|
||||
DUPLICATE_CUE_GAP_TOLERANCE_SECONDS,
|
||||
MIN_STREAM_RESIDUE_FRAMES,
|
||||
MIN_TIMING_ONLY_FRAMES,
|
||||
TIMING_ONLY_FRAME_MAX_SECONDS,
|
||||
} from '../subtitle-burst-constants';
|
||||
import {
|
||||
applyLexicalRemovals,
|
||||
makePlaceholders,
|
||||
planLexicalRemovalsForLines,
|
||||
toDbTimestamp,
|
||||
} from './query-shared';
|
||||
import { nowMs } from './time';
|
||||
|
||||
const MS_PER_DAY = 86_400_000;
|
||||
/** SQLite caps bound parameters per statement; stay well under it. */
|
||||
const ID_BATCH_SIZE = 400;
|
||||
const DEFAULT_SAMPLE_LIMIT = 20;
|
||||
|
||||
export interface DuplicateSubtitleLineCleanupOptions {
|
||||
/** Only consider lines recorded within this many days. Null or omitted = all history. */
|
||||
lookbackDays?: number | null;
|
||||
/** Measure without writing. */
|
||||
dryRun?: boolean;
|
||||
/** Identical contiguous lines needed before a run counts as an animation. */
|
||||
minRunLength?: number;
|
||||
/** Longest a single event may last and still look like an animation frame. */
|
||||
maxFrameSeconds?: number;
|
||||
/** How many of the largest runs to describe in the summary. */
|
||||
sampleLimit?: number;
|
||||
}
|
||||
|
||||
export interface DuplicateSubtitleLineBurst {
|
||||
sessionId: number;
|
||||
videoId: number;
|
||||
text: string;
|
||||
/** Kept line, extended to cover the whole run. */
|
||||
keptLineId: number;
|
||||
removedLineIds: number[];
|
||||
startMs: number;
|
||||
endMs: number;
|
||||
}
|
||||
|
||||
export interface DuplicateSubtitleLineSample {
|
||||
videoId: number;
|
||||
videoTitle: string | null;
|
||||
text: string;
|
||||
frames: number;
|
||||
removedLines: number;
|
||||
startMs: number;
|
||||
endMs: number;
|
||||
}
|
||||
|
||||
export interface DuplicateSubtitleLineCleanupSummary {
|
||||
dryRun: boolean;
|
||||
lookbackDays: number | null;
|
||||
scannedLines: number;
|
||||
burstGroups: number;
|
||||
removedLines: number;
|
||||
removedWordOccurrences: number;
|
||||
removedKanjiOccurrences: number;
|
||||
samples: DuplicateSubtitleLineSample[];
|
||||
}
|
||||
|
||||
export interface StoredSubtitleLineRow {
|
||||
lineId: number;
|
||||
sessionId: number;
|
||||
videoId: number;
|
||||
text: string;
|
||||
startMs: number;
|
||||
endMs: number;
|
||||
}
|
||||
|
||||
interface ResolvedBounds {
|
||||
lookbackDays: number | null;
|
||||
minRunLength: number;
|
||||
maxFrameMs: number;
|
||||
/** Shorter runs qualify only when every event sits under this much stricter bound. */
|
||||
residueMinRunLength: number;
|
||||
strictFrameMs: number;
|
||||
gapToleranceMs: number;
|
||||
sampleLimit: number;
|
||||
}
|
||||
|
||||
function resolveBounds(options: DuplicateSubtitleLineCleanupOptions): ResolvedBounds {
|
||||
const lookbackDays =
|
||||
typeof options.lookbackDays === 'number' && Number.isFinite(options.lookbackDays)
|
||||
? Math.max(1, Math.floor(options.lookbackDays))
|
||||
: null;
|
||||
const minRunLength =
|
||||
typeof options.minRunLength === 'number' && Number.isFinite(options.minRunLength)
|
||||
? Math.max(2, Math.floor(options.minRunLength))
|
||||
: MIN_TIMING_ONLY_FRAMES;
|
||||
const maxFrameSeconds =
|
||||
typeof options.maxFrameSeconds === 'number' &&
|
||||
Number.isFinite(options.maxFrameSeconds) &&
|
||||
options.maxFrameSeconds > 0
|
||||
? options.maxFrameSeconds
|
||||
: ANIMATION_FRAME_MAX_SECONDS;
|
||||
const sampleLimit =
|
||||
typeof options.sampleLimit === 'number' && options.sampleLimit >= 0
|
||||
? Math.floor(options.sampleLimit)
|
||||
: DEFAULT_SAMPLE_LIMIT;
|
||||
return {
|
||||
lookbackDays,
|
||||
minRunLength,
|
||||
maxFrameMs: Math.round(maxFrameSeconds * 1000),
|
||||
residueMinRunLength: Math.max(MIN_STREAM_RESIDUE_FRAMES, minRunLength - 1),
|
||||
strictFrameMs: Math.round(TIMING_ONLY_FRAME_MAX_SECONDS * 1000),
|
||||
gapToleranceMs: Math.round(DUPLICATE_CUE_GAP_TOLERANCE_SECONDS * 1000),
|
||||
sampleLimit,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* `CREATED_DATE` holds epoch milliseconds on rows this app wrote, but older and synced
|
||||
* rows can carry seconds, so normalize before comparing against the cutoff.
|
||||
*/
|
||||
const CREATED_MS_SQL = `
|
||||
CASE
|
||||
WHEN sl.CREATED_DATE < 10000000000 THEN sl.CREATED_DATE * 1000
|
||||
ELSE sl.CREATED_DATE
|
||||
END`;
|
||||
|
||||
function readCandidateLines(db: DatabaseSync, bounds: ResolvedBounds): StoredSubtitleLineRow[] {
|
||||
const scope =
|
||||
bounds.lookbackDays === null
|
||||
? ''
|
||||
: `AND sl.CREATED_DATE IS NOT NULL AND ${CREATED_MS_SQL} >= ?`;
|
||||
const params = bounds.lookbackDays === null ? [] : [nowMs() - bounds.lookbackDays * MS_PER_DAY];
|
||||
return db
|
||||
.prepare(
|
||||
`SELECT
|
||||
sl.line_id AS lineId,
|
||||
sl.session_id AS sessionId,
|
||||
sl.video_id AS videoId,
|
||||
sl.text AS text,
|
||||
sl.segment_start_ms AS startMs,
|
||||
sl.segment_end_ms AS endMs
|
||||
FROM imm_subtitle_lines sl
|
||||
WHERE sl.segment_start_ms IS NOT NULL
|
||||
AND sl.segment_end_ms IS NOT NULL
|
||||
${scope}
|
||||
ORDER BY sl.session_id, sl.video_id, sl.segment_start_ms, sl.line_id`,
|
||||
)
|
||||
.all(...params) as StoredSubtitleLineRow[];
|
||||
}
|
||||
|
||||
function isBurst(run: StoredSubtitleLineRow[], bounds: ResolvedBounds): boolean {
|
||||
const isShortFrame = (row: StoredSubtitleLineRow): boolean =>
|
||||
row.endMs - row.startMs <= bounds.maxFrameMs;
|
||||
// The residue the live gate leaves behind: it records the first frames of a burst
|
||||
// before the run is long enough to recognise, so one frame fewer than the timing-only
|
||||
// minimum, every one under the strict timing-only bound. No dialogue holds identical
|
||||
// sub-tenth-second lines back to back that many times.
|
||||
if (
|
||||
run.length >= bounds.residueMinRunLength &&
|
||||
run.every((row) => row.endMs - row.startMs <= bounds.strictFrameMs)
|
||||
) {
|
||||
return true;
|
||||
}
|
||||
if (run.length < bounds.minRunLength) {
|
||||
return false;
|
||||
}
|
||||
if (run.every(isShortFrame)) {
|
||||
return true;
|
||||
}
|
||||
// Karaoke commonly finishes its short animation frames with one long hold. Only the
|
||||
// final event may exceed the frame bound, and the short frames before it must already
|
||||
// meet the minimum run length on their own.
|
||||
return (
|
||||
run.length - 1 >= bounds.minRunLength &&
|
||||
run.slice(0, -1).every(isShortFrame) &&
|
||||
!isShortFrame(run[run.length - 1]!)
|
||||
);
|
||||
}
|
||||
|
||||
function toBurst(run: StoredSubtitleLineRow[]): DuplicateSubtitleLineBurst {
|
||||
const [first] = run;
|
||||
return {
|
||||
sessionId: first!.sessionId,
|
||||
videoId: first!.videoId,
|
||||
text: first!.text,
|
||||
keptLineId: first!.lineId,
|
||||
removedLineIds: run.slice(1).map((row) => row.lineId),
|
||||
startMs: first!.startMs,
|
||||
endMs: run.reduce((latest, row) => Math.max(latest, row.endMs), first!.endMs),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Group stored lines into animation runs.
|
||||
*
|
||||
* Rows are bucketed per (session, video, text) before chaining, the way the file-level
|
||||
* dedup buckets cues: dual-line karaoke interleaves two texts frame by frame, and
|
||||
* chaining across the interleave would break every run at length one.
|
||||
*
|
||||
* Runs never cross a session, which is what keeps a rewatch intact: the same episode
|
||||
* watched twice stores the same line twice, and those two belong to different sessions.
|
||||
*/
|
||||
export function findDuplicateSubtitleLineBursts(
|
||||
rows: readonly StoredSubtitleLineRow[],
|
||||
options: DuplicateSubtitleLineCleanupOptions = {},
|
||||
): DuplicateSubtitleLineBurst[] {
|
||||
const bounds = resolveBounds(options);
|
||||
|
||||
// Insertion order preserves the query's startMs ordering within each bucket.
|
||||
const rowsByKey = new Map<string, StoredSubtitleLineRow[]>();
|
||||
for (const row of rows) {
|
||||
const key = `${row.sessionId}|${row.videoId}|${row.text}`;
|
||||
const bucket = rowsByKey.get(key);
|
||||
if (bucket) {
|
||||
bucket.push(row);
|
||||
} else {
|
||||
rowsByKey.set(key, [row]);
|
||||
}
|
||||
}
|
||||
|
||||
const bursts: DuplicateSubtitleLineBurst[] = [];
|
||||
for (const bucket of rowsByKey.values()) {
|
||||
if (bucket.length < 2) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let run: StoredSubtitleLineRow[] = [];
|
||||
let chainEndMs = 0;
|
||||
|
||||
const closeRun = (): void => {
|
||||
if (run.length > 1 && isBurst(run, bounds)) {
|
||||
bursts.push(toBurst(run));
|
||||
}
|
||||
run = [];
|
||||
};
|
||||
|
||||
for (const row of bucket) {
|
||||
if (run.length > 0 && row.startMs <= chainEndMs + bounds.gapToleranceMs) {
|
||||
run.push(row);
|
||||
chainEndMs = Math.max(chainEndMs, row.endMs);
|
||||
continue;
|
||||
}
|
||||
closeRun();
|
||||
run = [row];
|
||||
chainEndMs = row.endMs;
|
||||
}
|
||||
closeRun();
|
||||
}
|
||||
|
||||
return bursts;
|
||||
}
|
||||
|
||||
function chunk<T>(values: T[], size: number): T[][] {
|
||||
const chunks: T[][] = [];
|
||||
for (let i = 0; i < values.length; i += size) {
|
||||
chunks.push(values.slice(i, i + size));
|
||||
}
|
||||
return chunks;
|
||||
}
|
||||
|
||||
function buildSamples(
|
||||
db: DatabaseSync,
|
||||
bursts: DuplicateSubtitleLineBurst[],
|
||||
sampleLimit: number,
|
||||
): DuplicateSubtitleLineSample[] {
|
||||
if (sampleLimit === 0 || bursts.length === 0) {
|
||||
return [];
|
||||
}
|
||||
const largest = [...bursts]
|
||||
.sort((a, b) => b.removedLineIds.length - a.removedLineIds.length)
|
||||
.slice(0, sampleLimit);
|
||||
const videoIds = [...new Set(largest.map((burst) => burst.videoId))];
|
||||
const titles = new Map<number, string>();
|
||||
for (const batch of chunk(videoIds, ID_BATCH_SIZE)) {
|
||||
const rows = db
|
||||
.prepare(
|
||||
`SELECT video_id AS videoId, canonical_title AS title
|
||||
FROM imm_videos
|
||||
WHERE video_id IN (${makePlaceholders(batch)})`,
|
||||
)
|
||||
.all(...batch) as Array<{ videoId: number; title: string | null }>;
|
||||
for (const row of rows) {
|
||||
if (row.title) titles.set(row.videoId, row.title);
|
||||
}
|
||||
}
|
||||
|
||||
return largest.map((burst) => ({
|
||||
videoId: burst.videoId,
|
||||
videoTitle: titles.get(burst.videoId) ?? null,
|
||||
text: burst.text,
|
||||
frames: burst.removedLineIds.length + 1,
|
||||
removedLines: burst.removedLineIds.length,
|
||||
startMs: burst.startMs,
|
||||
endMs: burst.endMs,
|
||||
}));
|
||||
}
|
||||
|
||||
function sumRemovedOccurrences(
|
||||
db: DatabaseSync,
|
||||
table: 'imm_word_line_occurrences' | 'imm_kanji_line_occurrences',
|
||||
lineIds: number[],
|
||||
): number {
|
||||
let total = 0;
|
||||
for (const batch of chunk(lineIds, ID_BATCH_SIZE)) {
|
||||
const row = db
|
||||
.prepare(
|
||||
`SELECT COALESCE(SUM(occurrence_count), 0) AS total
|
||||
FROM ${table}
|
||||
WHERE line_id IN (${makePlaceholders(batch)})`,
|
||||
)
|
||||
.get(...batch) as { total: number } | null;
|
||||
total += row?.total ?? 0;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
function applyBursts(db: DatabaseSync, bursts: DuplicateSubtitleLineBurst[]): void {
|
||||
const removedLineIds = bursts.flatMap((burst) => burst.removedLineIds);
|
||||
const currentMs = toDbTimestamp(nowMs());
|
||||
|
||||
db.exec('BEGIN IMMEDIATE');
|
||||
try {
|
||||
for (const batch of chunk(removedLineIds, ID_BATCH_SIZE)) {
|
||||
const placeholders = makePlaceholders(batch);
|
||||
// Measured before the delete, applied after it: `applyLexicalRemovals` checks the
|
||||
// surviving occurrences to decide whether a zeroed count really means the word is
|
||||
// gone, so the rows it inspects have to be the post-delete ones.
|
||||
const plan = planLexicalRemovalsForLines(db, batch);
|
||||
db.prepare(`DELETE FROM imm_word_line_occurrences WHERE line_id IN (${placeholders})`).run(
|
||||
...batch,
|
||||
);
|
||||
db.prepare(`DELETE FROM imm_kanji_line_occurrences WHERE line_id IN (${placeholders})`).run(
|
||||
...batch,
|
||||
);
|
||||
db.prepare(`DELETE FROM imm_subtitle_lines WHERE line_id IN (${placeholders})`).run(...batch);
|
||||
applyLexicalRemovals(db, plan);
|
||||
}
|
||||
|
||||
const extendStmt = db.prepare(
|
||||
`UPDATE imm_subtitle_lines
|
||||
SET segment_end_ms = ?, LAST_UPDATE_DATE = ?
|
||||
WHERE line_id = ? AND (segment_end_ms IS NULL OR segment_end_ms < ?)`,
|
||||
);
|
||||
for (const burst of bursts) {
|
||||
extendStmt.run(burst.endMs, currentMs, burst.keptLineId, burst.endMs);
|
||||
}
|
||||
db.exec('COMMIT');
|
||||
} catch (error) {
|
||||
try {
|
||||
db.exec('ROLLBACK');
|
||||
} catch {
|
||||
// Surface the transaction failure, not the rollback's.
|
||||
}
|
||||
throw error;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Collapse stored animation bursts down to one line each.
|
||||
*
|
||||
* A dry run measures exactly what an apply would remove, using the same scan, so the
|
||||
* numbers shown in a confirmation prompt are the numbers that will happen.
|
||||
*/
|
||||
export function cleanupDuplicateSubtitleLines(
|
||||
db: DatabaseSync,
|
||||
options: DuplicateSubtitleLineCleanupOptions = {},
|
||||
): DuplicateSubtitleLineCleanupSummary {
|
||||
const bounds = resolveBounds(options);
|
||||
const dryRun = options.dryRun === true;
|
||||
const rows = readCandidateLines(db, bounds);
|
||||
const bursts = findDuplicateSubtitleLineBursts(rows, options);
|
||||
const removedLineIds = bursts.flatMap((burst) => burst.removedLineIds);
|
||||
|
||||
const summary: DuplicateSubtitleLineCleanupSummary = {
|
||||
dryRun,
|
||||
lookbackDays: bounds.lookbackDays,
|
||||
scannedLines: rows.length,
|
||||
burstGroups: bursts.length,
|
||||
removedLines: removedLineIds.length,
|
||||
removedWordOccurrences: sumRemovedOccurrences(db, 'imm_word_line_occurrences', removedLineIds),
|
||||
removedKanjiOccurrences: sumRemovedOccurrences(
|
||||
db,
|
||||
'imm_kanji_line_occurrences',
|
||||
removedLineIds,
|
||||
),
|
||||
samples: buildSamples(db, bursts, bounds.sampleLimit),
|
||||
};
|
||||
|
||||
if (dryRun || removedLineIds.length === 0) {
|
||||
return summary;
|
||||
}
|
||||
|
||||
applyBursts(db, bursts);
|
||||
return summary;
|
||||
}
|
||||
@@ -276,6 +276,19 @@ export function planLexicalRemovalsForSessions(
|
||||
return planLexicalRemovals(db, `sl.session_id IN (${makePlaceholders(sessionIds)})`, sessionIds);
|
||||
}
|
||||
|
||||
/**
|
||||
* Measure what deleting these individual subtitle lines removes from the vocabulary
|
||||
* tables. Used by the duplicate-line cleanup, which drops animation frames out of the
|
||||
* middle of sessions that otherwise stay intact.
|
||||
*/
|
||||
export function planLexicalRemovalsForLines(
|
||||
db: DatabaseSync,
|
||||
lineIds: number[],
|
||||
): LexicalRemovalPlan {
|
||||
if (lineIds.length === 0) return EMPTY_LEXICAL_REMOVAL_PLAN;
|
||||
return planLexicalRemovals(db, `sl.line_id IN (${makePlaceholders(lineIds)})`, lineIds);
|
||||
}
|
||||
|
||||
/** Measure what deleting these videos removes from the vocabulary tables. */
|
||||
export function planLexicalRemovalsForVideos(
|
||||
db: DatabaseSync,
|
||||
|
||||
Reference in New Issue
Block a user