@inproceedings{c621e4a808cb4a538e7a5006668e2cc3,
title = "On-the-fly topic adaptation for YouTube video transcription",
abstract = "Automatic closed-captioning of video is a useful application of speech recognition technology but poses numerous challenges when applied to open-domain user-uploaded videos such as those on YouTube. In this work, we explore a strategy to improve decoding accuracy for video transcription by decoding each video with a language model (LM) adapted specifically to the topics that the video covers. Taxonomic topic classifiers are used to determine the topic content of videos and to build a large set of topic-specific LMs from web documents. We consider strategies for selecting and interpolating LMs in both supervised and unsupervised scenarios in a two-pass lattice rescoring framework. Experiments on a YouTube video corpus show a 3.6 absolute reduction in WER over generic single-pass transcriptions as well as a statistically significant 0.8 absolute improvement over rescoring with a very large non-adapted LM built from all the documents.",
author = "Kapil Thadani and Fadi Biadsy and Bikel, \{Daniel M.\}",
year = "2012",
month = dec,
day = "1",
language = "English",
isbn = "9781622767595",
series = "13th Annual Conference of the International Speech Communication Association 2012, INTERSPEECH 2012",
pages = "210--213",
booktitle = "13th Annual Conference of the International Speech Communication Association 2012, INTERSPEECH 2012",
note = "13th Annual Conference of the International Speech Communication Association 2012, INTERSPEECH 2012 ; Conference date: 09-09-2012 Through 13-09-2012",
}