@inproceedings{4a41b81cc76f4277a0b78a998533d664,
title = "Scikit-talk: A toolkit for processing real-world conversational speech data",
abstract = "We present Scikit-talk, an open-source toolkit for processing collections of real-world conversational speech in Python. First of its kind, the toolkit equips those interested in studying or modeling conversations with an easyto-use interface to build and explore large collections of transcriptions and annotations of talk-in-interaction. Designed for applications in speech processing and Conversational AI, Scikit-talk provides tools to custombuild datasets for tasks such as intent prototyping, dialog flow testing, and conversation design. Its preprocessor module comes with several pre-built interfaces for common transcription formats, which aim to make working across multiple data sources more accessible. The explorer module provides a collection of tools to explore and analyse this data type via string matching and unsupervised machine learning techniques. Scikit-talk serves as a platform to collect and connect different transcription formats and representations of talk, enabling the user to quickly build multilingual datasets of varying detail and granularity. Thus, the toolkit aims to make working with authentic conversational speech data in Python more accessible and to provide the user with comprehensive options to work with representations of talk in appropriate detail for any downstream task. For the latest updates and information on currently supported languages and language resources, please refer to: https://pypi.org/project/scikit-talk/.",
author = "Andreas Liesenfeld and G{\'a}bor Parti and Huang, {Chu Ren}",
note = "Publisher Copyright: {\textcopyright}2021 Association for Computational Linguistics.; 22nd Annual Meeting of the Special Interest Group on Discourse and Dialogue, SIGDIAL 2021 ; Conference date: 29-07-2021 Through 31-07-2021",
year = "2021",
month = jul,
doi = "10.18653/v1/2021.sigdial-1.26",
language = "English",
series = "SIGDIAL 2021 - 22nd Annual Meeting of the Special Interest Group on Discourse and Dialogue, Proceedings of the Conference",
publisher = "Association for Computational Linguistics (ACL)",
pages = "252--256",
editor = "Haizhou Li and Gina-Anne Levow and Zhou Yu and Chitralekha Gupta and Berrak Sisman and Siqi Cai and David Vandyke and Nina Dethlefs and Yan Wu and Li, {Junyi Jessy}",
booktitle = "SIGDIAL 2021 - 22nd Annual Meeting of the Special Interest Group on Discourse and Dialogue, Proceedings of the Conference",
address = "United States",
}