{"dcterms:modified":"2024-01-18","dcterms:creator":"heiDATA","@type":"ore:ResourceMap","schema:additionalType":"Dataverse OREMap Format v1.0.0","dvcore:generatedBy":{"@type":"schema:SoftwareApplication","schema:name":"Dataverse","schema:version":"6.1 build 1590-f5d1299","schema:url":"https://github.com/iqss/dataverse"},"@id":"https://heidata.uni-heidelberg.de/api/datasets/export?exporter=OAI_ORE&persistentId=https://doi.org/10.11588/data/TMEDTX","ore:describes":{"contributor":{"citation:contributorType":"Research Group","citation:contributorName":"Statistical NLP Group, Department of Computational Linguistics"},"author":[{"citation:authorName":"Beilharz, Benjamin","citation:authorAffiliation":"Department of Computational Linguistics, Heidelberg University, Heidelberg, Germany"},{"citation:authorName":"Sun, Xin","citation:authorAffiliation":"Department of Computational Linguistics, Heidelberg University, Heidelberg, Germany"}],"citation:dsDescription":{"citation:dsDescriptionValue":"This dataset is a corpus of sentence-aligned triples of German audio, German text, and English translation, based on German audio books. The corpus consists of over 100 hours of audio material and over 50k parallel sentences. The speech data are low in disfluencies because of the audio book setup. The quality of audio and sentence alignments has been checked by a manual evaluation, showing that that speech alignment is in general very high. The sentence alignment quality is comparable to well-used parallel translation data and can be adjusted by cutoffs on the automatic alignment score. To our knowledge, this corpus is to date the largest resource for end-to-end speech translation for German.","citation:dsDescriptionDate":"2019-10-17"},"citation:datasetContact":{"citation:datasetContactName":"Riezler, Stefan","citation:datasetContactAffiliation":"Department of Computational Linguistics, Heidelberg University, Heidelberg, Germany","citation:datasetContactEmail":"riezler@cl.uni-heidelberg.de"},"grantNumber":{"citation:grantNumberAgency":"German research foundation (DFG)","citation:grantNumberValue":"RI2221/4-1"},"citation:topicClassification":{"citation:topicClassValue":"Speech Translation"},"citation:keyword":{"citation:keywordValue":"Speech Translation Corpus German-English"},"citation:distributor":{"citation:distributorName":"Statistical NLP Group, Department of Computational Linguistics","citation:distributorAffiliation":"Heidelberg University","citation:distributorAbbreviation":"statNLPgroup","citation:distributorURL":"https://www.cl.uni-heidelberg.de/statnlpgroup/","citation:distributorLogoURL":"https://www.cl.uni-heidelberg.de/statnlpgroup/images/statnlp_logo.png"},"citation:producer":{"citation:producerName":"Statistical NLP Group, Department of Computational Linguistics","citation:producerAffiliation":"Heidelberg University","citation:producerAbbreviation":"statNLPgroup","citation:producerURL":"https://www.cl.uni-heidelberg.de/statnlpgroup/","citation:producerLogoURL":"https://www.cl.uni-heidelberg.de/statnlpgroup/images/statnlp_logo.png"},"publication":{"publicationCitation":"Beilharz et al. 2019, LibriVoxDeEn - A Corpus for German-to-English Speech Translation and Speech Recognition (arXiv preprint - arXiv:1910.07924)","publicationIDType":"arXiv","publicationIDNumber":"1910.07924","publicationURL":"https://arxiv.org/abs/1910.07924"},"citation:productionDate":"2019-10-17","language":["English","German"],"subject":"Computer and Information Science","citation:productionPlace":"Heidelberg","title":"LibriVoxDeEn - A Corpus for German-to-English Speech Translation and Speech Recognition","@id":"https://doi.org/10.11588/data/TMEDTX","@type":["ore:Aggregation","schema:Dataset"],"schema:version":"2.0","schema:name":"LibriVoxDeEn - A Corpus for German-to-English Speech Translation and Speech Recognition","schema:dateModified":"Sat Jun 13 09:01:19 CEST 2020","schema:datePublished":"2019-10-21","schema:creativeWorkStatus":"RELEASED","dvcore:termsOfUse":"Licensed under Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International (CC BY-NC-SA 4.0) \r\n","dvcore:fileTermsOfAccess":{"dvcore:fileRequestAccess":false},"schema:includedInDataCatalog":"heiDATA","schema:isPartOf":{"schema:name":"Statistical Natural Language Processing Group","@id":"https://heidata.uni-heidelberg.de/dataverse/statnlpgroup","schema:description":"The Statistical Natural Language Processing Group is part of the Department of Computational Linguistics.\r\n
\r\nOur research addresses various aspects of the problem of the confusion of languages, by means of statistical learning techniques.\r\n
\r\nResearch topics include the following:\r\n