Recent advances in Automatic Speech Recognition (ASR) have made it possible to reliably produce automatic transcripts of clinician-patient conversations. However, access to clinical datasets is heavily restricted due to patient privacy, thus slowing down normal research practices. We detail the development of a public access, high quality dataset comprising of57 mocked primary care consultations, including audio recordings, their manual utterance-level transcriptions, and the associated consultation notes. Our work illustrates how the dataset can be used as a benchmark for conversational medical ASR as well as consultation note generation from transcripts.
@article{arxiv.2204.00333,
title = {PriMock57: A Dataset Of Primary Care Mock Consultations},
author = {Alex Papadopoulos Korfiatis and Francesco Moramarco and Radmila Sarac and Aleksandar Savkov},
journal= {arXiv preprint arXiv:2204.00333},
year = {2022}
}