# TranscribeBench pilot test set: openly licensed English speech with human reference transcripts.
#
# Nothing here identifies anyone: speakers are known only by the corpora's own anonymous IDs and,
# for EdAcc, the first languages they reported. No attempt is ever made to find out who they are.
#
# The audio is not committed. Re-create and verify it with:
#   .venv/bin/python data/transcribe/fetch_samples.py      (writes data/private/transcribe-audio/<id>.wav)
# Clips were chosen by the fixed rules in select_samples.py before any tool was run.
#
# Fields: id, condition (clean-read | hard-read | noisy | meeting | accented), source, license,
# license_url, credit (what the license asks us to show, and what we changed), url (the corpus),
# build (how the clip is cut or mixed: exact utterances or a time window), duration_s, sample_rate,
# sha256 (of the WAV file this script writes), reference (the committed transcript file),
# reference_words, reference_from, speakers, first_languages (as the speakers reported them, EdAcc only),
# snr_db (noisy only).
#
# Reference transcripts in references/ are shared under the license of the clip's source:
# CC BY 4.0 (LibriSpeech, AMI) or CC BY-SA 4.0 (EdAcc), with the credit above.
version: 1
samples:
- id: ls-clean-1
  condition: clean-read
  source: librispeech
  build:
    type: librispeech
    split: test-clean
    speaker: '2961'
    utterances:
    - 2961-960-0000
    - 2961-960-0001
    - 2961-960-0002
    - 2961-960-0003
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 68.1
  sample_rate: 16000
  sha256: 36d0d116dc43fa03955fdbd70f74f77015af0a9ddafd437281ad78fd1d8b001f
  reference: references/ls-clean-1.txt
  reference_words: 141
- id: ls-clean-2
  condition: clean-read
  source: librispeech
  build:
    type: librispeech
    split: test-clean
    speaker: '4446'
    utterances:
    - 4446-2271-0000
    - 4446-2271-0001
    - 4446-2271-0002
    - 4446-2271-0003
    - 4446-2271-0004
    - 4446-2271-0005
    - 4446-2271-0006
    - 4446-2271-0007
    - 4446-2271-0008
    - 4446-2271-0009
    - 4446-2271-0010
    - 4446-2271-0011
    - 4446-2271-0012
    - 4446-2271-0013
    - 4446-2271-0014
    - 4446-2271-0015
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 72.85
  sample_rate: 16000
  sha256: 38fb5cb970f2777123494fc03ef1533c0badf4893cac12ccd5e316a9f0ffda34
  reference: references/ls-clean-2.txt
  reference_words: 231
- id: ls-clean-3
  condition: clean-read
  source: librispeech
  build:
    type: librispeech
    split: test-clean
    speaker: '1221'
    utterances:
    - 1221-135766-0000
    - 1221-135766-0001
    - 1221-135766-0002
    - 1221-135766-0003
    - 1221-135766-0004
    - 1221-135766-0005
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 71.78
  sample_rate: 16000
  sha256: c67186d0c6bea5d26e5150197ac3b53ed9cb2523c7d084169dae4ce44276f1c9
  reference: references/ls-clean-3.txt
  reference_words: 194
- id: ls-clean-4
  condition: clean-read
  source: librispeech
  build:
    type: librispeech
    split: test-clean
    speaker: '4077'
    utterances:
    - 4077-13751-0000
    - 4077-13751-0001
    - 4077-13751-0002
    - 4077-13751-0003
    - 4077-13751-0004
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 61.45
  sample_rate: 16000
  sha256: de9dd674cf99a3bb1a91e9306cdf0ffae5d573baf94de140bb124b5177e62060
  reference: references/ls-clean-4.txt
  reference_words: 151
- id: ls-clean-5
  condition: clean-read
  source: librispeech
  build:
    type: librispeech
    split: test-clean
    speaker: '8230'
    utterances:
    - 8230-279154-0000
    - 8230-279154-0001
    - 8230-279154-0002
    - 8230-279154-0003
    - 8230-279154-0004
    - 8230-279154-0005
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 67.64
  sample_rate: 16000
  sha256: 088c887e36173d29188a6f7429fa1c6a9e01f8ee6f8fb71bf928c26164605cee
  reference: references/ls-clean-5.txt
  reference_words: 161
- id: ls-other-1
  condition: hard-read
  source: librispeech
  build:
    type: librispeech
    split: test-other
    speaker: '3080'
    utterances:
    - 3080-5032-0000
    - 3080-5032-0001
    - 3080-5032-0002
    - 3080-5032-0003
    - 3080-5032-0004
    - 3080-5032-0005
    - 3080-5032-0006
    - 3080-5032-0007
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 71.82
  sample_rate: 16000
  sha256: 8bf4c81b2e1cb1a2643a7510bbcdeafc2c5fd685baffd10acb04482c32330629
  reference: references/ls-other-1.txt
  reference_words: 190
- id: ls-other-2
  condition: hard-read
  source: librispeech
  build:
    type: librispeech
    split: test-other
    speaker: '3997'
    utterances:
    - 3997-180294-0000
    - 3997-180294-0001
    - 3997-180294-0002
    - 3997-180294-0003
    - 3997-180294-0004
    - 3997-180294-0005
    - 3997-180294-0006
    - 3997-180294-0007
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 66.69
  sample_rate: 16000
  sha256: d13f35c344d3cc7c6383ca78257958864e89ef7ffe06359df28c7bd6f3dd7035
  reference: references/ls-other-2.txt
  reference_words: 178
- id: ls-other-3
  condition: hard-read
  source: librispeech
  build:
    type: librispeech
    split: test-other
    speaker: '3005'
    utterances:
    - 3005-163389-0000
    - 3005-163389-0001
    - 3005-163389-0002
    - 3005-163389-0003
    - 3005-163389-0004
    - 3005-163389-0005
    - 3005-163389-0006
    - 3005-163389-0007
    - 3005-163389-0008
    - 3005-163389-0009
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 65.98
  sample_rate: 16000
  sha256: 7072446fbde25beebcbec5841f26bd53149ba7a551ac5585b98565a41374c458
  reference: references/ls-other-3.txt
  reference_words: 167
- id: ls-other-4
  condition: hard-read
  source: librispeech
  build:
    type: librispeech
    split: test-other
    speaker: '5764'
    utterances:
    - 5764-299665-0000
    - 5764-299665-0001
    - 5764-299665-0002
    - 5764-299665-0003
    - 5764-299665-0004
    - 5764-299665-0005
    - 5764-299665-0006
    - 5764-299665-0007
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0 (Panayotov, Chen, Povey and Khudanpur, ICASSP 2015),
    from LibriVox public-domain audiobooks. Changed: consecutive utterances joined into one clip.'
  url: https://www.openslr.org/12
  reference_from: The corpus's own transcript (trans.txt), unchanged except that the utterances are joined with spaces.
  duration_s: 68.84
  sample_rate: 16000
  sha256: 5d459c6509677eec5b9361c39a20151597038ca6a3c360e208bb8c2f79a67f81
  reference: references/ls-other-4.txt
  reference_words: 141
- id: noisy-1
  condition: noisy
  source: librispeech+noise
  build:
    type: mix
    speech: ls-clean-1
    noise: restaurant-ambience
    snr_db: 5
    noise_offset_s: 33.351
  license: CC BY 4.0 (speech); public domain (noise)
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'Speech: LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0. Noise: "Restaurant ambience" by stephan,
    public domain (pdsounds.org, via Wikimedia Commons: https://commons.wikimedia.org/wiki/File:Restaurant_ambience.ogg).
    Changed: utterances joined and mixed with the noise at the stated SNR.'
  url: https://www.openslr.org/12
  reference_from: Same as the clean clip it is made from.
  duration_s: 68.1
  sample_rate: 16000
  sha256: 9acc4dbe9ecdb6c18425f2e79a5f20dd98a0c5e4fcb85c62691ceba33db34f53
  reference: references/noisy-1.txt
  reference_words: 141
  snr_db: 5
- id: noisy-2
  condition: noisy
  source: librispeech+noise
  build:
    type: mix
    speech: ls-clean-2
    noise: restaurant-ambience
    snr_db: 5
    noise_offset_s: 45.105
  license: CC BY 4.0 (speech); public domain (noise)
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'Speech: LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0. Noise: "Restaurant ambience" by stephan,
    public domain (pdsounds.org, via Wikimedia Commons: https://commons.wikimedia.org/wiki/File:Restaurant_ambience.ogg).
    Changed: utterances joined and mixed with the noise at the stated SNR.'
  url: https://www.openslr.org/12
  reference_from: Same as the clean clip it is made from.
  duration_s: 72.85
  sample_rate: 16000
  sha256: 4954d85993782cb4661cf03ece848e60240093044334bf468ce0cbc5be034784
  reference: references/noisy-2.txt
  reference_words: 231
  snr_db: 5
- id: noisy-3
  condition: noisy
  source: librispeech+noise
  build:
    type: mix
    speech: ls-clean-3
    noise: restaurant-ambience
    snr_db: 5
    noise_offset_s: 49.291
  license: CC BY 4.0 (speech); public domain (noise)
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'Speech: LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0. Noise: "Restaurant ambience" by stephan,
    public domain (pdsounds.org, via Wikimedia Commons: https://commons.wikimedia.org/wiki/File:Restaurant_ambience.ogg).
    Changed: utterances joined and mixed with the noise at the stated SNR.'
  url: https://www.openslr.org/12
  reference_from: Same as the clean clip it is made from.
  duration_s: 71.78
  sample_rate: 16000
  sha256: 41bb872e86be5f174687b6125a608e5835ecb4e50b3d6f9d24456de92d816632
  reference: references/noisy-3.txt
  reference_words: 194
  snr_db: 5
- id: noisy-4
  condition: noisy
  source: librispeech+noise
  build:
    type: mix
    speech: ls-clean-4
    noise: restaurant-ambience
    snr_db: 5
    noise_offset_s: 28.477
  license: CC BY 4.0 (speech); public domain (noise)
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'Speech: LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0. Noise: "Restaurant ambience" by stephan,
    public domain (pdsounds.org, via Wikimedia Commons: https://commons.wikimedia.org/wiki/File:Restaurant_ambience.ogg).
    Changed: utterances joined and mixed with the noise at the stated SNR.'
  url: https://www.openslr.org/12
  reference_from: Same as the clean clip it is made from.
  duration_s: 61.45
  sample_rate: 16000
  sha256: 4948753a660d5eb84d596bfbadd89f4dd75d7f072f747cfe017a9e97d49dff70
  reference: references/noisy-4.txt
  reference_words: 151
  snr_db: 5
- id: noisy-5
  condition: noisy
  source: librispeech+noise
  build:
    type: mix
    speech: ls-clean-5
    noise: restaurant-ambience
    snr_db: 5
    noise_offset_s: 59.653
  license: CC BY 4.0 (speech); public domain (noise)
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'Speech: LibriSpeech ASR corpus, (c) 2014 Vassil Panayotov, CC BY 4.0. Noise: "Restaurant ambience" by stephan,
    public domain (pdsounds.org, via Wikimedia Commons: https://commons.wikimedia.org/wiki/File:Restaurant_ambience.ogg).
    Changed: utterances joined and mixed with the noise at the stated SNR.'
  url: https://www.openslr.org/12
  reference_from: Same as the clean clip it is made from.
  duration_s: 67.64
  sample_rate: 16000
  sha256: 80f1b570671e454b0051289ab4eb4e5e784a7977c8a91d902abd77489b255fb0
  reference: references/noisy-5.txt
  reference_words: 161
  snr_db: 5
- id: ami-1
  condition: meeting
  source: ami
  build:
    type: ami
    meeting: ES2004a
    start_s: 305.708
    end_s: 427.092
  speakers: 4
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'AMI Meeting Corpus (headset mix and manual annotations v1.6.2), CC BY 4.0, https://groups.inf.ed.ac.uk/ami/corpus/license.shtml
    (Carletta et al., 2005). Changed: a window cut from the meeting.'
  url: https://groups.inf.ed.ac.uk/ami/corpus/
  reference_from: 'AMI manual word annotations v1.6.2: every segment wholly inside the window, ordered by start time, words
    in order; truncated words, vocal sounds and non-speech marks left out.'
  duration_s: 121.38
  sample_rate: 16000
  sha256: 94625497e8aefbe2939546283c6c3c2b3d9f878f3a72f7cb6da0cfa3317cb232
  reference: references/ami-1.txt
  reference_words: 290
- id: ami-2
  condition: meeting
  source: ami
  build:
    type: ami
    meeting: IS1009a
    start_s: 307.02
    end_s: 428.116
  speakers: 4
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'AMI Meeting Corpus (headset mix and manual annotations v1.6.2), CC BY 4.0, https://groups.inf.ed.ac.uk/ami/corpus/license.shtml
    (Carletta et al., 2005). Changed: a window cut from the meeting.'
  url: https://groups.inf.ed.ac.uk/ami/corpus/
  reference_from: 'AMI manual word annotations v1.6.2: every segment wholly inside the window, ordered by start time, words
    in order; truncated words, vocal sounds and non-speech marks left out.'
  duration_s: 121.1
  sample_rate: 16000
  sha256: d48568b99412e67266d5a17204f1b70402bff2028fd790b41b49386f11cc92b2
  reference: references/ami-2.txt
  reference_words: 220
- id: ami-3
  condition: meeting
  source: ami
  build:
    type: ami
    meeting: TS3003a
    start_s: 329.262
    end_s: 450.116
  speakers: 3
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'AMI Meeting Corpus (headset mix and manual annotations v1.6.2), CC BY 4.0, https://groups.inf.ed.ac.uk/ami/corpus/license.shtml
    (Carletta et al., 2005). Changed: a window cut from the meeting.'
  url: https://groups.inf.ed.ac.uk/ami/corpus/
  reference_from: 'AMI manual word annotations v1.6.2: every segment wholly inside the window, ordered by start time, words
    in order; truncated words, vocal sounds and non-speech marks left out.'
  duration_s: 120.85
  sample_rate: 16000
  sha256: 1d195406d68893d06e09fb363577866b7d143df8465de6ffe6569c049b3d7b24
  reference: references/ami-3.txt
  reference_words: 287
- id: ami-4
  condition: meeting
  source: ami
  build:
    type: ami
    meeting: EN2002a
    start_s: 320.86
    end_s: 494.26
  speakers: 4
  license: CC BY 4.0
  license_url: https://creativecommons.org/licenses/by/4.0/
  credit: 'AMI Meeting Corpus (headset mix and manual annotations v1.6.2), CC BY 4.0, https://groups.inf.ed.ac.uk/ami/corpus/license.shtml
    (Carletta et al., 2005). Changed: a window cut from the meeting.'
  url: https://groups.inf.ed.ac.uk/ami/corpus/
  reference_from: 'AMI manual word annotations v1.6.2: every segment wholly inside the window, ordered by start time, words
    in order; truncated words, vocal sounds and non-speech marks left out.'
  duration_s: 173.4
  sample_rate: 16000
  sha256: a0a28adde985c155cbc3deff4e9e0e8949d15245b0edc82f9ba9c4bfe2c64f11
  reference: references/ami-4.txt
  reference_words: 700
- id: edacc-1
  condition: accented
  source: edacc
  build:
    type: edacc
    conversation: EDACC-C02
    start_s: 121.73
    end_s: 218.72
  speakers: 2
  first_languages:
  - Bulgarian
  - Lithuanian
  license: CC BY-SA 4.0
  license_url: https://creativecommons.org/licenses/by-sa/4.0/
  credit: 'The Edinburgh International Accents of English Corpus (EdAcc), Sanabria, Markl, Carmantini, Klejch, Bell and Bogoychev,
    University of Edinburgh, 2025, doi:10.7488/ds/7914, CC BY-SA 4.0. Changed: a window cut from the conversation. The reference
    transcript is shared under CC BY-SA 4.0.'
  url: https://doi.org/10.7488/ds/7914
  reference_from: 'EdAcc''s own test-set transcript (text, segments): every segment wholly inside the window, ordered by start
    time. Special tokens such as <overlap> and <laugh> are kept in the file and removed by the normalization.'
  duration_s: 96.99
  sample_rate: 32000
  sha256: 8b7f79aec7d5ce07ac13467170ac2e647573b6a9d509c0f17d186f6c812492a8
  reference: references/edacc-1.txt
  reference_words: 196
- id: edacc-2
  condition: accented
  source: edacc
  build:
    type: edacc
    conversation: EDACC-C05_P0
    start_s: 119.835
    end_s: 217.31
  speakers: 2
  first_languages:
  - English
  - Hindi
  - Konkani
  - Marathi
  - Romanian
  license: CC BY-SA 4.0
  license_url: https://creativecommons.org/licenses/by-sa/4.0/
  credit: 'The Edinburgh International Accents of English Corpus (EdAcc), Sanabria, Markl, Carmantini, Klejch, Bell and Bogoychev,
    University of Edinburgh, 2025, doi:10.7488/ds/7914, CC BY-SA 4.0. Changed: a window cut from the conversation. The reference
    transcript is shared under CC BY-SA 4.0.'
  url: https://doi.org/10.7488/ds/7914
  reference_from: 'EdAcc''s own test-set transcript (text, segments): every segment wholly inside the window, ordered by start
    time. Special tokens such as <overlap> and <laugh> are kept in the file and removed by the normalization.'
  duration_s: 97.47
  sample_rate: 32000
  sha256: bbfb45abaaab208828200a5df11c31ac4183ce6c6bb2269898066e2e811a271d
  reference: references/edacc-2.txt
  reference_words: 328
- id: edacc-3
  condition: accented
  source: edacc
  build:
    type: edacc
    conversation: EDACC-C10
    start_s: 127.175
    end_s: 220.39
  speakers: 2
  first_languages:
  - Sinhalese
  license: CC BY-SA 4.0
  license_url: https://creativecommons.org/licenses/by-sa/4.0/
  credit: 'The Edinburgh International Accents of English Corpus (EdAcc), Sanabria, Markl, Carmantini, Klejch, Bell and Bogoychev,
    University of Edinburgh, 2025, doi:10.7488/ds/7914, CC BY-SA 4.0. Changed: a window cut from the conversation. The reference
    transcript is shared under CC BY-SA 4.0.'
  url: https://doi.org/10.7488/ds/7914
  reference_from: 'EdAcc''s own test-set transcript (text, segments): every segment wholly inside the window, ordered by start
    time. Special tokens such as <overlap> and <laugh> are kept in the file and removed by the normalization.'
  duration_s: 93.22
  sample_rate: 32000
  sha256: 51982fe36f1ed081a978716f6b2e447f115514277a52c436aaa305d70d579104
  reference: references/edacc-3.txt
  reference_words: 182
- id: edacc-4
  condition: accented
  source: edacc
  build:
    type: edacc
    conversation: EDACC-C19
    start_s: 165.93
    end_s: 288.35
  speakers: 2
  first_languages:
  - English
  - Mandarin
  - Shona
  license: CC BY-SA 4.0
  license_url: https://creativecommons.org/licenses/by-sa/4.0/
  credit: 'The Edinburgh International Accents of English Corpus (EdAcc), Sanabria, Markl, Carmantini, Klejch, Bell and Bogoychev,
    University of Edinburgh, 2025, doi:10.7488/ds/7914, CC BY-SA 4.0. Changed: a window cut from the conversation. The reference
    transcript is shared under CC BY-SA 4.0.'
  url: https://doi.org/10.7488/ds/7914
  reference_from: 'EdAcc''s own test-set transcript (text, segments): every segment wholly inside the window, ordered by start
    time. Special tokens such as <overlap> and <laugh> are kept in the file and removed by the normalization.'
  duration_s: 122.42
  sample_rate: 32000
  sha256: 7f4168d9fa946ef2deb9d3e512904de29281d594810d3c163fc0961c5ea94363
  reference: references/edacc-4.txt
  reference_words: 323
- id: edacc-5
  condition: accented
  source: edacc
  build:
    type: edacc
    conversation: EDACC-C20
    start_s: 120.465
    end_s: 231.02
  speakers: 2
  first_languages:
  - Catalan
  - Spanish
  license: CC BY-SA 4.0
  license_url: https://creativecommons.org/licenses/by-sa/4.0/
  credit: 'The Edinburgh International Accents of English Corpus (EdAcc), Sanabria, Markl, Carmantini, Klejch, Bell and Bogoychev,
    University of Edinburgh, 2025, doi:10.7488/ds/7914, CC BY-SA 4.0. Changed: a window cut from the conversation. The reference
    transcript is shared under CC BY-SA 4.0.'
  url: https://doi.org/10.7488/ds/7914
  reference_from: 'EdAcc''s own test-set transcript (text, segments): every segment wholly inside the window, ordered by start
    time. Special tokens such as <overlap> and <laugh> are kept in the file and removed by the normalization.'
  duration_s: 110.56
  sample_rate: 32000
  sha256: cc4d58cf88aa78dbdd4ad86da2013b2dfd2bd30d7507a0e8050c51af022ad53a
  reference: references/edacc-5.txt
  reference_words: 284
total_duration_s: 2014.36
clips: 23
