GOESB

librispeech-en-streaming

v1.0.0

openCC-BY-4.0scores against whisper-medium-en-streaming

Audio

15 utterances · 145.195s total · 16000 Hz

Integrity

sha256: f7df4ba02fc3ac70ebb606143b03bf74ffaf952b102d9b58e8870a1d0b0610d9

Full manifest

{
  "id": "librispeech-en-streaming",
  "version": "1.0.0",
  "sha256": "f7df4ba02fc3ac70ebb606143b03bf74ffaf952b102d9b58e8870a1d0b0610d9",
  "profile_id": "whisper-medium-en-streaming",
  "visibility": "open",
  "license": "CC-BY-4.0",
  "audio": {
    "count": 15,
    "total_duration_s": 145.195,
    "sample_rate_hz": 16000,
    "manifest_sha256": "6b9cdcf81a565496b7d4ef4881b035d98593eb941e61d87780591f0aba5852d9",
    "source": {
      "type": "librispeech",
      "params": {
        "speaker": "1272",
        "chapter": "128104",
        "split": "dev-clean"
      },
      "fetch_instructions": "Identical audio to librispeech-en-batch — auto-fetched the same way (same speaker/chapter). Manual fallback: run scripts/fetch_librispeech_subset.py for that pack, then pass `--audio-dir packs/librispeech-en-batch/audio` to `goesb run`, or symlink/copy that pack's audio/ directory here."
    }
  },
  "metadata": {
    "language": "en-US",
    "dialect": "general-american",
    "age_group": "mixed",
    "recording_environment": "quiet",
    "microphone": "consumer usb",
    "sample_rate_hz": 16000,
    "background_noise": "none",
    "num_speakers": 100,
    "speech_style": "read",
    "transcription_style": "verbatim",
    "tags": [
      "librispeech",
      "english",
      "clean",
      "streaming"
    ]
  }
}