{
  "@context": "https://schema.org",
  "@type": "Dataset",
  "@id": "https://openspeaker.ai/data/long-form-tts-speed-benchmark.json#dataset",
  "identifier": "openspeaker-long-form-tts-production-snapshot-2026-08-02",
  "name": "OpenSpeaker long-form TTS RTF production benchmark — 2 August 2026",
  "description": "Privacy-safe production evidence from one 153,072-character single-speaker job, two 100,000+ character dialogue jobs and a 45-job long-form single-speaker cohort, with transcript generation disabled and final MP3 duration measured for RTF.",
  "url": "https://openspeaker.ai/long-form-tts-speed-benchmark",
  "distribution": {
    "@type": "DataDownload",
    "encodingFormat": "application/json",
    "contentUrl": "https://openspeaker.ai/data/long-form-tts-speed-benchmark.json"
  },
  "creator": {
    "@type": "Organization",
    "name": "OpenSpeaker",
    "legalName": "Vivoo Global CO., LTD.",
    "url": "https://openspeaker.ai"
  },
  "datePublished": "2026-08-02",
  "dateModified": "2026-08-02",
  "measurement": {
    "source": "Production BullMQ completed-job state joined to completed tasks; transcript flags were checked in both job data and task metadata, and final MP3 duration was measured from tasks.output_uri with ffprobe without publishing the URL",
    "queue": "tts-queue",
    "snapshotAtUtc": "2026-08-02T11:41:51.625Z",
    "timeZone": "Asia/Ho_Chi_Minh",
    "retentionScope": "Completed jobs retained by each measured production queue at the dated snapshot; not all historical traffic",
    "percentileMethod": "nearest-rank",
    "formulas": {
      "activeProcessingSeconds": "finishedOn - processedOn",
      "processingRtf": "activeProcessingSeconds / finalOutputAudioDurationSeconds",
      "realTimeSpeedup": "finalOutputAudioDurationSeconds / activeProcessingSeconds"
    },
    "audioDurationMethod": "ffprobe of the final completed MP3 referenced by tasks.output_uri; output URLs are not included in this dataset",
    "activeProcessingScope": "Production worker time through orchestration, synthesis, merge and post-processing, upload and durable task completion; not isolated model inference",
    "privacy": "Aggregate values only. Input text, prompts, user IDs, task IDs, output URLs and credentials are excluded.",
    "limitations": [
      "Completed jobs only; this is not a success-rate study.",
      "First-party production snapshot, not an independent laboratory benchmark.",
      "No same-script competitor comparison was performed.",
      "The 100K+ evidence contains one single-speaker narration and two multi-speaker dialogue jobs; it is not all historical traffic.",
      "The broader earlier tts-queue snapshot was not stratified by transcript option and is published as workload context.",
      "Active processing uses one worker clock. Recorded end-to-end can inherit sub-second cross-host clock skew, so queue-wait deltas are not used as a headline claim.",
      "RTF measures production speed, not perceptual voice quality or MOS."
    ]
  },
  "singleSpeakerValidation": {
    "snapshotAtUtc": "2026-08-02T11:41:23.195Z",
    "queue": "tts-queue",
    "jobName": "tts-job",
    "provider": "elevenlabs",
    "providerLabel": "ElevenLabs voice source",
    "greaterThanInputCharacters": 100000,
    "completedJobs": 1,
    "transcriptGeneration": false,
    "transcriptEvidencePaths": [
      "job.data.with_transcript",
      "tasks.metadata.data.with_transcript"
    ],
    "inputCharacters": 153072,
    "activeProcessingSeconds": 158.396,
    "recordedEndToEndSeconds": 158.296,
    "outputAudioDurationSeconds": 8746.475,
    "outputAudioDurationMinutes": 145.775,
    "outputAudioDurationLabel": "2h 25m 46.475s",
    "processingRtf": 0.0181,
    "realTimeSpeedup": 55.22,
    "outputAudioFormat": {
      "container": "MP3",
      "sampleRateHz": 44100,
      "channels": 2,
      "bitrateKbps": 128
    },
    "audioDurationEvidence": "ffprobe of the final tasks.output_uri MP3; the URL is not published",
    "clockNote": "Recorded end-to-end is 0.100 seconds below active processing because producer and worker clocks differ slightly. Active processing uses one worker clock and is the headline timing.",
    "claimScope": "One retained completed single-speaker production job over 100,000 characters, not a universal guarantee or competitor benchmark."
  },
  "singleSpeakerLongFormCohort": {
    "snapshotAtUtc": "2026-08-02T11:41:23.195Z",
    "queue": "tts-queue",
    "retainedCompletedJobs": 100,
    "minimumInputCharacters": 5000,
    "completedJobs": 45,
    "transcriptGeneration": false,
    "validAudioDurationJobs": 45,
    "audioDurationProbeFailures": 0,
    "finishedWindowUtc": {
      "from": "2026-08-02T11:33:21.288Z",
      "to": "2026-08-02T11:41:21.412Z"
    },
    "inputCharacterRange": [
      5263,
      153072
    ],
    "medianInputCharacters": 19934,
    "totalInputCharacters": 1385427,
    "totalOutputAudioDurationSeconds": 95050.848,
    "totalOutputAudioDurationHours": 26.403,
    "medianOutputAudioDurationSeconds": 1282.403,
    "activeProcessingSecondsRange": [
      20.959,
      158.396
    ],
    "medianActiveProcessingSeconds": 44.452,
    "p95ActiveProcessingSeconds": 126.812,
    "processingRtfRange": [
      0.0129,
      0.0789
    ],
    "medianProcessingRtf": 0.0346,
    "p95ProcessingRtf": 0.0652,
    "medianRealTimeSpeedup": 28.9,
    "realTimeSpeedupRange": [
      12.68,
      77.22
    ],
    "completedFasterThanRealTime": 45,
    "methodology": "Single-speaker jobs with at least 5,000 characters, explicit transcript=false in BullMQ and task metadata, valid worker-clock timing, and final MP3 duration successfully measured with ffprobe.",
    "claimScope": "The 45 qualifying jobs among the 100 completed jobs retained by tts-queue at the snapshot time; not all historical traffic or a success-rate calculation."
  },
  "combined100kValidation": {
    "snapshotAtUtc": "2026-08-02T11:41:51.625Z",
    "greaterThanInputCharacters": 100000,
    "completedJobs": 3,
    "jobTypes": [
      "single-speaker narration",
      "multi-speaker dialogue"
    ],
    "transcriptGeneration": false,
    "totalInputCharacters": 357627,
    "totalActiveProcessingSeconds": 366.168,
    "totalOutputAudioDurationSeconds": 20698.514,
    "totalOutputAudioDurationHours": 5.75,
    "aggregateProcessingRtf": 0.0177,
    "aggregateRealTimeSpeedup": 56.53,
    "processingRtfRange": [
      0.017,
      0.0181
    ],
    "realTimeSpeedupRange": [
      55.22,
      58.91
    ],
    "claimScope": "Three retained completed production jobs over 100,000 characters with transcript generation explicitly disabled: one single-speaker narration and two multi-speaker dialogues."
  },
  "ultraLongFormValidation": {
    "snapshotAtUtc": "2026-08-02T11:41:51.625Z",
    "queue": "v3-tts-dialogue-queue",
    "jobName": "v3-tts-dialogue-job",
    "retainedCompletedJobs": 100,
    "retainedFinishedWindowUtc": {
      "from": "2026-08-01T15:11:28.423Z",
      "to": "2026-08-02T10:40:04.034Z"
    },
    "matchedFinishedWindowUtc": {
      "from": "2026-08-02T03:36:52.881Z",
      "to": "2026-08-02T04:34:12.009Z"
    },
    "greaterThanInputCharacters": 100000,
    "completedJobs": 2,
    "transcriptGeneration": false,
    "transcriptEvidencePath": "tasks.metadata.payload.with_transcript",
    "inputCharacterRange": [
      101481,
      103074
    ],
    "activeProcessingSecondsRange": [
      103.321,
      104.451
    ],
    "recordedEndToEndSecondsRange": [
      103.342,
      104.475
    ],
    "outputAudioDurationSecondsRange": [
      5798.583,
      6153.456
    ],
    "outputAudioFormat": {
      "container": "MP3",
      "sampleRateHz": 44100,
      "channels": 2,
      "bitrateKbps": 128
    },
    "processingRtfRange": [
      0.017,
      0.0178
    ],
    "realTimeSpeedupRange": [
      56.12,
      58.91
    ],
    "completedWithin120Seconds": 2,
    "measurements": [
      {
        "label": "A",
        "inputCharacters": 101481,
        "activeProcessingSeconds": 103.321,
        "recordedEndToEndSeconds": 103.342,
        "outputAudioDurationSeconds": 5798.583,
        "processingRtf": 0.0178,
        "realTimeSpeedup": 56.12
      },
      {
        "label": "B",
        "inputCharacters": 103074,
        "activeProcessingSeconds": 104.451,
        "recordedEndToEndSeconds": 104.475,
        "outputAudioDurationSeconds": 6153.456,
        "processingRtf": 0.017,
        "realTimeSpeedup": 58.91
      }
    ],
    "dialogueDelayNote": "Both measured dialogue jobs had 18 lines and configured inter-line delay=0, so their final audio duration was not inflated by the dialogue delay setting.",
    "audioDurationEvidence": "ffprobe of each final tasks.output_uri MP3; URLs are not published",
    "clockNote": "Active processing uses worker-clock finishedOn minus processedOn. Recorded end-to-end uses finishedOn minus producer-clock timestamp and can inherit sub-second cross-host clock skew; queue-wait deltas are therefore not used as a headline claim.",
    "claimScope": "Two retained completed multi-speaker dialogue jobs over 100,000 characters, not all historical traffic or a competitor benchmark."
  },
  "allMeasured": {
    "completedJobs": 98,
    "minInputCharacters": 883,
    "medianInputCharacters": 8369,
    "p95InputCharacters": 61540,
    "maxInputCharacters": 87810,
    "totalInputCharacters": 1641265,
    "medianQueueWaitSeconds": 0.03,
    "p95QueueWaitSeconds": 0.25,
    "medianProcessingSeconds": 34.98,
    "p95ProcessingSeconds": 104.75,
    "medianEndToEndSeconds": 35,
    "p90EndToEndSeconds": 92.62,
    "p95EndToEndSeconds": 104.78,
    "maxEndToEndSeconds": 142.9,
    "medianProcessingCharactersPerSecond": 246.08,
    "p95ProcessingCharactersPerSecond": 748.2,
    "completedWithin120Seconds": 97
  },
  "longFormFocus": {
    "minimumInputCharacters": 5000,
    "completedJobs": 59,
    "minObservedInputCharacters": 5026,
    "medianInputCharacters": 17382,
    "p95InputCharacters": 70057,
    "maxObservedInputCharacters": 87810,
    "totalInputCharacters": 1550967,
    "medianQueueWaitSeconds": 0.03,
    "p95QueueWaitSeconds": 0.25,
    "medianProcessingSeconds": 59.49,
    "p95ProcessingSeconds": 110.21,
    "medianEndToEndSeconds": 59.52,
    "p90EndToEndSeconds": 103.52,
    "p95EndToEndSeconds": 110.23,
    "maxEndToEndSeconds": 142.9,
    "medianProcessingCharactersPerSecond": 360.86,
    "p95ProcessingCharactersPerSecond": 790.42,
    "completedWithin120Seconds": 58
  },
  "rows": [
    {
      "key": "all",
      "inputRange": "883–87,810",
      "completedJobs": 98,
      "medianQueueWaitSeconds": 0.03,
      "medianProcessingSeconds": 34.98,
      "medianEndToEndSeconds": 35,
      "p95EndToEndSeconds": 104.78,
      "completedWithin120Seconds": 97
    },
    {
      "key": "5k-20k",
      "inputRange": "5,026–19,804",
      "completedJobs": 31,
      "medianQueueWaitSeconds": 0.03,
      "medianProcessingSeconds": 40.1,
      "medianEndToEndSeconds": 40.36,
      "p95EndToEndSeconds": 68.37,
      "completedWithin120Seconds": 31
    },
    {
      "key": "20k-plus",
      "inputRange": "22,403–87,810",
      "completedJobs": 28,
      "medianQueueWaitSeconds": 0.03,
      "medianProcessingSeconds": 75.78,
      "medianEndToEndSeconds": 75.8,
      "p95EndToEndSeconds": 111.99,
      "completedWithin120Seconds": 27
    }
  ],
  "voiceSourceLabelMix": [
    {
      "label": "ElevenLabs",
      "completedJobs": 72
    },
    {
      "label": "MiniMax",
      "completedJobs": 22
    },
    {
      "label": "Fish Audio",
      "completedJobs": 3
    },
    {
      "label": "Vbee",
      "completedJobs": 1
    }
  ],
  "voiceSourceDisclosure": "The 153K single-speaker task metadata identified provider=elevenlabs, presented here as an ElevenLabs voice source. OpenSpeaker core high-quality TTS uses its own orchestration and synthesis bridge, so this does not promise output identical to a direct ElevenLabs request.",
  "humanReadablePages": {
    "en": "https://openspeaker.ai/long-form-tts-speed-benchmark",
    "vi": "https://openspeaker.ai/vi/benchmark-toc-do-tts-van-ban-dai"
  }
}