From a537b4032648296a60d3aea88222fd5945fc3621 Mon Sep 17 00:00:00 2001 From: Stackie Jia Date: Tue, 14 Jul 2026 20:08:24 +0800 Subject: [PATCH 1/7] feat(player): support selectable HLS and TS audio tracks --- package.json | 1 + tools/devlab/README.md | 39 +- tools/devlab/devlab.py | 255 +++++++- tsconfig.json | 1 + .../src/components/player/player-controls.tsx | 63 +- web-ui/src/components/player/video-player.tsx | 63 +- web-ui/src/i18n/player.ts | 9 + web-ui/src/lib/player-storage.ts | 10 + .../playback-engine/audio/pcm-audio-player.ts | 60 ++ .../backends/mse-playback-backend.ts | 7 +- .../backends/native-playback-backend.ts | 72 ++- .../src/playback-engine/demux/pat-pmt-pes.ts | 12 + .../playback-engine/demux/ts-demuxer.test.ts | 101 +++ .../src/playback-engine/demux/ts-demuxer.ts | 203 +++--- web-ui/src/playback-engine/hls/hls-source.ts | 92 ++- web-ui/src/playback-engine/hls/m3u8.test.ts | 94 +++ web-ui/src/playback-engine/hls/m3u8.ts | 92 ++- web-ui/src/playback-engine/index.ts | 3 + .../src/playback-engine/mse/media-source.ts | 143 +++-- .../mse/playback-controller.ts | 67 +- .../src/playback-engine/remux/mp4-remuxer.ts | 38 ++ web-ui/src/playback-engine/types.ts | 35 +- web-ui/src/playback-engine/worker/messages.ts | 11 +- web-ui/src/playback-engine/worker/pipeline.ts | 123 +++- .../playback-engine/worker/transmux-worker.ts | 578 ++++++++++++++++-- 25 files changed, 1892 insertions(+), 280 deletions(-) create mode 100644 web-ui/src/playback-engine/demux/ts-demuxer.test.ts create mode 100644 web-ui/src/playback-engine/hls/m3u8.test.ts diff --git a/package.json b/package.json index f33125d9..d37aed3c 100644 --- a/package.json +++ b/package.json @@ -8,6 +8,7 @@ "type-check": "pnpm run type-check:tsc && pnpm run type-check:basedpyright", "type-check:tsc": "tsc --noEmit", "type-check:basedpyright": "uv run --group dev basedpyright", + "test:web-ui": "node --experimental-strip-types --test web-ui/src/playback-engine/hls/*.test.ts web-ui/src/playback-engine/demux/*.test.ts", "lint": "pnpm run lint:biome && pnpm run lint:ruff && pnpm run lint:clang", "lint:biome": "biome check", "lint:ruff": "uv run --group dev ruff check && uv run --group dev ruff format --check", diff --git a/tools/devlab/README.md b/tools/devlab/README.md index c6e29dfb..7ed77683 100644 --- a/tools/devlab/README.md +++ b/tools/devlab/README.md @@ -4,15 +4,18 @@ rtp2httpd can proxy, so you can develop the web player against the scenarios it needs to support: -| Scenario | Delivery | rtp2httpd proxy | Codecs | -| ------------------ | ------------------------- | --------------- | ----------------------------------- | -| HLS-TS live | HTTP `.m3u8` + `.ts` | `/http` | `h264-mp2`, `hevc-aac` | -| HLS-fMP4 live | HTTP `.m3u8` + `.m4s` | `/http` | `h264-aac`, `hevc-aac` | -| HLS catchup | HTTP HLS VOD (`playseek`) | `/http` | all of the above | -| mpegts (RTSP) | RTSP TS live + catchup | `/rtsp` | `h264-mp2`, `hevc-aac` | -| mpegts (multicast) | RTP multicast live | `/rtp` | `h264-mp2`, `hevc-ac3`, `hevc-eac3` | -| mpegts (scan) | RTP multicast live | `/rtp` | 1080i / 1080p / 2160p (see below) | -| external file | RTP multicast (looped) | `/rtp` | whatever the `.ts` file contains | +| Scenario | Delivery | rtp2httpd proxy | Codecs | +| ------------------------ | --------------------------------- | --------------- | ----------------------------------- | +| HLS-TS live | HTTP `.m3u8` + `.ts` | `/http` | `h264-mp2`, `hevc-aac` | +| HLS-fMP4 live | HTTP `.m3u8` + `.m4s` | `/http` | `h264-aac`, `hevc-aac` | +| HLS alternate audio | HTTP master + split A/V playlists | `/http` | AAC / MP2 / AC-3 (TS), AAC (fMP4) | +| HLS internal multi-audio | HTTP `.m3u8` + multi-PID `.ts` | `/http` | H.264 + AAC / MP2 / AC-3 | +| HTTP internal multi-audio | continuous multi-PID MPEG-TS | `/http` | H.264 + AAC / MP2 / AC-3 | +| HLS catchup | HTTP HLS VOD (`playseek`) | `/http` | all of the above | +| mpegts (RTSP) | RTSP TS live + catchup | `/rtsp` | `h264-mp2`, `hevc-aac` | +| mpegts (multicast) | RTP multicast live | `/rtp` | `h264-mp2`, `hevc-ac3`, `hevc-eac3` | +| mpegts (scan) | RTP multicast live | `/rtp` | 1080i / 1080p / 2160p (see below) | +| external file | RTP multicast (looped) | `/rtp` | whatever the `.ts` file contains | - `h264-mp2` = H.264 video + MPEG-1/2 Layer II audio - `hevc-aac` = H.265/HEVC video + AAC audio @@ -23,6 +26,18 @@ HLS live is offered in both segment specs: **HLS-TS** (MPEG-TS `.ts` segments) and **HLS-fMP4** (an `init.mp4` referenced via `#EXT-X-MAP` + `.m4s` fragments). fMP4 carries AAC audio (MP2-in-MP4 is unsupported), so fMP4 channels use AAC. +The two **alternate audio** channels use a video-only playlist plus English +(440 Hz), Chinese (880 Hz), and Spanish (1320 Hz) renditions declared with +`EXT-X-MEDIA`. The TS channel combines AAC, MP2, and AC-3; the fMP4 channel uses +AAC for all three renditions. They exercise default-track selection, the player +controls selector, audible track switching, codec changes, and per-channel +selection persistence. + +The **internal multi-audio** channels carry those same three languages and +tones as separate PIDs in one MPEG-TS program. One is segmented as HLS-TS and +the other is a continuous HTTP TS response, covering both inputs accepted by +the custom demux/transmux pipeline. + **HLS catchup** is a real HLS VOD: each `playseek` window returns an `index.m3u8` (`#EXT-X-PLAYLIST-TYPE:VOD`) listing fixed-duration `.ts` slices. The playlist is produced instantly; each slice is encoded lazily on first @@ -65,7 +80,7 @@ exact codecs/bitstream are relayed) — useful for reproducing a stream attached to a bug report: ```bash -uv run python tools/devlab/devlab.py --ts-file /path/to/user-report.ts +uv run tools/devlab/devlab.py --ts-file /path/to/user-report.ts # adds a "file " channel proxied through rtp2httpd ``` @@ -80,7 +95,7 @@ UTC` — proving the time → picture mapping end to end. ```bash # 1. start the upstreams (writes /tmp/r2h-devlab.conf) -uv run python tools/devlab/devlab.py +uv run tools/devlab/devlab.py # The script prints the rtp2httpd command. # 2. in another shell, run rtp2httpd against the generated config @@ -104,7 +119,7 @@ ffmpeg -hide_banner -filters | grep drawtext ``` Defaults: HTTP origin `127.0.0.1:8881`, RTSP origin `127.0.0.1:8554`, -rtp2httpd port `5140`. Run `uv run python tools/devlab/devlab.py --help` for +rtp2httpd port `5140`. Run `uv run tools/devlab/devlab.py --help` for options. ## Verifying a scenario without the browser diff --git a/tools/devlab/devlab.py b/tools/devlab/devlab.py index df1b8930..6dbb926e 100644 --- a/tools/devlab/devlab.py +++ b/tools/devlab/devlab.py @@ -46,6 +46,7 @@ import time from datetime import datetime, timezone from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from typing import NamedTuple from urllib.parse import parse_qs, urlparse FONT: str | None = None @@ -91,6 +92,35 @@ ("HLS-fMP4 (hevc-aac)", "hevc-aac", "fmp4", "hevc-aac-fmp4"), ) +# Alternate-audio HLS masters: video-only rendition plus three independently +# segmented audio renditions. The tones make track switching audible. +# Tuple: (channel_label, seg_type, live_key). +ALT_AUDIO_HLS_CHANNELS = ( + ("HLS-TS alternate audio", "mpegts", "h264-aac-ts-alt-audio"), + ("HLS-fMP4 alternate audio", "fmp4", "h264-aac-fmp4-alt-audio"), +) + +# One MPEG-TS program containing AAC, MP2 and AC-3 elementary audio streams. +INTERNAL_AUDIO_HLS_CHANNELS = (("HLS-TS internal multi audio", "h264-internal-multi-audio"),) + + +class AudioRendition(NamedTuple): + directory: str + name: str + hls_language: str + ts_language: str + tone: int + ts_profile: str + codec: str + bitrate: str + + +AUDIO_RENDITIONS = ( + AudioRendition("audio-en", "English", "en", "eng", 440, "h264-aac", "aac", "128k"), + AudioRendition("audio-zh", "中文", "zh", "zho", 880, "h264-mp2", "mp2", "192k"), + AudioRendition("audio-es", "Español", "es", "spa", 1320, "h264-ac3", "ac3", "192k"), +) + # --------------------------------------------------------------------------- # ffmpeg argument helpers # --------------------------------------------------------------------------- @@ -243,6 +273,31 @@ def lavfi_inputs(size: str = "1280x720", rate: int = 25) -> list[str]: ] +def multi_audio_media_args() -> list[str]: + """H.264 plus three distinctly pitched/language-tagged audio PIDs.""" + args = [ + "-f", + "lavfi", + "-i", + "testsrc2=size=1280x720:rate=25", + ] + for rendition in AUDIO_RENDITIONS: + args += ["-f", "lavfi", "-i", f"sine=frequency={rendition.tone}:sample_rate=48000"] + args += ["-map", "0:v"] + for index, rendition in enumerate(AUDIO_RENDITIONS): + args += [ + "-map", + f"{index + 1}:a", + f"-metadata:s:a:{index}", + f"language={rendition.ts_language}", + f"-c:a:{index}", + rendition.codec, + f"-b:a:{index}", + rendition.bitrate, + ] + return [*args, "-vf", live_filter("h264 multi-audio"), *video_args("h264-aac"), "-ar:a", "48000", "-ac:a", "2"] + + # --------------------------------------------------------------------------- # Time parsing for playseek # --------------------------------------------------------------------------- @@ -293,11 +348,21 @@ class LiveHLS: or ``fmp4`` (HLS-fMP4, an init.mp4 + .m4s fragments referenced via EXT-X-MAP). """ - def __init__(self, ffmpeg: str, profile: str, outdir: str, seg_type: str = "mpegts"): + def __init__( + self, + ffmpeg: str, + profile: str, + outdir: str, + seg_type: str = "mpegts", + media: str = "muxed", + tone: int = 440, + ): self.ffmpeg = ffmpeg self.profile = profile self.outdir = outdir self.seg_type = seg_type + self.media = media + self.tone = tone self.log_path = os.path.join(outdir, "ffmpeg.log") self.proc: subprocess.Popen[bytes] | None = None self._stderr = None @@ -330,20 +395,38 @@ def start(self) -> None: "-hls_segment_filename", os.path.join(self.outdir, "seg_%05d.ts"), ] - cmd = [ - self.ffmpeg, - "-hide_banner", - "-loglevel", - "error", - "-re", - *lavfi_inputs(), - "-vf", - live_filter(self.profile), - *video_args(self.profile), - *audio_args(self.profile), - *hls_opts, - os.path.join(self.outdir, "index.m3u8"), - ] + common = [self.ffmpeg, "-hide_banner", "-loglevel", "error", "-re"] + if self.media == "video": + media_args = [ + "-f", + "lavfi", + "-i", + "testsrc2=size=1280x720:rate=25", + "-vf", + live_filter(self.profile), + *video_args(self.profile), + "-an", + ] + elif self.media == "audio": + media_args = [ + "-f", + "lavfi", + "-i", + f"sine=frequency={self.tone}:sample_rate=48000", + "-vn", + *audio_args(self.profile), + ] + elif self.media == "multi-audio": + media_args = multi_audio_media_args() + else: + media_args = [ + *lavfi_inputs(), + "-vf", + live_filter(self.profile), + *video_args(self.profile), + *audio_args(self.profile), + ] + cmd = [*common, *media_args, *hls_opts, os.path.join(self.outdir, "index.m3u8")] self._stderr = open(self.log_path, "ab") self.proc = subprocess.Popen(cmd, stdout=subprocess.DEVNULL, stderr=self._stderr) @@ -362,7 +445,74 @@ def stop(self) -> None: self._stderr.close() -def wait_for_live_hls(gens: list[LiveHLS], timeout: float) -> None: +class AlternateAudioHLS: + """Builds a master playlist from one video-only and multiple audio-only HLS generators.""" + + def __init__(self, ffmpeg: str, outdir: str, seg_type: str): + self.profile = "h264-aac-alt-audio" + self.seg_type = seg_type + self.outdir = outdir + self.log_path = os.path.join(outdir, "ffmpeg.log") + self.proc: subprocess.Popen[bytes] | None = None + self.generators = [ + LiveHLS(ffmpeg, "h264-aac", os.path.join(outdir, "video"), seg_type, media="video"), + *( + LiveHLS( + ffmpeg, + rendition.ts_profile if seg_type == "mpegts" else "h264-aac", + os.path.join(outdir, rendition.directory), + seg_type, + media="audio", + tone=rendition.tone, + ) + for rendition in AUDIO_RENDITIONS + ), + ] + + def start(self) -> None: + os.makedirs(self.outdir, exist_ok=True) + for generator in self.generators: + generator.start() + + def ready(self) -> bool: + if not all(generator.ready() for generator in self.generators): + return False + master = os.path.join(self.outdir, "index.m3u8") + if not os.path.isfile(master): + audio_lines = [ + '#EXT-X-MEDIA:TYPE=AUDIO,GROUP-ID="audio",' + f'NAME="{rendition.name}",DEFAULT={"YES" if index == 0 else "NO"},' + f'AUTOSELECT=YES,LANGUAGE="{rendition.hls_language}",URI="{rendition.directory}/index.m3u8"' + for index, rendition in enumerate(AUDIO_RENDITIONS) + ] + content = "\n".join( + [ + "#EXTM3U", + "#EXT-X-VERSION:3", + *audio_lines, + '#EXT-X-STREAM-INF:BANDWIDTH=2400000,CODECS="avc1.64001f,mp4a.40.2",' + 'RESOLUTION=1280x720,FRAME-RATE=25.000,AUDIO="audio"', + "video/index.m3u8", + ] + ) + with open(master, "w", encoding="utf-8") as fh: + fh.write(f"{content}\n") + return True + + def failure_message(self) -> str: + failures = [ + generator.failure_message() + for generator in self.generators + if generator.proc is not None and generator.proc.poll() is not None + ] + return "\n".join(failures) if failures else f"alternate-audio HLS generator failed; root: {self.outdir}" + + def stop(self) -> None: + for generator in self.generators: + generator.stop() + + +def wait_for_live_hls(gens: list[LiveHLS | AlternateAudioHLS], timeout: float) -> None: if timeout <= 0: return @@ -474,9 +624,9 @@ def _serve_file(self, path: str) -> None: self._send(200, ctype, fh.read()) def _live(self, parts: list[str]) -> None: - # /live// - profile, fname = parts[1], parts[2] - self._serve_file(os.path.join(live_root, profile, fname)) + # /live// + profile = parts[1] + self._serve_file(os.path.join(live_root, profile, *parts[2:])) def _catchup_playlist(self, profile: str, qs: dict[str, list[str]]) -> None: playseek = (qs.get("playseek") or qs.get("tvdr") or [""])[0] @@ -507,18 +657,53 @@ def _catchup_seg(self, profile: str, begin: int, idx: int) -> None: if proc.poll() is None: proc.terminate() + def _multi_ts(self) -> None: + self.send_response(200) + self.send_header("Content-Type", "video/mp2t") + self.send_header("Access-Control-Allow-Origin", "*") + self.end_headers() + proc = subprocess.Popen( + [ + catchup.ffmpeg, + "-hide_banner", + "-loglevel", + "error", + "-re", + *multi_audio_media_args(), + "-f", + "mpegts", + "-", + ], + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + ) + assert proc.stdout is not None + try: + while True: + chunk = proc.stdout.read(65536) + if not chunk: + break + self.wfile.write(chunk) + except BrokenPipeError, ConnectionError: + pass + finally: + if proc.poll() is None: + proc.terminate() + def do_GET(self) -> None: # noqa: N802 (http.server API) parsed = urlparse(self.path) qs = parse_qs(parsed.query) parts = [p for p in parsed.path.split("/") if p] try: - if len(parts) == 3 and parts[0] == "live": + if len(parts) >= 3 and parts[0] == "live": self._live(parts) elif len(parts) >= 2 and parts[0] == "catchup" and parts[-1] == "index.m3u8": self._catchup_playlist(parts[1], qs) elif len(parts) == 4 and parts[0] == "catchup-seg": # /catchup-seg///.ts self._catchup_seg(parts[1], int(parts[2]), int(parts[3].removesuffix(".ts"))) + elif parts == ["multi-audio.ts"]: + self._multi_ts() else: self._send(404, "text/plain", b"not found") except BrokenPipeError, ConnectionError: @@ -791,6 +976,20 @@ def build_services_m3u(http_hostport: str, rtsp_hostport: str, mcast_channels: l f'#EXTINF:-1 group-title="HLS" catchup="default" catchup-source="{src}",{label}', f"http://{http_hostport}/live/{key}/index.m3u8", ] + for label, _seg_type, key in ALT_AUDIO_HLS_CHANNELS: + lines += [ + f'#EXTINF:-1 group-title="HLS",{label}', + f"http://{http_hostport}/live/{key}/index.m3u8", + ] + for label, key in INTERNAL_AUDIO_HLS_CHANNELS: + lines += [ + f'#EXTINF:-1 group-title="HLS",{label}', + f"http://{http_hostport}/live/{key}/index.m3u8", + ] + lines += [ + '#EXTINF:-1 group-title="mpegts (HTTP)",HTTP TS internal multi audio', + f"http://{http_hostport}/multi-audio.ts", + ] for prof in RTSP_PROFILES: # mpegts-over-RTSP live + RTSP TS catchup window: covers mpegts 回看 src = f"rtsp://{rtsp_hostport}/catchup/{prof}?{tpl}" @@ -865,10 +1064,24 @@ def main() -> int: live_root = os.path.join(args.workdir, "live") - live = [ + live: list[LiveHLS | AlternateAudioHLS] = [ LiveHLS(args.ffmpeg, prof, os.path.join(live_root, key), seg_type) for _label, prof, seg_type, key in HLS_CHANNELS ] + live += [ + AlternateAudioHLS(args.ffmpeg, os.path.join(live_root, key), seg_type) + for _label, seg_type, key in ALT_AUDIO_HLS_CHANNELS + ] + live += [ + LiveHLS( + args.ffmpeg, + "h264-multi-audio", + os.path.join(live_root, key), + "mpegts", + media="multi-audio", + ) + for _label, key in INTERNAL_AUDIO_HLS_CHANNELS + ] for gen in live: gen.start() try: diff --git a/tsconfig.json b/tsconfig.json index 9c956da1..ef4ed34e 100644 --- a/tsconfig.json +++ b/tsconfig.json @@ -7,6 +7,7 @@ "skipLibCheck": true, "esModuleInterop": true, "allowSyntheticDefaultImports": true, + "allowImportingTsExtensions": true, "strict": true, "forceConsistentCasingInFileNames": true, "module": "ESNext", diff --git a/web-ui/src/components/player/player-controls.tsx b/web-ui/src/components/player/player-controls.tsx index d9466bda..e64af989 100644 --- a/web-ui/src/components/player/player-controls.tsx +++ b/web-ui/src/components/player/player-controls.tsx @@ -1,6 +1,7 @@ import { clsx } from "clsx"; import { History, + Languages, Maximize, Minimize, PanelRightClose, @@ -17,7 +18,7 @@ import { memo, useCallback, useMemo, useRef, useState } from "react"; import { usePlayerTranslation } from "../../hooks/use-player-translation"; import type { Locale } from "../../lib/locale"; import { createProgramTimeline, programProgressToWallClock } from "../../lib/program-timeline"; -import type { PlayerMediaInfo, PlayerRenderState } from "../../playback-engine"; +import type { PlayerAudioTrackState, PlayerMediaInfo, PlayerRenderState } from "../../playback-engine"; import { isNearLiveWallClock, type LiveSessionAnchor, mseToWallClock } from "../../playback-engine/timeline/wall-clock"; import type { Channel, EPGProgram } from "../../types/player"; import { PLAYER_CONTROL_BUTTON_CLASS, PLAYER_OVERLAY_SURFACE_CLASS } from "./classnames"; @@ -40,6 +41,8 @@ interface PlayerControlsProps { locale: Locale; // Technical information for the currently visible player slot mediaInfo: PlayerMediaInfo | null; + audioTrackState: PlayerAudioTrackState; + onAudioTrackChange: (trackId: string) => void; renderState: PlayerRenderState; autoDeinterlace: boolean; // The absolute time of the last seek position (null for live mode) @@ -383,6 +386,8 @@ function PlayerControlsComponent({ onScrubbingChange, locale, mediaInfo, + audioTrackState, + onAudioTrackChange, renderState, autoDeinterlace, seekStartTime, @@ -408,6 +413,9 @@ function PlayerControlsComponent({ const isEffectivelyMuted = isMuted || volume <= 0; const isCatchupSupported = channel.sources.some((s) => s.catchup && s.catchupSource); const hasTimeline = isCatchupSupported || Boolean(currentProgram); + const selectedAudioTrack = audioTrackState.tracks.find( + (track) => track.id === (audioTrackState.pendingTrackId ?? audioTrackState.selectedTrackId), + ); return (
)} + {/* Audio Track Selector */} + {audioTrackState.tracks.length > 1 && ( +
+ +
+ + {audioTrackState.tracks.map((track) => ( + + ))} +
+
+ )} + {/* Fullscreen */}