extract pssh from media segments

This commit is contained in:
Nirvana
2026-10-01 09:14:42 +02:00
parent 107bb88222
commit 74de08d403
4 changed files with 275 additions and 15 deletions
+71 -13
View File
@@ -19,7 +19,8 @@ and get_catchup_content_drm_configs):
2. fetch manifest, parse ONCE → (is_encrypted, pssh_list); analysis failures
degrade gracefully rather than crashing the request
3. verified-clear short-circuit
4. generic plugin phase (with stub-PSSH upgrade via init segment)
4. generic plugin phase (with stub-PSSH upgrade via init segment, then
first media segment for providers that put the pssh in the moof)
5. provider DRM configs (generics become the base list if provider has none)
6. system-specific plugin loop with incremental ClearKey coverage checks
7. final composition: generic merge, ClearKey validation, reinstatement
@@ -57,6 +58,10 @@ _PSSH_TAG_RE = re.compile(r'<(?:cenc:)?pssh[^>]*>', re.IGNORECASE)
# kwargs that produce distinct DRM results and therefore belong in cache keys
_VARIANT_KEYS = ("drm_variant", "preferred_quality", "preferred_format")
# The media-segment PSSH fallback only needs the moof at the start of the
# segment; request just this many bytes (MP4PSSHExtractor reads 100 KB at most).
_MEDIA_SEGMENT_PROBE_BYTES = 100 * 1024
class TTLCache:
"""Thread-safe TTL cache with LRU eviction and a hard size bound.
@@ -450,7 +455,8 @@ class DRMOperations:
1. PSSH TTL cache (may hold a previous init-segment-upgraded result —
still valid even if THIS call's parse failed)
2. The list parsed during manifest analysis (populates the cache)
3. Full extraction: manifest (re-)fetch + init-segment fallback
3. Full extraction: manifest (re-)fetch + init-segment fallback +
first-media-segment fallback
Note: the cache key includes catchup start/end times, so each timeshift
window gets its own entry even though PSSH is likely identical per
@@ -578,12 +584,15 @@ class DRMOperations:
pssh_list: Optional[List] = None,
**kwargs,
) -> List:
"""Extract PSSH data from a manifest, falling back to the init segment.
"""Extract PSSH data from a manifest, falling back to segments.
Levels, each tried only while the previous one left a PSSH unresolved:
1. manifest, 2. init segment (moov), 3. first media segment (moof).
manifest_headers now defaults to None so the legacy facade call
(which passes only the URL) works. pssh_list lets callers that already
parsed the manifest skip the redundant re-parse and go straight to
init-segment extraction (passing [] implies the manifest was already
segment extraction (passing [] implies the manifest was already
parsed and yielded nothing).
**kwargs (which contain start_time/end_time for catchup) are passed to
@@ -626,21 +635,54 @@ class DRMOperations:
if needs_segment_extraction:
if not manifest_content:
# Content is needed to locate the init segment URL.
# Content is needed to locate the segment URLs.
manifest_content = self._fetch_manifest_text(http, manifest_url, manifest_headers)
if manifest_content:
init_segment_url = ManifestParser.extract_single_init_segment_url(
if not manifest_content:
return pssh_list
# Level 2: init segment (pssh in moov)
init_segment_url = ManifestParser.extract_single_init_segment_url(
manifest_content, manifest_url
)
if init_segment_url:
segment_pssh = DRMExtractor._extract_from_single_segment(
init_segment_url,
[p.system_id for p in pssh_list] if pssh_list else [],
headers=segment_headers,
http_manager=http,
)
if segment_pssh:
pssh_list = DRMExtractor._merge_pssh_data(pssh_list, segment_pssh)
# Level 3: first media segment (pssh in moof). Only while a
# system that should carry a PSSH is still unresolved, so
# providers whose init segment works never pay for this and
# ClearKey stubs (legitimately PSSH-less) don't trigger it.
if self._has_unresolved_pssh(pssh_list):
media_segment_url = ManifestParser.extract_first_media_segment_url(
manifest_content, manifest_url
)
if init_segment_url:
segment_pssh = DRMExtractor._extract_from_single_segment(
init_segment_url,
if media_segment_url:
# Only the moof at the start of the segment is needed.
probe_headers = {
**(segment_headers or {}),
"Range": f"bytes=0-{_MEDIA_SEGMENT_PROBE_BYTES - 1}",
}
media_pssh = DRMExtractor._extract_from_single_segment(
media_segment_url,
[p.system_id for p in pssh_list] if pssh_list else [],
headers=segment_headers,
headers=probe_headers,
http_manager=http,
)
if segment_pssh:
return DRMExtractor._merge_pssh_data(pssh_list, segment_pssh)
if media_pssh:
pssh_list = DRMExtractor._merge_pssh_data(pssh_list, media_pssh)
if self._has_unresolved_pssh(pssh_list):
logger.warning(
f"DRMOperations: PSSH still unresolved after manifest, init and "
f"media segment extraction for {channel_id or manifest_url}; "
f"license acquisition may fail"
)
return pssh_list
@@ -717,6 +759,22 @@ class DRMOperations:
return True
return False
@staticmethod
def _has_unresolved_pssh(pssh_list: Optional[List]) -> bool:
"""True if a system that should carry a PSSH is still a stub.
Like _has_stub_pssh, but ClearKey stubs don't count: ClearKey content
legitimately has no PSSH box, so chasing one would only cost a
pointless segment download on every uncached resolution. An empty
list counts as unresolved (nothing found yet).
"""
if not pssh_list:
return True
return any(
(not p.pssh_box or not p.key_ids) and p.drm_system != DRMSystem.CLEARKEY
for p in pssh_list
)
def _needs_pssh_extraction(self, drm_configs) -> bool:
config_systems = {config.system for config in drm_configs}
plugin_systems = {
@@ -1,6 +1,6 @@
# streaming_providers/base/utils/manifest_parser.py
"""
DASH manifest parser for extracting init segment URLs.
DASH manifest parser for extracting init and first-media segment URLs.
For PSSH/DRM extraction, use drm_extractor module.
"""
@@ -157,6 +157,88 @@ class ManifestParser:
logger.warning("Could not find init segment URL in manifest")
return None
@staticmethod
def extract_first_media_segment_url(
manifest_content: str,
manifest_url: str
) -> Optional[str]:
"""
Extract the URL of the FIRST media segment from a DASH manifest.
Counterpart to extract_single_init_segment_url(), used as a PSSH
fallback for providers that put the pssh box in the moof of each media
segment instead of the manifest or the init segment.
Only SegmentTemplate manifests are supported (SegmentBase has no
separate media segments). Template variables are resolved as follows:
$RepresentationID$ first Representation ID of the AdaptationSet
$Bandwidth$ bandwidth of the first Representation
$Time$ t of the first <S> of the SegmentTimeline (else 0)
$Number$ startNumber of the SegmentTemplate (else 1)
Templates with format specifiers (e.g. $Number%05d$) are not resolved
and are skipped rather than requested with a broken URL.
Args:
manifest_content: Full manifest XML content
manifest_url: URL where the manifest was fetched from
Returns:
Full URL to the first media segment, or None if not found
"""
base_urls = ManifestUtils.extract_base_urls(manifest_content)
effective_base = URLResolver.build_effective_base_url(manifest_url, base_urls)
adaptation_sets = ManifestUtils.parse_adaptation_sets(manifest_content)
video_sets, audio_sets = ManifestUtils.separate_video_audio_sets(adaptation_sets)
# Try video first, then audio (same order as the init segment lookup)
for ad_set_info in video_sets + audio_sets:
media_template = ManifestUtils.extract_segment_template_media(
ad_set_info.content
)
if not media_template:
continue
logger.debug(f"Found media template: {media_template}")
rep_id = ManifestUtils.extract_first_representation_id(ad_set_info.content)
if not rep_id:
logger.debug("No Representation ID found in AdaptationSet")
continue
bandwidth = ManifestUtils.extract_first_representation_bandwidth(
ad_set_info.content
) or "0"
first_time = ManifestUtils.extract_first_segment_time(ad_set_info.content) or "0"
start_number = (
ManifestUtils.extract_segment_template_start_number(ad_set_info.content)
or "1"
)
media_url = URLResolver.substitute_template_variables(
media_template,
representation_id=rep_id,
bandwidth=bandwidth,
time=first_time,
number=start_number
)
if "$" in media_url:
logger.debug(f"Unresolved template variables in media URL: {media_url}")
continue
full_url = URLResolver.construct_full_url(
effective_base,
media_url,
url_encode_filename=True
)
logger.info(f"Constructed media segment URL (SegmentTemplate): {full_url}")
return full_url
logger.debug("Could not find media segment URL in manifest")
return None
@staticmethod
def extract_segment_urls(manifest_content: str, manifest_url: str) -> List[str]:
"""
@@ -128,6 +128,76 @@ class ManifestUtils:
)
return match.group(1) if match else None
@staticmethod
def _find_segment_template_tag(ad_set_content: str, attribute: str) -> Optional[str]:
"""Return the first <SegmentTemplate ...> opening tag that carries `attribute`."""
for match in re.finditer(r"<SegmentTemplate\b[^>]*>", ad_set_content, re.IGNORECASE):
if re.search(rf'\b{attribute}="', match.group(0)):
return match.group(0)
return None
@staticmethod
def extract_segment_template_media(ad_set_content: str) -> Optional[str]:
"""
Extract media attribute from SegmentTemplate.
Mirrors extract_segment_template_initialization().
Args:
ad_set_content: AdaptationSet XML content
Returns:
Media template string or None if not found
"""
tag = ManifestUtils._find_segment_template_tag(ad_set_content, "media")
if not tag:
return None
match = re.search(r'\bmedia="([^"]+)"', tag)
return match.group(1) if match else None
@staticmethod
def extract_segment_template_start_number(ad_set_content: str) -> Optional[str]:
"""
Extract startNumber from the SegmentTemplate that carries the media template.
Returns:
startNumber as string, or None if the attribute is absent
(DASH default is 1, the caller decides).
"""
tag = ManifestUtils._find_segment_template_tag(ad_set_content, "media")
if not tag:
return None
match = re.search(r'\bstartNumber="(\d+)"', tag)
return match.group(1) if match else None
@staticmethod
def extract_first_segment_time(ad_set_content: str) -> Optional[str]:
"""
Extract the t attribute of the FIRST <S> element of a SegmentTimeline.
The attribute is read from that one tag only (attribute order is free
in XML, and a first <S> without t must not pick up a later one's t).
Returns:
Start time as string, or None if there is no timeline or the
first <S> has no explicit t.
"""
timeline = re.search(
r"<SegmentTimeline[^>]*>\s*(<S\b[^>]*>)",
ad_set_content,
re.IGNORECASE,
)
if not timeline:
return None
match = re.search(r'\bt="(\d+)"', timeline.group(1))
return match.group(1) if match else None
@staticmethod
def extract_first_representation_bandwidth(ad_set_content: str) -> Optional[str]:
"""Extract the bandwidth of the first Representation in an AdaptationSet."""
match = re.search(r'<Representation[^>]*\bbandwidth="(\d+)"', ad_set_content)
return match.group(1) if match else None
@staticmethod
def extract_base_urls(manifest_content: str) -> List[str]:
"""
@@ -97,11 +97,17 @@ class MP4PSSHExtractor:
box_type = data[offset + 4: offset + 8]
if box_type == b"moov":
# Look for PSSH in moov container
# Look for PSSH in moov container (init segments)
moov_data = data[offset: offset + box_size]
pssh_in_moov = MP4PSSHExtractor._extract_from_moov(moov_data)
pssh_data_list.extend(pssh_in_moov)
elif box_type == b"moof":
# Look for PSSH in moof container (media segments, e.g. Allente)
moof_data = data[offset: offset + box_size]
pssh_in_moof = MP4PSSHExtractor._extract_from_moof(moof_data)
pssh_data_list.extend(pssh_in_moof)
elif box_type == b"pssh":
# Found standalone PSSH box
pssh_box = MP4PSSHExtractor._parse_pssh_box(
@@ -254,6 +260,50 @@ class MP4PSSHExtractor:
return pssh_list
@staticmethod
def _extract_from_moof(moof_data: bytes) -> List[PSSHData]:
"""
Extract PSSH boxes from a moof (movie fragment) box.
Media segments carry moof where init segments carry moov. Per
ISO/IEC 23001-7 a pssh box may live in moov or moof (as a sibling of
mfhd/traf), so only the direct children of moof are inspected.
Args:
moof_data: Raw moof box data
Returns:
List of PSSHData objects found in moof
"""
pssh_list = []
offset = 8 # Skip moof header
while offset < len(moof_data):
try:
if offset + 8 > len(moof_data):
break
box_size = struct.unpack(">I", moof_data[offset: offset + 4])[0]
box_type = moof_data[offset + 4: offset + 8]
if box_size < 8 or offset + box_size > len(moof_data):
break
if box_type == b"pssh":
pssh_box = MP4PSSHExtractor._parse_pssh_box(
moof_data[offset: offset + box_size]
)
if pssh_box:
pssh_list.append(pssh_box)
offset += box_size
except Exception as e:
logger.debug(f"Error parsing moof box at offset {offset}: {e}")
break
return pssh_list
@staticmethod
def _extract_from_trak(trak_data: bytes) -> List[PSSHData]:
"""Extract PSSH from trak box"""