diff --git a/lib/streaming_providers/base/drm_operations.py b/lib/streaming_providers/base/drm_operations.py index c694c36..5e8815c 100644 --- a/lib/streaming_providers/base/drm_operations.py +++ b/lib/streaming_providers/base/drm_operations.py @@ -19,7 +19,8 @@ and get_catchup_content_drm_configs): 2. fetch manifest, parse ONCE → (is_encrypted, pssh_list); analysis failures degrade gracefully rather than crashing the request 3. verified-clear short-circuit -4. generic plugin phase (with stub-PSSH upgrade via init segment) +4. generic plugin phase (with stub-PSSH upgrade via init segment, then + first media segment for providers that put the pssh in the moof) 5. provider DRM configs (generics become the base list if provider has none) 6. system-specific plugin loop with incremental ClearKey coverage checks 7. final composition: generic merge, ClearKey validation, reinstatement @@ -57,6 +58,10 @@ _PSSH_TAG_RE = re.compile(r'<(?:cenc:)?pssh[^>]*>', re.IGNORECASE) # kwargs that produce distinct DRM results and therefore belong in cache keys _VARIANT_KEYS = ("drm_variant", "preferred_quality", "preferred_format") +# The media-segment PSSH fallback only needs the moof at the start of the +# segment; request just this many bytes (MP4PSSHExtractor reads 100 KB at most). +_MEDIA_SEGMENT_PROBE_BYTES = 100 * 1024 + class TTLCache: """Thread-safe TTL cache with LRU eviction and a hard size bound. @@ -450,7 +455,8 @@ class DRMOperations: 1. PSSH TTL cache (may hold a previous init-segment-upgraded result — still valid even if THIS call's parse failed) 2. The list parsed during manifest analysis (populates the cache) - 3. Full extraction: manifest (re-)fetch + init-segment fallback + 3. Full extraction: manifest (re-)fetch + init-segment fallback + + first-media-segment fallback Note: the cache key includes catchup start/end times, so each timeshift window gets its own entry even though PSSH is likely identical per @@ -578,12 +584,15 @@ class DRMOperations: pssh_list: Optional[List] = None, **kwargs, ) -> List: - """Extract PSSH data from a manifest, falling back to the init segment. + """Extract PSSH data from a manifest, falling back to segments. + + Levels, each tried only while the previous one left a PSSH unresolved: + 1. manifest, 2. init segment (moov), 3. first media segment (moof). manifest_headers now defaults to None so the legacy facade call (which passes only the URL) works. pssh_list lets callers that already parsed the manifest skip the redundant re-parse and go straight to - init-segment extraction (passing [] implies the manifest was already + segment extraction (passing [] implies the manifest was already parsed and yielded nothing). **kwargs (which contain start_time/end_time for catchup) are passed to @@ -626,21 +635,54 @@ class DRMOperations: if needs_segment_extraction: if not manifest_content: - # Content is needed to locate the init segment URL. + # Content is needed to locate the segment URLs. manifest_content = self._fetch_manifest_text(http, manifest_url, manifest_headers) - if manifest_content: - init_segment_url = ManifestParser.extract_single_init_segment_url( + if not manifest_content: + return pssh_list + + # Level 2: init segment (pssh in moov) + init_segment_url = ManifestParser.extract_single_init_segment_url( + manifest_content, manifest_url + ) + if init_segment_url: + segment_pssh = DRMExtractor._extract_from_single_segment( + init_segment_url, + [p.system_id for p in pssh_list] if pssh_list else [], + headers=segment_headers, + http_manager=http, + ) + if segment_pssh: + pssh_list = DRMExtractor._merge_pssh_data(pssh_list, segment_pssh) + + # Level 3: first media segment (pssh in moof). Only while a + # system that should carry a PSSH is still unresolved, so + # providers whose init segment works never pay for this and + # ClearKey stubs (legitimately PSSH-less) don't trigger it. + if self._has_unresolved_pssh(pssh_list): + media_segment_url = ManifestParser.extract_first_media_segment_url( manifest_content, manifest_url ) - if init_segment_url: - segment_pssh = DRMExtractor._extract_from_single_segment( - init_segment_url, + if media_segment_url: + # Only the moof at the start of the segment is needed. + probe_headers = { + **(segment_headers or {}), + "Range": f"bytes=0-{_MEDIA_SEGMENT_PROBE_BYTES - 1}", + } + media_pssh = DRMExtractor._extract_from_single_segment( + media_segment_url, [p.system_id for p in pssh_list] if pssh_list else [], - headers=segment_headers, + headers=probe_headers, http_manager=http, ) - if segment_pssh: - return DRMExtractor._merge_pssh_data(pssh_list, segment_pssh) + if media_pssh: + pssh_list = DRMExtractor._merge_pssh_data(pssh_list, media_pssh) + + if self._has_unresolved_pssh(pssh_list): + logger.warning( + f"DRMOperations: PSSH still unresolved after manifest, init and " + f"media segment extraction for {channel_id or manifest_url}; " + f"license acquisition may fail" + ) return pssh_list @@ -717,6 +759,22 @@ class DRMOperations: return True return False + @staticmethod + def _has_unresolved_pssh(pssh_list: Optional[List]) -> bool: + """True if a system that should carry a PSSH is still a stub. + + Like _has_stub_pssh, but ClearKey stubs don't count: ClearKey content + legitimately has no PSSH box, so chasing one would only cost a + pointless segment download on every uncached resolution. An empty + list counts as unresolved (nothing found yet). + """ + if not pssh_list: + return True + return any( + (not p.pssh_box or not p.key_ids) and p.drm_system != DRMSystem.CLEARKEY + for p in pssh_list + ) + def _needs_pssh_extraction(self, drm_configs) -> bool: config_systems = {config.system for config in drm_configs} plugin_systems = { diff --git a/lib/streaming_providers/base/utils/manifest_parser.py b/lib/streaming_providers/base/utils/manifest_parser.py index 7e7ca94..28f078d 100644 --- a/lib/streaming_providers/base/utils/manifest_parser.py +++ b/lib/streaming_providers/base/utils/manifest_parser.py @@ -1,6 +1,6 @@ # streaming_providers/base/utils/manifest_parser.py """ -DASH manifest parser for extracting init segment URLs. +DASH manifest parser for extracting init and first-media segment URLs. For PSSH/DRM extraction, use drm_extractor module. """ @@ -157,6 +157,88 @@ class ManifestParser: logger.warning("Could not find init segment URL in manifest") return None + @staticmethod + def extract_first_media_segment_url( + manifest_content: str, + manifest_url: str + ) -> Optional[str]: + """ + Extract the URL of the FIRST media segment from a DASH manifest. + + Counterpart to extract_single_init_segment_url(), used as a PSSH + fallback for providers that put the pssh box in the moof of each media + segment instead of the manifest or the init segment. + + Only SegmentTemplate manifests are supported (SegmentBase has no + separate media segments). Template variables are resolved as follows: + $RepresentationID$ first Representation ID of the AdaptationSet + $Bandwidth$ bandwidth of the first Representation + $Time$ t of the first of the SegmentTimeline (else 0) + $Number$ startNumber of the SegmentTemplate (else 1) + Templates with format specifiers (e.g. $Number%05d$) are not resolved + and are skipped rather than requested with a broken URL. + + Args: + manifest_content: Full manifest XML content + manifest_url: URL where the manifest was fetched from + + Returns: + Full URL to the first media segment, or None if not found + """ + base_urls = ManifestUtils.extract_base_urls(manifest_content) + effective_base = URLResolver.build_effective_base_url(manifest_url, base_urls) + + adaptation_sets = ManifestUtils.parse_adaptation_sets(manifest_content) + video_sets, audio_sets = ManifestUtils.separate_video_audio_sets(adaptation_sets) + + # Try video first, then audio (same order as the init segment lookup) + for ad_set_info in video_sets + audio_sets: + media_template = ManifestUtils.extract_segment_template_media( + ad_set_info.content + ) + if not media_template: + continue + + logger.debug(f"Found media template: {media_template}") + + rep_id = ManifestUtils.extract_first_representation_id(ad_set_info.content) + if not rep_id: + logger.debug("No Representation ID found in AdaptationSet") + continue + + bandwidth = ManifestUtils.extract_first_representation_bandwidth( + ad_set_info.content + ) or "0" + first_time = ManifestUtils.extract_first_segment_time(ad_set_info.content) or "0" + start_number = ( + ManifestUtils.extract_segment_template_start_number(ad_set_info.content) + or "1" + ) + + media_url = URLResolver.substitute_template_variables( + media_template, + representation_id=rep_id, + bandwidth=bandwidth, + time=first_time, + number=start_number + ) + + if "$" in media_url: + logger.debug(f"Unresolved template variables in media URL: {media_url}") + continue + + full_url = URLResolver.construct_full_url( + effective_base, + media_url, + url_encode_filename=True + ) + + logger.info(f"Constructed media segment URL (SegmentTemplate): {full_url}") + return full_url + + logger.debug("Could not find media segment URL in manifest") + return None + @staticmethod def extract_segment_urls(manifest_content: str, manifest_url: str) -> List[str]: """ diff --git a/lib/streaming_providers/base/utils/manifest_utils.py b/lib/streaming_providers/base/utils/manifest_utils.py index 75206b1..489760c 100644 --- a/lib/streaming_providers/base/utils/manifest_utils.py +++ b/lib/streaming_providers/base/utils/manifest_utils.py @@ -128,6 +128,76 @@ class ManifestUtils: ) return match.group(1) if match else None + @staticmethod + def _find_segment_template_tag(ad_set_content: str, attribute: str) -> Optional[str]: + """Return the first opening tag that carries `attribute`.""" + for match in re.finditer(r"]*>", ad_set_content, re.IGNORECASE): + if re.search(rf'\b{attribute}="', match.group(0)): + return match.group(0) + return None + + @staticmethod + def extract_segment_template_media(ad_set_content: str) -> Optional[str]: + """ + Extract media attribute from SegmentTemplate. + + Mirrors extract_segment_template_initialization(). + + Args: + ad_set_content: AdaptationSet XML content + + Returns: + Media template string or None if not found + """ + tag = ManifestUtils._find_segment_template_tag(ad_set_content, "media") + if not tag: + return None + match = re.search(r'\bmedia="([^"]+)"', tag) + return match.group(1) if match else None + + @staticmethod + def extract_segment_template_start_number(ad_set_content: str) -> Optional[str]: + """ + Extract startNumber from the SegmentTemplate that carries the media template. + + Returns: + startNumber as string, or None if the attribute is absent + (DASH default is 1, the caller decides). + """ + tag = ManifestUtils._find_segment_template_tag(ad_set_content, "media") + if not tag: + return None + match = re.search(r'\bstartNumber="(\d+)"', tag) + return match.group(1) if match else None + + @staticmethod + def extract_first_segment_time(ad_set_content: str) -> Optional[str]: + """ + Extract the t attribute of the FIRST element of a SegmentTimeline. + + The attribute is read from that one tag only (attribute order is free + in XML, and a first without t must not pick up a later one's t). + + Returns: + Start time as string, or None if there is no timeline or the + first has no explicit t. + """ + timeline = re.search( + r"]*>\s*(]*>)", + ad_set_content, + re.IGNORECASE, + ) + if not timeline: + return None + match = re.search(r'\bt="(\d+)"', timeline.group(1)) + return match.group(1) if match else None + + @staticmethod + def extract_first_representation_bandwidth(ad_set_content: str) -> Optional[str]: + """Extract the bandwidth of the first Representation in an AdaptationSet.""" + match = re.search(r']*\bbandwidth="(\d+)"', ad_set_content) + return match.group(1) if match else None + @staticmethod def extract_base_urls(manifest_content: str) -> List[str]: """ diff --git a/lib/streaming_providers/base/utils/mp4_pssh_extractor.py b/lib/streaming_providers/base/utils/mp4_pssh_extractor.py index 495db2f..c570fb9 100644 --- a/lib/streaming_providers/base/utils/mp4_pssh_extractor.py +++ b/lib/streaming_providers/base/utils/mp4_pssh_extractor.py @@ -97,11 +97,17 @@ class MP4PSSHExtractor: box_type = data[offset + 4: offset + 8] if box_type == b"moov": - # Look for PSSH in moov container + # Look for PSSH in moov container (init segments) moov_data = data[offset: offset + box_size] pssh_in_moov = MP4PSSHExtractor._extract_from_moov(moov_data) pssh_data_list.extend(pssh_in_moov) + elif box_type == b"moof": + # Look for PSSH in moof container (media segments, e.g. Allente) + moof_data = data[offset: offset + box_size] + pssh_in_moof = MP4PSSHExtractor._extract_from_moof(moof_data) + pssh_data_list.extend(pssh_in_moof) + elif box_type == b"pssh": # Found standalone PSSH box pssh_box = MP4PSSHExtractor._parse_pssh_box( @@ -254,6 +260,50 @@ class MP4PSSHExtractor: return pssh_list + @staticmethod + def _extract_from_moof(moof_data: bytes) -> List[PSSHData]: + """ + Extract PSSH boxes from a moof (movie fragment) box. + + Media segments carry moof where init segments carry moov. Per + ISO/IEC 23001-7 a pssh box may live in moov or moof (as a sibling of + mfhd/traf), so only the direct children of moof are inspected. + + Args: + moof_data: Raw moof box data + + Returns: + List of PSSHData objects found in moof + """ + pssh_list = [] + offset = 8 # Skip moof header + + while offset < len(moof_data): + try: + if offset + 8 > len(moof_data): + break + + box_size = struct.unpack(">I", moof_data[offset: offset + 4])[0] + box_type = moof_data[offset + 4: offset + 8] + + if box_size < 8 or offset + box_size > len(moof_data): + break + + if box_type == b"pssh": + pssh_box = MP4PSSHExtractor._parse_pssh_box( + moof_data[offset: offset + box_size] + ) + if pssh_box: + pssh_list.append(pssh_box) + + offset += box_size + + except Exception as e: + logger.debug(f"Error parsing moof box at offset {offset}: {e}") + break + + return pssh_list + @staticmethod def _extract_from_trak(trak_data: bytes) -> List[PSSHData]: """Extract PSSH from trak box"""