diff --git a/lib/streaming_providers/base/drm_operations.py b/lib/streaming_providers/base/drm_operations.py index 5e8815c..13346b7 100644 --- a/lib/streaming_providers/base/drm_operations.py +++ b/lib/streaming_providers/base/drm_operations.py @@ -20,7 +20,7 @@ and get_catchup_content_drm_configs): degrade gracefully rather than crashing the request 3. verified-clear short-circuit 4. generic plugin phase (with stub-PSSH upgrade via init segment, then - first media segment for providers that put the pssh in the moof) + a media segment for providers that put the pssh in the moof) 5. provider DRM configs (generics become the base list if provider has none) 6. system-specific plugin loop with incremental ClearKey coverage checks 7. final composition: generic merge, ClearKey validation, reinstatement @@ -456,7 +456,7 @@ class DRMOperations: still valid even if THIS call's parse failed) 2. The list parsed during manifest analysis (populates the cache) 3. Full extraction: manifest (re-)fetch + init-segment fallback + - first-media-segment fallback + media-segment fallback Note: the cache key includes catchup start/end times, so each timeshift window gets its own entry even though PSSH is likely identical per @@ -587,7 +587,7 @@ class DRMOperations: """Extract PSSH data from a manifest, falling back to segments. Levels, each tried only while the previous one left a PSSH unresolved: - 1. manifest, 2. init segment (moov), 3. first media segment (moof). + 1. manifest, 2. init segment (moov), 3. a media segment (moof). manifest_headers now defaults to None so the legacy facade call (which passes only the URL) works. pssh_list lets callers that already @@ -654,12 +654,12 @@ class DRMOperations: if segment_pssh: pssh_list = DRMExtractor._merge_pssh_data(pssh_list, segment_pssh) - # Level 3: first media segment (pssh in moof). Only while a + # Level 3: media segment (pssh in moof). Only while a # system that should carry a PSSH is still unresolved, so # providers whose init segment works never pay for this and # ClearKey stubs (legitimately PSSH-less) don't trigger it. if self._has_unresolved_pssh(pssh_list): - media_segment_url = ManifestParser.extract_first_media_segment_url( + media_segment_url = ManifestParser.extract_media_segment_url( manifest_content, manifest_url ) if media_segment_url: diff --git a/lib/streaming_providers/base/utils/manifest_parser.py b/lib/streaming_providers/base/utils/manifest_parser.py index 28f078d..ec12c79 100644 --- a/lib/streaming_providers/base/utils/manifest_parser.py +++ b/lib/streaming_providers/base/utils/manifest_parser.py @@ -1,6 +1,6 @@ # streaming_providers/base/utils/manifest_parser.py """ -DASH manifest parser for extracting init and first-media segment URLs. +DASH manifest parser for extracting init and media segment URLs. For PSSH/DRM extraction, use drm_extractor module. """ @@ -158,23 +158,28 @@ class ManifestParser: return None @staticmethod - def extract_first_media_segment_url( + def extract_media_segment_url( manifest_content: str, manifest_url: str ) -> Optional[str]: """ - Extract the URL of the FIRST media segment from a DASH manifest. + Extract the URL of ONE media segment from a DASH manifest. Counterpart to extract_single_init_segment_url(), used as a PSSH fallback for providers that put the pssh box in the moof of each media segment instead of the manifest or the init segment. + The segment is taken from the MIDDLE of the SegmentTimeline, not the + first entry: in a live manifest the first entry sits at the very edge + of the time-shift window and is typically evicted (404) by the time it + is requested. + Only SegmentTemplate manifests are supported (SegmentBase has no separate media segments). Template variables are resolved as follows: $RepresentationID$ first Representation ID of the AdaptationSet $Bandwidth$ bandwidth of the first Representation - $Time$ t of the first of the SegmentTimeline (else 0) - $Number$ startNumber of the SegmentTemplate (else 1) + $Time$ start time of the chosen SegmentTimeline entry (else 0) + $Number$ startNumber + index of the chosen entry (else startNumber, default 1) Templates with format specifiers (e.g. $Number%05d$) are not resolved and are skipped rather than requested with a broken URL. @@ -183,7 +188,7 @@ class ManifestParser: manifest_url: URL where the manifest was fetched from Returns: - Full URL to the first media segment, or None if not found + Full URL to the chosen media segment, or None if not found """ base_urls = ManifestUtils.extract_base_urls(manifest_content) effective_base = URLResolver.build_effective_base_url(manifest_url, base_urls) @@ -209,18 +214,21 @@ class ManifestParser: bandwidth = ManifestUtils.extract_first_representation_bandwidth( ad_set_info.content ) or "0" - first_time = ManifestUtils.extract_first_segment_time(ad_set_info.content) or "0" - start_number = ( + segment_time, segment_index = ( + ManifestUtils.extract_segment_timeline_position(ad_set_info.content) + or (0, 0) + ) + start_number = int( ManifestUtils.extract_segment_template_start_number(ad_set_info.content) - or "1" + or 1 ) media_url = URLResolver.substitute_template_variables( media_template, representation_id=rep_id, bandwidth=bandwidth, - time=first_time, - number=start_number + time=str(segment_time), + number=str(start_number + segment_index) ) if "$" in media_url: diff --git a/lib/streaming_providers/base/utils/manifest_utils.py b/lib/streaming_providers/base/utils/manifest_utils.py index 489760c..faf3ead 100644 --- a/lib/streaming_providers/base/utils/manifest_utils.py +++ b/lib/streaming_providers/base/utils/manifest_utils.py @@ -171,26 +171,59 @@ class ManifestUtils: return match.group(1) if match else None @staticmethod - def extract_first_segment_time(ad_set_content: str) -> Optional[str]: + def extract_segment_timeline_position( + ad_set_content: str, + position: float = 0.5, + ) -> Optional[Tuple[int, int]]: """ - Extract the t attribute of the FIRST element of a SegmentTimeline. + Locate a segment inside a SegmentTimeline. - The attribute is read from that one tag only (attribute order is free - in XML, and a first without t must not pick up a later one's t). + Walks the entries (r = repeat count, t optional and + continuing from the previous entry) and returns the start time and the + zero-based index of the segment at `position` (0.0 = first, 0.5 = middle, + 1.0 = last). Returns: - Start time as string, or None if there is no timeline or the - first has no explicit t. + (start_time_ticks, index), or None if there is no usable timeline + (absent, an without d, or an open-ended r="-1"). """ timeline = re.search( - r"]*>\s*(]*>)", + r"]*>(.*?)", ad_set_content, - re.IGNORECASE, + re.IGNORECASE | re.DOTALL, ) if not timeline: return None - match = re.search(r'\bt="(\d+)"', timeline.group(1)) - return match.group(1) if match else None + + entries = [] # (explicit t or None, duration, segment count) + for tag in re.finditer(r"]*>", timeline.group(1), re.IGNORECASE): + attrs = tag.group(0) + d = re.search(r'\bd="(\d+)"', attrs) + if not d: + return None + t = re.search(r'\bt="(\d+)"', attrs) + r = re.search(r'\br="(-?\d+)"', attrs) + repeat = int(r.group(1)) if r else 0 + if repeat < 0: + return None # open-ended repeat: segment count unknown + entries.append((int(t.group(1)) if t else None, int(d.group(1)), repeat + 1)) + + if not entries: + return None + + total = sum(count for _, _, count in entries) + target = min(total - 1, max(0, int(total * position))) + + time = 0 + index = 0 + for explicit_t, duration, count in entries: + if explicit_t is not None: + time = explicit_t + if target < index + count: + return time + (target - index) * duration, target + time += count * duration + index += count + return None @staticmethod def extract_first_representation_bandwidth(ad_set_content: str) -> Optional[str]: diff --git a/service.py b/service.py index 2d8812e..0452c04 100644 --- a/service.py +++ b/service.py @@ -398,6 +398,11 @@ class UltimateService: return manifest_response.text, ttl, provider_proxy_url, segment_headers, manifest_response.url def _make_kid_resolver(self, provider: str, segment_headers: Optional[dict]): + """ + Build the init-URL -> KID callable for MPDRewriter (tenc lookup when the + MPD carries no KID in multi-key mode). The resolver and its cache are + process-wide; headers and HTTP manager are bound per provider here. + """ resolver = get_init_kid_resolver() http_manager = self.manager.get_provider_http_manager(provider) return lambda init_url: resolver.resolve( @@ -589,7 +594,7 @@ class UltimateService: self.media_proxy_url, provider_proxy_url, keyids, highest_quality_only, provider=provider, channel=channel_id, clearkey_receiver_side=receiver_side, segment_headers=segment_headers, - id_resolver=self._make_kid_resolver(provider, segment_headers), + kid_resolver=self._make_kid_resolver(provider, segment_headers), ) rewritten_mpd = rewriter.rewrite_mpd(manifest_text, effective_url) return rewritten_mpd, min(ttl, 10) # holds key material — keep exposure window short @@ -645,7 +650,7 @@ class UltimateService: self.media_proxy_url, provider_proxy_url, keyids, highest_quality_only, provider=provider, channel=channel_id, clearkey_receiver_side=receiver_side, segment_headers=segment_headers, - id_resolver=self._make_kid_resolver(provider, segment_headers), + kid_resolver=self._make_kid_resolver(provider, segment_headers), ) rewritten_mpd = rewriter.rewrite_mpd(manifest_text, effective_url) return rewritten_mpd, min(ttl, 30)