From ecb2d7d3707b12fe5fa214ef6e926521ad17a44d Mon Sep 17 00:00:00 2001 From: PiratesIRC <98669745+PiratesIRC@users.noreply.github.com> Date: Sun, 6 Sep 2026 10:51:22 -0500 Subject: [PATCH 1/3] feat(epg): scan for placeholder name families no pattern covers (#43) Adds the action Scan for Placeholder Patterns. It groups every stream name into a numbered family by replacing its numbers with a slot, and reports the families that no configured Placeholder Name Pattern covers, each with an anchored regular expression to paste. Requested as issue #43. The setting only ever helped with the naming schemes the operator already thought to write down, and nothing in the interface told apart "this installation has no placeholder families" from "the patterns you wrote match none of them". The grouping lives in a new stdlib-only module, Stream-Mapparr/placeholder_scan.py, with no Django import, so it is tested directly. The action is the thin wrapper that loads the streams and writes the readout to /config/stream-mapparr/placeholder-name-scan.txt. Measured against 25,068 live stream names, and three decisions came from that rather than from assumption: A digit immediately followed by K is left alone, because 4K and 8K are resolution tags. With the rule off, 15 further templates covering 418 streams group by their resolution tag instead of by a slot number. A family needs three different numbers in its slot, not three streams. A separate stream-count threshold was written and removed: three distinct numbers already implies three streams, so it could never refuse anything the first rule admitted, and no test could show it doing work. Families whose streams carry EPG data are reported first and separately. On this installation 131 families are uncovered and only 16 hold a stream with an EPG identifier, so one undifferentiated list would report a problem eight times larger than the one worth acting on. The readout is plain ASCII with any other character written as a backslash-u escape, the same rule as the CSV export preamble. Provider names here really do carry such characters, and an earlier version of the ASCII test used only synthetic names, so it passed while the live report broke the rule. Python regular expressions accept that form; all 132 suggestions generated from live data were checked against their own example name. Every guard was proven by mutation rather than by reading: each was broken in turn and a named test confirmed to fail, with a comment-only control leaving the suite green. That found one vacuous test, which sized its own input from the threshold it was checking and so moved with it. Reports only. No setting is changed and nothing is written to the database. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017ZBufwmvsXqzkq2h2VkAwB --- CHANGELOG.md | 52 ++++ README.md | 3 +- Stream-Mapparr/__init__.py | 2 +- Stream-Mapparr/placeholder_scan.py | 310 ++++++++++++++++++++++++ Stream-Mapparr/plugin.json | 10 +- Stream-Mapparr/plugin.py | 103 +++++++- tests/test_action_buttons.py | 1 + tests/test_action_order.py | 1 + tests/test_placeholder_scan.py | 369 +++++++++++++++++++++++++++++ 9 files changed, 847 insertions(+), 4 deletions(-) create mode 100644 Stream-Mapparr/placeholder_scan.py create mode 100644 tests/test_placeholder_scan.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 0f96aa5..89f5f52 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,57 @@ # Stream-Mapparr CHANGELOG +## 1.26.2491549 (2026-09-06) + +### Added + +- **New action, Scan for Placeholder Patterns, which finds the numbered stream + name families your Placeholder Name Patterns do not cover.** Requested as + issue #43. The setting only ever helped with the naming schemes you already + thought to write down, and nothing in the interface told apart "this + installation has no placeholder families" from "the patterns you wrote match + none of them". The reporter found two whole uncovered families, one of them + their largest, only by pulling every stream name through the API by hand. + + The scan replaces the numbers in every stream name with a slot, so MAX 100 and + MAX 101 become the one family MAX #, and reports the families that no + configured pattern covers, each with an anchored regular expression to paste. + It reads one database column that matching already loads, opens no provider + connection, changes no setting and writes nothing to the database. The full + readout goes to `/config/stream-mapparr/placeholder-name-scan.txt`, because a + notification shows only about 280 characters. + + Three decisions in it were measured rather than assumed, against 25,068 live + stream names: + + A digit immediately followed by K is left alone, because `4K` and `8K` are + resolution tags rather than slot numbers. With that rule off, 15 further + templates covering 418 streams are grouped by their resolution tag instead of + by a slot number. + + A family needs at least three different numbers in its slot, not merely three + streams. Five rows of `HBO 1` from five sources are five sources for one name, + not five slots. A separate stream-count threshold was written and then removed, + because three distinct numbers already implies three streams, so the second + rule could never refuse anything the first admitted. + + Families whose streams carry EPG data are reported first and separately from + those carrying none, because a placeholder can only ever be resolved when + there is guide data to resolve it from. This matters more than expected: on + this installation 131 families are uncovered and only 16 hold a stream with an + EPG identifier, so a single undifferentiated list would report a problem eight + times larger than the one worth acting on. The detailed list is capped, and + says how many it left out rather than cutting silently. + + The readout is plain ASCII, the same rule as the CSV export preamble, with any + other character written as a backslash-u escape. Provider names really do carry + such characters here. Python regular expressions accept that form, so every + suggested pattern still matches the name it came from; all 132 suggestions + generated from live data were checked against their own example name. + + This reports only. Nothing is ever added to your pattern list, and not every + numbered family is a placeholder: a numbered channel family whose names are + already informative matches better as it is. + ## v1.26.2481756 (September 5, 2026) ### Fixed diff --git a/README.md b/README.md index c92cfc8..41c485d 100644 --- a/README.md +++ b/README.md @@ -200,7 +200,7 @@ the operation lock prevents concurrent runs and auto-expires after 10 minutes. | **Stream Prefix Countries** | string | (empty) | Tell the country filter what a provider prefix means, as comma-separated `PREFIX=COUNTRY` entries such as `NOW=UK, GO=US`. Use it when a prefix names a platform rather than a country, which the plugin cannot know: NOW is Sky's service in the United Kingdom and also in Italy, so no default is right for everyone. Consulted last, so it fills a gap and never overrules a country the provider stated. Matches only at the start of a name, never a word inside a title | | **Keep Same-Named Streams From One Source** | boolean | False | Enable if your provider publishes several genuinely different feeds under one identical name. By default those are treated as duplicates | | **Enable EPG-Based Placeholder Matching** | boolean | False | Match a placeholder-named channel or stream by the programme currently airing on it, taken from EPG data, instead of by its literal name. For providers that name event slots generically, such as `PPV EVENT 04`, and put the real event only in the guide. Channel and stream names are never modified | -| **Placeholder Name Patterns** | string | (see plugin) | One regex per line. A name is only ever treated as a placeholder if it matches one of these, so nothing else changes behaviour | +| **Placeholder Name Patterns** | string | (see plugin) | One regex per line. A name is only ever treated as a placeholder if it matches one of these, so nothing else changes behaviour. Press **Scan for Placeholder Patterns** to find the families your patterns miss | | **EPG Title Cleanup Rules** | string | (see plugin) | JSON list of `[find, replace]` pairs applied to the raw programme title before it is used for matching, for example stripping a `Next Event: X at 6:00AM` wrapper down to `X` | | **Skip Titles** | string | (see plugin) | Comma-separated. If the cleaned programme title matches one of these, the channel keeps its literal name for that pass, because an idle slot carries no useful event | | **Channel Schedule Suffix Cleanup Rules** | string | (see plugin) | JSON list of `[find, replace]` pairs that strip a schedule annotation such as `\| Monday @ 5` from the channel name before it is compared against a programme title | @@ -228,6 +228,7 @@ the operation lock prevents concurrent runs and auto-expires after 10 minutes. | **Validate Settings** | Check configuration, profiles, groups and databases | | **Test Regex Rules** | Preview what your regex rules would change, with before and after samples and invisible characters made visible. Writes the full readout to `/config/stream-mapparr/test-regex-rules.txt`, since a notification shows only about 280 characters | | **Check Stream Country Labels** | Compare each stream's group country against its EPG identifier suffix and report where they disagree. Reads two database columns, opens no provider connection and changes nothing. A disagreement is not automatically a fault: a channel carried in one country and made in another is ordinary | +| **Scan for Placeholder Patterns** | Group every stream name into a numbered family, by replacing its numbers with a slot, and report the families that no Placeholder Name Pattern covers, each with a regex to paste. Families whose streams carry EPG data are listed first, because a placeholder can only be resolved when there is guide data to resolve it from. A digit followed by K is left alone, since `4K` is a resolution tag rather than a slot number. Writes the full readout to `/config/stream-mapparr/placeholder-name-scan.txt`. Reads one database column, opens no provider connection and changes nothing | | **Load/Process Channels** | Load channel and stream data from the database | | **Preview Changes** | Dry run with a CSV export | | **Match & Assign Streams** | Fuzzy match and assign streams to channels | diff --git a/Stream-Mapparr/__init__.py b/Stream-Mapparr/__init__.py index a2107d5..798681c 100644 --- a/Stream-Mapparr/__init__.py +++ b/Stream-Mapparr/__init__.py @@ -5,5 +5,5 @@ from .plugin import Plugin -__version__ = "1.26.2481756" +__version__ = "1.26.2491549" __all__ = ["Plugin"] \ No newline at end of file diff --git a/Stream-Mapparr/placeholder_scan.py b/Stream-Mapparr/placeholder_scan.py new file mode 100644 index 0000000..df20b66 --- /dev/null +++ b/Stream-Mapparr/placeholder_scan.py @@ -0,0 +1,310 @@ +"""Find numbered stream-name families that no placeholder pattern covers. + +GitHub issue #43. The setting `epg_placeholder_name_patterns` only ever helps +with the naming schemes the operator already thought to write down, and nothing +in the interface separates "this installation has no placeholder families" from +"the patterns you wrote match none of them". The reporter discovered two whole +uncovered families, one of them their largest, only by pulling every stream +name through the API by hand and grouping it. + +This module is the grouping, as a pure stdlib unit with no Django and no +plugin import, so it can be tested directly. The plugin action is the thin +wrapper that loads the streams and writes the readout. + +TWO THINGS ARE BUILT IN RATHER THAN LEFT TO BE DISCOVERED LATER, both measured +on a live installation of 25,323 stream names before any of this was written: + + A digit immediately followed by K is a resolution tag, not a slot number. + Replacing it turns `4K` and `8K` into `#K`, which merged 293 unrelated names + into a single false family driven entirely by the resolution tag. + + Not every numbered family is a placeholder. `UK: BBC RED BUTTON #`, + `US: HULU ORIGINALS #` and `UK: KARAOKE #` are numbered channel families + whose names are perfectly informative, and adding a placeholder pattern for + them would make matching worse. A placeholder can only ever be RESOLVED if + its streams carry EPG data, so families are ranked by how many of their + members carry an EPG identifier and the merely-numbered ones sink. + +This reports. It never edits the setting, and nothing here writes to the +database. +""" + +# A family needs at least this many DIFFERENT numbers in its slot before it is +# worth reporting. Counting streams instead would let one stream name carried by +# several M3U accounts look like a numbered family: five rows of `HBO 1` are +# five sources for one name, not five slots. A count threshold as well was +# tried and removed, because a family with three distinct numbers already has +# at least three streams, so the second rule could never refuse anything the +# first one admitted and no test could tell whether it was doing any work. +MIN_DISTINCT_NUMBERS = 3 + +# How many uncovered families are described in full, with a pattern to paste. +# MEASURED on 25,068 live stream names: 131 families are uncovered. Describing +# every one produces the long, mostly unactionable readout this feature exists +# to replace, so the rest are listed as one line each and the count of what was +# left out is stated rather than the cut being silent. +DETAIL_LIMIT = 25 + +# The slot marker in a template. A name containing a literal one is stored +# with it backslash-escaped so the two can never be confused. +SLOT = "#" + +# Characters that need escaping to appear literally in the suggested regex. +# Deliberately NOT re.escape: since Python 3.7 that also escapes the space and +# the hash, which would turn a readable, pasteable suggestion such as +# ^Triller TV \| Event \d+$ into an unreadable one. +_REGEX_SPECIALS = set(r"()[]{}?*+-|^$\.") + + +def ascii_safe(text): + """Rewrite every non-ASCII character as a backslash-u escape. + + The readout is plain ASCII on purpose, the same rule as the CSV export + preamble: it is opened in a text editor that may be using another codepage, + where a non-ASCII byte becomes mojibake. Provider stream names really do + carry such characters. MEASURED on this installation: family templates + include a circled bullet and superscript letters, so writing names through + unescaped would have broken the rule on live data while every synthetic + test still passed. + + The escape is not merely readable, it is also correct inside a suggested + pattern: Python's regular expressions accept backslash-u, so a pattern written + this way still matches the real name. + """ + return "".join(ch if ord(ch) < 128 else "\\u%04x" % ord(ch) for ch in text) + + +def _escape_literal(text): + """Escape `text` so it matches itself inside a regular expression.""" + return "".join("\\" + ch if ch in _REGEX_SPECIALS else ch for ch in text) + + +def template_of(name): + """Split `name` into its digit-stripped template and the numbers removed. + + Returns `(template, numbers)`. `numbers` is a tuple of the digit runs in + the order they appeared, so a caller can count distinct slot values, and it + is empty when the name holds no slot at all. + + A digit run is left alone, and so contributes no slot, when it starts at a + word boundary and is immediately followed by the letter K. That is a + resolution tag (`4K`, `8k`), not a numbered slot. The boundary is required: + in `X264K` the run is glued to a letter on the left, so it is a slot and + the template is `X#K`. + """ + if not name: + return ("", ()) + parts = [] + numbers = [] + index = 0 + length = len(name) + while index < length: + char = name[index] + if not char.isdigit(): + parts.append("\\" + SLOT if char == SLOT else char) + index += 1 + continue + start = index + while index < length and name[index].isdigit(): + index += 1 + run = name[start:index] + at_boundary = start == 0 or not name[start - 1].isalnum() + followed_by_k = index < length and name[index] in ("k", "K") + if at_boundary and followed_by_k: + parts.append(run) + continue + parts.append(SLOT) + numbers.append(run) + return ("".join(parts), tuple(numbers)) + + +def suggested_pattern(template): + """Turn a template into an anchored regex the operator can paste. + + Every slot becomes `\\d+`, an escaped literal hash becomes a literal hash, + and everything else is escaped so it matches itself. Anchored at both ends + because the plugin decides placeholder eligibility with `fullmatch` over + the whole name. + """ + out = [] + literal = [] + index = 0 + length = len(template) + while index < length: + char = template[index] + if char == "\\" and index + 1 < length and template[index + 1] == SLOT: + literal.append(SLOT) + index += 2 + continue + if char == SLOT: + out.append(_escape_literal("".join(literal))) + literal = [] + out.append(r"\d+") + index += 1 + continue + literal.append(char) + index += 1 + out.append(_escape_literal("".join(literal))) + return ascii_safe("^" + "".join(out) + "$") + + +def _has_epg_identifier(stream): + return bool((stream.get("tvg_id") or "").strip()) + + +def scan_families(streams, patterns): + """Group `streams` by template and describe every numbered family found. + + `patterns` is the list of compiled placeholder patterns already configured. + Coverage is decided with `fullmatch`, the same way the matcher itself + decides whether a name is a placeholder, so an unanchored pattern that + would merely search-match does not count as covering anything. + + Returns a list of dicts, best candidate first, ranked by how many members + carry an EPG identifier and then by size. A family is reported whether or + not it is covered; the caller decides what to show. + """ + grouped = {} + for stream in streams: + name = stream.get("name") or "" + if not name: + continue + template, numbers = template_of(name) + if not numbers: + continue + family = grouped.setdefault(template, { + "template": template, + "count": 0, + "numbers": set(), + "example": name, + "with_epg_id": 0, + "covered": 0, + }) + family["count"] += 1 + family["numbers"].add(numbers) + if name < family["example"]: + family["example"] = name + if _has_epg_identifier(stream): + family["with_epg_id"] += 1 + if patterns and any(p.fullmatch(name) for p in patterns): + family["covered"] += 1 + + families = [] + for family in grouped.values(): + distinct = len(family["numbers"]) + if distinct < MIN_DISTINCT_NUMBERS: + continue + families.append({ + "template": family["template"], + "count": family["count"], + "distinct_numbers": distinct, + "example": family["example"], + "with_epg_id": family["with_epg_id"], + "covered": family["covered"], + "uncovered": family["count"] - family["covered"], + "suggested": suggested_pattern(family["template"]), + }) + families.sort(key=lambda f: (-f["with_epg_id"], -f["count"], f["template"])) + return families + + +def render_report(families, total_streams, pattern_count, feature_enabled): + """The operator-facing readout, plain ASCII. + + ASCII on purpose, the same rule as the CSV export preamble: this file is + opened in a text editor that may be using another codepage, and a non-ASCII + character there becomes mojibake. + """ + uncovered = [f for f in families if f["uncovered"] > 0] + covered = [f for f in families if f["uncovered"] == 0] + + lines = [ + "Stream-Mapparr: placeholder name family scan", + "============================================", + "", + "Every stream name is grouped by replacing its numbers with a slot, so", + "MAX 100 and MAX 101 become the one family MAX #. A family is reported", + f"when at least {MIN_DISTINCT_NUMBERS} different numbers appear in its slot,", + "which also means at least that many streams. Counting streams alone", + "would read one name carried by several sources as a family.", + "", + "A character outside plain ASCII is written as a backslash-u escape. Python", + "regular expressions accept that form, so a suggested pattern still", + "matches the real name.", + "", + "A digit followed by K is left alone, because 4K and 8K are resolution", + "tags rather than slot numbers.", + "", + "This reports only. No setting is changed and nothing is written to the", + "database. Read a suggestion before pasting it: not every numbered", + "family is a placeholder. A numbered channel family whose names are", + "already informative is matched better as it is.", + "", + f"Streams scanned : {total_streams}", + f"Numbered families found : {len(families)}", + f"Placeholder patterns configured : {pattern_count}", + f"Families no pattern covers : {len(uncovered)}", + "", + ] + + if not feature_enabled: + lines += [ + "NOTE: EPG-Based Placeholder Matching is OFF, so the pattern list is", + "never consulted and no family is covered in practice, whatever is", + "written in it. Turn the feature on for these patterns to do anything.", + "", + ] + + actionable = [f for f in uncovered if f["with_epg_id"] > 0] + no_epg = [f for f in uncovered if f["with_epg_id"] == 0] + + if uncovered: + word = "family" if len(actionable) == 1 else "families" + lines.append(f"{len(actionable)} likely placeholder {word} not covered by your " + f"current patterns") + lines.append("whose streams carry EPG data, best candidate first. The ranking is") + lines.append("how many streams in the family carry an EPG identifier, because a") + lines.append("placeholder can only ever be resolved when its streams carry EPG") + lines.append("data to resolve it from.") + lines.append("") + if not actionable: + lines.append(" None. Every uncovered family is listed below instead.") + lines.append("") + for family in actionable[:DETAIL_LIMIT]: + lines.append(" " + ascii_safe(family["template"])) + lines.append(f" streams : {family['count']}" + f" ({family['uncovered']} not covered)") + lines.append(f" different slot numbers : {family['distinct_numbers']}") + lines.append(f" carrying an EPG id : {family['with_epg_id']}") + lines.append(" example : " + ascii_safe(family["example"])) + lines.append(f" pattern to paste : {family['suggested']}") + lines.append("") + if len(actionable) > DETAIL_LIMIT: + lines.append(f" and {len(actionable) - DETAIL_LIMIT} more, not described here.") + lines.append("") + + if no_epg: + lines.append(f"{len(no_epg)} further uncovered families carry no EPG data at all.") + lines.append("A placeholder pattern for one of these could not resolve anything") + lines.append("today, so no pattern is suggested. They are listed because a family") + lines.append("can start carrying EPG data later, and because a numbered channel") + lines.append("family whose names are already informative belongs here rather than") + lines.append("in the list above.") + lines.append("") + for family in no_epg[:DETAIL_LIMIT * 4]: + lines.append(" " + ascii_safe(family["template"]) + f" ({family['count']} streams)") + if len(no_epg) > DETAIL_LIMIT * 4: + lines.append(f" and {len(no_epg) - DETAIL_LIMIT * 4} more, not listed here.") + lines.append("") + + if not uncovered: + lines.append("No uncovered numbered families were found.") + lines.append("") + + if covered: + lines.append("Families your patterns already cover in full:") + for family in covered: + lines.append(" " + ascii_safe(family["template"]) + f" ({family['count']} streams)") + lines.append("") + + return "\n".join(lines) diff --git a/Stream-Mapparr/plugin.json b/Stream-Mapparr/plugin.json index ea8069d..1894882 100644 --- a/Stream-Mapparr/plugin.json +++ b/Stream-Mapparr/plugin.json @@ -1,6 +1,6 @@ { "name": "Stream-Mapparr", - "version": "1.26.2481756", + "version": "1.26.2491549", "description": "Automatically add matching streams to channels based on name similarity and quality precedence. Supports unlimited stream matching, channel visibility management, and CSV export cleanup.", "author": "PiratesIRC", "license": "MIT", @@ -147,6 +147,14 @@ "button_color": "blue", "button_label": "🌍 Check Countries" }, + { + "id": "scan_placeholder_names", + "label": "🔍 Scan for Placeholder Patterns", + "description": "Group every stream name into a numbered family and report the families no Placeholder Name Pattern covers, with a regex to paste. Reads one database column, changes nothing", + "button_variant": "outline", + "button_color": "blue", + "button_label": "🔍 Scan Placeholders" + }, { "id": "email_report_now", "label": "📧 Email Report Now", diff --git a/Stream-Mapparr/plugin.py b/Stream-Mapparr/plugin.py index b5784d3..14b82cb 100644 --- a/Stream-Mapparr/plugin.py +++ b/Stream-Mapparr/plugin.py @@ -303,7 +303,7 @@ class PluginConfig: """ # === PLUGIN METADATA === - PLUGIN_VERSION = "1.26.2481756" + PLUGIN_VERSION = "1.26.2491549" FUZZY_MATCHER_MIN_VERSION = "25.358.0200" # Requires custom ignore tags Unicode fix # Match sensitivity presets (maps select value to threshold number) @@ -1572,6 +1572,14 @@ def fields(self): "button_color": "blue", "button_label": "🌍 Check Countries", }, + { + "id": "scan_placeholder_names", + "label": "🔍 Scan for Placeholder Patterns", + "description": "Group every stream name into a numbered family and report the families no Placeholder Name Pattern covers, with a regex to paste. Reads one database column, changes nothing", + "button_variant": "outline", + "button_color": "blue", + "button_label": "🔍 Scan Placeholders", + }, { "id": "clear_csv_exports", "label": "🗑️ Clear CSV Exports", @@ -2589,6 +2597,98 @@ def check_stream_countries_action(self, settings, logger, context=None): result["file"] = path return result + @staticmethod + def _placeholder_scan(): + """The pure grouping module behind the placeholder family scan.""" + try: + from . import placeholder_scan + except ImportError: + import placeholder_scan + return placeholder_scan + + def scan_placeholder_names_action(self, settings, logger, context=None): + """Report numbered stream-name families no placeholder pattern covers. + + GitHub issue #43. The Placeholder Name Patterns setting only helps with + the naming schemes the operator already thought of, and nothing in the + interface tells apart "this installation has no placeholder families" + from "the patterns written here match none of them". The reporter found + two whole uncovered families, one of them their largest, only by pulling + every stream name through the API by hand. + + Reads one database column that matching already loads, opens no + provider connection, changes no setting and writes no channel data. + + The patterns are resolved WITHOUT the feature toggle gating them, unlike + _resolve_epg_matching_settings, because a pattern list is worth checking + for coverage whether or not the feature happens to be switched on. The + readout says plainly when it is off, since a list that is never + consulted covers nothing in practice. + """ + scan = self._placeholder_scan() + try: + streams = self._get_all_streams(logger) + except Exception as e: + logger.error(f"[Stream-Mapparr] Could not load streams: {e}") + return {"status": "error", + "error": f"Could not load the stream list ({e})."} + + settings = settings if isinstance(settings, dict) else {} + patterns = self._resolve_epg_placeholder_patterns(settings) + enabled = self._get_bool_setting( + settings, 'epg_placeholder_matching_enabled', + PluginConfig.DEFAULT_EPG_PLACEHOLDER_MATCHING_ENABLED) + + families = scan.scan_families(streams, patterns) + text = scan.render_report(families, total_streams=len(streams), + pattern_count=len(patterns), + feature_enabled=enabled) + uncovered = [f for f in families if f["uncovered"] > 0] + + path = None + try: + os.makedirs(self.BUG_REPORT_DIR, exist_ok=True) + path = os.path.join(self.BUG_REPORT_DIR, "placeholder-name-scan.txt") + with open(path, "w", encoding="utf-8") as fh: + fh.write(text) + except Exception as e: + # A toast cannot carry the readout, so say plainly that the file is + # missing rather than reporting a success the operator cannot read. + logger.warning(f"[Stream-Mapparr] Could not write the placeholder scan: {e}") + path = None + + if not families: + parts = [f"No numbered stream-name families were found across " + f"{len(streams)} streams, so there is nothing to cover."] + elif not uncovered: + parts = [f"{len(families)} numbered families found, and your " + f"{len(patterns)} pattern(s) cover them all."] + else: + # The headline counts the families whose streams carry EPG data, + # because only those can ever be resolved. MEASURED on this + # installation: 131 families are uncovered and 16 of them hold a + # stream with an EPG identifier, so counting all of them would + # report a problem eight times larger than the one worth acting on. + actionable = [f for f in uncovered if f["with_epg_id"] > 0] + word = "family" if len(actionable) == 1 else "families" + parts = [ + f"{len(actionable)} uncovered placeholder {word} carrying EPG data, " + f"out of {len(uncovered)} uncovered in total.", + ] + if actionable: + top = actionable[0] + parts.append(f"Best candidate: {top['template']} " + f"({top['count']} streams, " + f"{top['with_epg_id']} with an EPG id).") + parts.append(f"Pattern to paste: {top['suggested']}") + if not enabled: + parts.append("EPG-Based Placeholder Matching is currently off.") + + result = {"status": "success", "message": self._fit_toast(parts)} + if path: + result["file"] = path + return result + def _own_csv_exports(self): """Full paths of the CSV exports THIS plugin wrote, newest first not implied. @@ -6620,6 +6720,7 @@ def run(self, action, settings, context=None): "view_last_results": self.view_last_results_action, "test_regex_rules": self.test_regex_rules_action, "check_stream_countries": self.check_stream_countries_action, + "scan_placeholder_names": self.scan_placeholder_names_action, } if action in background_actions: diff --git a/tests/test_action_buttons.py b/tests/test_action_buttons.py index d901b3d..dca8992 100644 --- a/tests/test_action_buttons.py +++ b/tests/test_action_buttons.py @@ -68,6 +68,7 @@ "view_last_results": "blue", "test_regex_rules": "blue", "check_stream_countries": "blue", + "scan_placeholder_names": "blue", } diff --git a/tests/test_action_order.py b/tests/test_action_order.py index bf95eb5..949fd27 100644 --- a/tests/test_action_order.py +++ b/tests/test_action_order.py @@ -44,6 +44,7 @@ def _ids(plugin_module): # Tools and recovery "test_regex_rules", "check_stream_countries", + "scan_placeholder_names", "clear_csv_exports", "clear_operation_lock", "report_a_bug", diff --git a/tests/test_placeholder_scan.py b/tests/test_placeholder_scan.py new file mode 100644 index 0000000..d386de7 --- /dev/null +++ b/tests/test_placeholder_scan.py @@ -0,0 +1,369 @@ +"""The uncovered-placeholder scan, GitHub issue #43. + +`epg_placeholder_name_patterns` only helps with the naming schemes the +operator already thought of, and nothing in the interface separates "no +placeholders here" from "your patterns match nothing". The reporter found two +whole families (`MAX #` at 128 streams, the largest they had) only by pulling +every stream name through the API by hand. + +The scan groups every stream name by a digit-stripped template and reports the +recurring numbered families that no configured pattern covers. Two findings +from testing the idea against 25,323 local names are built in rather than +discovered later: + + A digit immediately followed by K is a resolution tag, not a slot number. + Stripping it turned `4K` and `8K` into `#K` and merged 293 unrelated names + into one false family. + + Not every numbered family is a placeholder (`UK: BBC RED BUTTON #`, + `US: HULU ORIGINALS #`). A placeholder can only be resolved if its streams + carry EPG data, so candidates are ranked by how many members carry an EPG + identifier, which pushes the merely-numbered families down. + +Report only. Nothing is added to the setting. +""" +import re +import string + +from placeholder_scan import ( + DETAIL_LIMIT, MIN_DISTINCT_NUMBERS, template_of, suggested_pattern, + scan_families, render_report, +) + + +def _pat(*patterns): + return [re.compile(p, re.IGNORECASE) for p in patterns] + + +# --------------------------------------------------------------------------- # +# The template +# --------------------------------------------------------------------------- # + +def test_a_digit_run_becomes_one_slot(): + assert template_of("MAX 100") == ("MAX #", ("100",)) + assert template_of("Triller TV | Event 07") == ("Triller TV | Event #", ("07",)) + + +def test_every_digit_run_is_a_slot_and_the_numbers_come_back_in_order(): + assert template_of("LIVE EVENT 04 - 11am") == ("LIVE EVENT # - #am", ("04", "11")) + + +def test_a_resolution_tag_is_not_a_slot(): + """`4K` and `8K` are resolution tags. Stripping them merged 293 unrelated + names under one template on a real installation.""" + assert template_of("US: CINEMANIA 4K") == ("US: CINEMANIA 4K", ()) + assert template_of("PPV 12 | 4K") == ("PPV # | 4K", ("12",)) + assert template_of("Movie 8k") == ("Movie 8k", ()) + + +def test_a_resolution_tag_needs_a_boundary_before_the_digit(): + """Only a digit run that STARTS at a word boundary and is immediately + followed by K is left alone; a run glued to letters on the left is a slot.""" + assert template_of("X264K") == ("X#K", ("264",)) + + +def test_a_name_with_no_digits_has_no_slot(): + assert template_of("CNN") == ("CNN", ()) + + +def test_the_template_is_built_from_the_exact_name_not_a_normalised_one(): + """Case and spacing are kept so the suggested regex matches the real names.""" + assert template_of(" Max 100") == (" Max #", ("100",)) + + +# --------------------------------------------------------------------------- # +# The suggested regex round-trips +# --------------------------------------------------------------------------- # + +def test_the_suggested_pattern_matches_every_member_and_is_anchored(): + pattern = suggested_pattern("Triller TV | Event #") + assert pattern == r"^Triller TV \| Event \d+$" + compiled = re.compile(pattern, re.IGNORECASE) + assert compiled.fullmatch("Triller TV | Event 07") + assert compiled.fullmatch("triller tv | event 12") + assert not compiled.fullmatch("Triller TV | Event 07 HD") + + +def test_a_kept_resolution_tag_survives_in_the_suggested_pattern(): + pattern = suggested_pattern("PPV # | 4K") + assert pattern == r"^PPV \d+ \| 4K$" + assert re.compile(pattern).fullmatch("PPV 12 | 4K") + + +def test_a_literal_hash_in_a_name_is_escaped_not_read_as_a_slot(): + """`#` is the slot marker in a template. A name that really contains one + is stored with it escaped so it cannot be mistaken for a slot.""" + template, numbers = template_of("Event #7") + assert numbers == ("7",) + assert template != "Event ##" + compiled = re.compile(suggested_pattern(template), re.IGNORECASE) + assert compiled.fullmatch("Event #7") + assert compiled.fullmatch("Event #12") + assert not compiled.fullmatch("Event 7") + + +# --------------------------------------------------------------------------- # +# Which families are candidates +# --------------------------------------------------------------------------- # + +def _streams(template, numbers, tvg="", start_id=1): + out = [] + for i, n in enumerate(numbers): + out.append({"id": start_id + i, + "name": template.replace("#", str(n)), + "tvg_id": tvg}) + return out + + +def test_a_recurring_numbered_family_is_a_candidate(): + streams = _streams("MAX #", [100, 101, 102, 103]) + families = scan_families(streams, []) + assert [f["template"] for f in families] == ["MAX #"] + fam = families[0] + assert fam["count"] == 4 + assert fam["distinct_numbers"] == 4 + assert fam["example"] == "MAX 100" + assert fam["suggested"] == r"^MAX \d+$" + + +def test_the_threshold_is_three_distinct_numbers(): + """Written as a literal, not derived from the constant. A test that sizes + its own input from the threshold it is checking moves with the threshold + and can never fail: that is how the first version of this file let the + limit be lowered to 1 with every test still green.""" + assert MIN_DISTINCT_NUMBERS == 3 + assert scan_families(_streams("MAX #", [1, 2]), []) == [] + assert [f["template"] for f in scan_families(_streams("MAX #", [1, 2, 3]), [])] == ["MAX #"] + + +def test_many_streams_sharing_one_number_are_not_a_family(): + """Five rows of `HBO 1` are five sources for one name, not five slots. The + input is well over any plausible stream-count threshold, so this cannot + pass by accident of a size rule.""" + streams = _streams("HBO #", [1] * 9) + assert len(streams) == 9 + assert scan_families(streams, []) == [] + + +def test_a_name_with_no_slot_is_never_a_family(): + streams = [{"id": i, "name": "CNN", "tvg_id": ""} for i in range(10)] + assert scan_families(streams, []) == [] + + +def test_a_family_fully_covered_by_a_pattern_is_reported_as_covered(): + streams = _streams("PPV EVENT #", [1, 2, 3, 4]) + families = scan_families(streams, _pat(r"^PPV EVENT \d+$")) + assert families[0]["covered"] == 4 + assert families[0]["uncovered"] == 0 + + +def test_a_partly_covered_family_says_how_many_members_miss(): + """A pattern written against one suffix shape silently misses a member + with another shape. Partial coverage is the finding, so it is counted, + not rounded to covered.""" + streams = _streams("PPV # |", [1, 2, 3]) + _streams("PPV #", [4, 5, 6]) + families = scan_families(streams, _pat(r"^PPV \d+ \|$")) + by_template = {f["template"]: f for f in families} + assert by_template["PPV # |"]["uncovered"] == 0 + assert by_template["PPV #"]["uncovered"] == 3 + + +def test_coverage_uses_fullmatch_like_the_matcher_does(): + """The plugin decides eligibility with fullmatch, so an unanchored pattern + that would search-match must not count as coverage here either.""" + streams = _streams("MAX # HD", [1, 2, 3]) + families = scan_families(streams, _pat(r"MAX \d+")) + assert families[0]["uncovered"] == 3 + + +# --------------------------------------------------------------------------- # +# Ranking +# --------------------------------------------------------------------------- # + +def test_families_whose_streams_carry_epg_identifiers_rank_first(): + """`BBC RED BUTTON #` is bigger, but its streams carry no EPG identifier, so + a placeholder pattern for it could never resolve anything.""" + red_button = _streams("UK: BBC RED BUTTON #", list(range(1, 11)), tvg="") + events = _streams("PPV EVENT #", [1, 2, 3], tvg="ppv.uk", start_id=100) + families = scan_families(red_button + events, []) + assert [f["template"] for f in families] == ["PPV EVENT #", "UK: BBC RED BUTTON #"] + assert families[0]["with_epg_id"] == 3 + assert families[1]["with_epg_id"] == 0 + + +def test_equal_epg_evidence_ranks_by_size(): + small = _streams("A #", [1, 2, 3], tvg="") + big = _streams("B #", [1, 2, 3, 4, 5], tvg="", start_id=50) + families = scan_families(small + big, []) + assert [f["template"] for f in families] == ["B #", "A #"] + + +def test_a_blank_or_missing_identifier_does_not_count_as_epg_evidence(): + streams = [{"id": 1, "name": "X 1", "tvg_id": " "}, + {"id": 2, "name": "X 2"}, + {"id": 3, "name": "X 3", "tvg_id": None}] + assert scan_families(streams, [])[0]["with_epg_id"] == 0 + + +def test_a_missing_name_is_skipped_not_fatal(): + streams = [{"id": 1, "name": None}, {"id": 2}] + _streams("Y #", [1, 2, 3]) + assert [f["template"] for f in scan_families(streams, [])] == ["Y #"] + + +# --------------------------------------------------------------------------- # +# The readout +# --------------------------------------------------------------------------- # + +def test_the_report_leads_with_uncovered_families_and_gives_a_pasteable_regex(): + streams = (_streams("MAX #", [100, 101, 102], tvg="max.us") + + _streams("PPV EVENT #", [1, 2, 3], tvg="ppv.us", start_id=50)) + families = scan_families(streams, _pat(r"^PPV EVENT \d+$")) + text = render_report(families, total_streams=len(streams), pattern_count=1, + feature_enabled=True) + assert "1 likely placeholder famil" in text + assert r"^MAX \d+$" in text + assert text.index("MAX #") < text.index("PPV EVENT #") + assert "Streams scanned" in text + + +def test_families_carrying_no_epg_data_are_separated_not_mixed_in(): + """MEASURED on 25,068 live stream names: 131 families are uncovered and + only 16 hold a stream carrying an EPG identifier. A pattern for one of the + other 115 could never resolve anything, so listing all 131 together is the + long, mostly unactionable report this feature exists to avoid.""" + with_epg = _streams("PPV #", [1, 2, 3], tvg="ppv.us") + without = _streams("KARAOKE #", list(range(1, 21)), tvg="", start_id=50) + text = render_report(scan_families(with_epg + without, []), + total_streams=23, pattern_count=0, feature_enabled=True) + assert "pattern to paste" in text + head, tail = text.split("carry no EPG data", 1) + assert "PPV #" in head + assert "KARAOKE #" in tail + assert "pattern to paste" not in tail + + +def test_the_detailed_list_is_capped_and_says_how_many_it_left_out(): + """131 families in full would be unreadable, and a silent cut is worse than + a cut the reader can see.""" + # Names must differ by LETTERS, not by digits: two names differing only in a + # digit are the same family by construction, which is the point of the module. + labels = [a + b for a in string.ascii_uppercase for b in string.ascii_uppercase] + streams = [] + for i, label in enumerate(labels[:DETAIL_LIMIT + 5]): + streams += _streams("FAM %s #" % label, [1, 2, 3], tvg="x", start_id=100 * i) + text = render_report(scan_families(streams, []), total_streams=len(streams), + pattern_count=0, feature_enabled=True) + assert text.count("pattern to paste") == DETAIL_LIMIT + assert "5 more" in text + + +def test_the_report_says_when_nothing_is_uncovered(): + streams = _streams("PPV EVENT #", [1, 2, 3]) + families = scan_families(streams, _pat(r"^PPV EVENT \d+$")) + text = render_report(families, total_streams=3, pattern_count=1, feature_enabled=True) + assert "No uncovered" in text + + +def test_the_report_warns_when_the_feature_is_off(): + """A pattern list that is never consulted covers nothing, whatever it says.""" + text = render_report([], total_streams=0, pattern_count=0, feature_enabled=False) + assert "OFF" in text + + +def test_the_report_is_plain_ascii_even_for_a_name_that_is_not(): + """A file opened under another codepage turns any non-ASCII byte into + mojibake, the same rule as the CSV preamble. The name here is the real + shape: MEASURED on this installation, family templates carry a circled + bullet and superscript letters. An earlier version of this test used only + synthetic ASCII names, so it passed while the live report broke the rule.""" + bullet = chr(0x25C9) + streams = _streams("UK: HIGH STREET TV # " + bullet, [1, 2, 3], tvg="x") + text = render_report(scan_families(streams, []), total_streams=3, + pattern_count=0, feature_enabled=True) + text.encode("ascii") + assert chr(92) + "u25c9" in text + + +def test_an_escaped_pattern_still_matches_the_real_name(): + """The escape has to be correct, not merely printable. Python regular + expressions accept the backslash-u form, so the suggestion a user + pastes must match the name it was derived from.""" + name = "UK: HIGH STREET TV 4 " + chr(0x25C9) + template, _ = template_of(name) + pattern = suggested_pattern(template) + pattern.encode("ascii") + assert re.compile(pattern, re.IGNORECASE).fullmatch(name) + + +# --------------------------------------------------------------------------- # +# The action wrapper +# --------------------------------------------------------------------------- # + +class _Logger: + def _record(self, msg, *a, **k): + pass + info = debug = warning = error = _record + + +def _plugin(plugin_module, tmp_path, monkeypatch, streams): + P = plugin_module.Plugin + inst = P.__new__(P) + inst.version = "test" + monkeypatch.setattr(P, "BUG_REPORT_DIR", str(tmp_path / "config"), raising=False) + monkeypatch.setattr(P, "_get_all_streams", lambda self, logger: streams, + raising=False) + return inst + + +def test_the_action_writes_the_readout_and_returns_its_path( + plugin_module, tmp_path, monkeypatch): + streams = _streams("MAX #", [100, 101, 102], tvg="max.us") + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + result = plugin.scan_placeholder_names_action( + {"epg_placeholder_matching_enabled": True, + "epg_placeholder_name_patterns": ""}, _Logger()) + assert result["status"] == "success" + assert result["file"].endswith("placeholder-name-scan.txt") + with open(result["file"], encoding="utf-8") as handle: + text = handle.read() + assert r"^MAX \d+$" in text + assert r"^MAX \d+$" in result["message"] + + +def test_the_action_resolves_patterns_even_when_the_feature_is_off( + plugin_module, tmp_path, monkeypatch): + """_resolve_epg_matching_settings returns no patterns when the toggle is + off. The scan must not inherit that, or every family would report as + uncovered the moment somebody switched the feature off.""" + streams = _streams("PPV EVENT #", [1, 2, 3], tvg="ppv.us") + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + result = plugin.scan_placeholder_names_action( + {"epg_placeholder_matching_enabled": False, + "epg_placeholder_name_patterns": r"^PPV EVENT \d+$"}, _Logger()) + with open(result["file"], encoding="utf-8") as handle: + text = handle.read() + assert "No uncovered" in text + assert "off" in result["message"] + + +def test_the_action_reports_an_unreadable_stream_list_as_an_error( + plugin_module, tmp_path, monkeypatch): + P = plugin_module.Plugin + plugin = _plugin(plugin_module, tmp_path, monkeypatch, []) + + def _boom(self, logger): + raise RuntimeError("database is down") + monkeypatch.setattr(P, "_get_all_streams", _boom, raising=False) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert result["status"] == "error" + assert "database is down" in result["error"] + + +def test_the_action_message_fits_a_toast(plugin_module, tmp_path, monkeypatch): + streams = [] + for i in range(40): + streams += _streams("A very long provider family name number %d #" % i, + [1, 2, 3], tvg="x", start_id=100 * i) + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert len(result["message"]) <= plugin_module.PluginConfig.TOAST_BUDGET_CHARS From 557f7c6032253132ed5d46afdfe1fa4b00c310d1 Mon Sep 17 00:00:00 2001 From: PiratesIRC <98669745+PiratesIRC@users.noreply.github.com> Date: Sun, 6 Sep 2026 11:05:57 -0500 Subject: [PATCH 2/3] fix(epg): close four review findings in the placeholder scan (#43) Two reviews of the first version of the placeholder family scan found four defects. Each was reproduced before it was changed, and each is now covered by a named test that fails when the fix is reverted. A character above the basic multilingual plane, an emoji or a flag, was escaped with the lower-case backslash-u form. That form carries a MINIMUM of four hex digits, not exactly four, so an emoji produced five and Python read only the first four. The pattern compiled, matched nothing, and the family kept reporting as uncovered with nothing saying why. The existing round-trip test used a character inside the basic plane, which is the one class the old form already handled. Characters above the plane now use the upper-case eight-digit form. A literal backslash was left unescaped in the template while a literal hash was escaped, so a backslash followed by a digit produced the same two characters as an escaped hash. Two different names grouped into one family and the suggested pattern matched neither. The scan ran inside the request with no yield, no input cap and no time budget. The pattern safety gate deliberately admits patterns that can backtrack polynomially, on the stated promise that the runtime bounds the input, and that promise was kept only on the regex pre-processing path. The walk now hands the worker back every 500 names, skips a name over 500 characters and stops at a five second budget, using the same PluginConfig limits as that path. A stopped walk is reported as partial rather than passed off as finished: a family the walk never reached is missing, not covered. A failure to write the readout returned a plain success. The readout exists only in that file, so the notification now says the file is missing, and says it FIRST, because the notification helper drops lines from the end and a test caught the warning being truncated away. Also from those reviews: distinct slot numbers are counted within one slot rather than across slots, so two slots holding two values each no longer count as four; a resolution tag groups regardless of the case of its K, since patterns are compiled case-insensitively; a suggestion longer than the setting will accept is marked as such instead of being offered as though it worked; a database row that is not a dictionary is skipped rather than raising out of the walk; and the second listing is capped and says how many it left out. Verification: 1,506 tests pass. Twenty-four mutations were applied one at a time and every one was caught by a named test, with a comment-only control leaving the suite green. One mutation initially escaped, because the test asserted the argument NAMES appeared in the call and setting them to None keeps the names; the assertion now pins the values. Re-measured over the same 25,068 live stream names: 132 families, 131 uncovered, 16 carrying an EPG identifier, 0.26 seconds, no name skipped, budget not tripped, output plain ASCII, and all 132 suggestions match their own example name. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017ZBufwmvsXqzkq2h2VkAwB --- CHANGELOG.md | 35 +++++ Stream-Mapparr/placeholder_scan.py | 143 +++++++++++++++-- Stream-Mapparr/plugin.py | 35 ++++- tests/test_placeholder_scan.py | 238 ++++++++++++++++++++++++++++- tests/test_worker_yield.py | 40 ++++- 5 files changed, 473 insertions(+), 18 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 89f5f52..4257375 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -52,6 +52,41 @@ numbered family is a placeholder: a numbered channel family whose names are already informative matches better as it is. + Two code reviews of the first version found four defects, all reproduced + before they were changed and all now covered by a named test that fails when + the fix is reverted: + + A character above the basic plane, such as an emoji or a flag, was escaped in + the four-digit form. That form carries a minimum of four digits, not exactly + four, so an emoji produced five and Python read only the first four. The + pattern compiled, matched nothing, and the family kept reporting as uncovered + with nothing saying why. The earlier test used a character inside the basic + plane, the one class that already worked. + + A backslash followed by a digit produced the same two characters as an escaped + literal hash, so two different names grouped into one family and the suggested + pattern matched neither. + + The scan ran inside the request with no yield, no input cap and no time + budget, while the pattern safety gate deliberately admits patterns that can + backtrack polynomially on the promise that the runtime bounds them. It now + hands the worker back every 500 names, skips a name over 500 characters and + stops at a five second budget, using the same limits as the regex + pre-processing path. A stopped scan is reported as partial rather than passed + off as finished, because a family the walk never reached is missing, not + covered. + + A failure to write the readout returned a plain success. The full readout + exists only in that file, so the notification now says the file is missing, + and says it first, because the notification drops lines from the end. + + Also from those reviews: distinct slot numbers are counted within one slot + rather than across slots, a resolution tag groups regardless of the case of + its K, a suggestion too long for the setting to accept is marked as such + instead of being offered as though it worked, a database row that is not a + dictionary is skipped rather than raising, and the second listing is capped + and says how many it left out. + ## v1.26.2481756 (September 5, 2026) ### Fixed diff --git a/Stream-Mapparr/placeholder_scan.py b/Stream-Mapparr/placeholder_scan.py index df20b66..8c97ea8 100644 --- a/Stream-Mapparr/placeholder_scan.py +++ b/Stream-Mapparr/placeholder_scan.py @@ -28,6 +28,7 @@ This reports. It never edits the setting, and nothing here writes to the database. """ +import time # A family needs at least this many DIFFERENT numbers in its slot before it is # worth reporting. Counting streams instead would let one stream name carried by @@ -45,6 +46,17 @@ # left out is stated rather than the cut being silent. DETAIL_LIMIT = 25 +# The Placeholder Name Patterns setting refuses a pattern longer than this, so a +# suggestion over the limit would be skipped in silence if it were pasted. The +# readout says so rather than offering it as though it worked. The number is +# PluginConfig.REGEX_PATTERN_MAX_LEN, repeated here because this module has no +# plugin import; tests/test_placeholder_scan.py pins the two together. +SUGGESTION_MAX_LEN = 500 + +# How often the walk hands control back to the caller. The plugin runs under +# gevent, where a loop that never yields freezes the entire worker. +YIELD_EVERY = 500 + # The slot marker in a template. A name containing a literal one is stored # with it backslash-escaped so the two can never be confused. SLOT = "#" @@ -71,7 +83,20 @@ def ascii_safe(text): pattern: Python's regular expressions accept backslash-u, so a pattern written this way still matches the real name. """ - return "".join(ch if ord(ch) < 128 else "\\u%04x" % ord(ch) for ch in text) + out = [] + for ch in text: + code = ord(ch) + if code < 128: + out.append(ch) + elif code <= 0xFFFF: + out.append("\\u%04x" % code) + else: + # The lower-case form has a MINIMUM width of four, not a fixed + # one, so an emoji produced five hex digits and Python read only + # the first four. The pasted pattern then matched nothing and the + # family kept reporting as uncovered with nothing saying why. + out.append("\\U%08x" % code) + return "".join(out) def _escape_literal(text): @@ -101,7 +126,14 @@ def template_of(name): while index < length: char = name[index] if not char.isdigit(): - parts.append("\\" + SLOT if char == SLOT else char) + if char in (SLOT, "\\"): + # Both are escaped, and the backslash MUST be, or a backslash + # followed by a digit produces the same two characters as an + # escaped hash: two different names then group into one family + # and neither is matched by the pattern the family suggests. + parts.append("\\" + char) + else: + parts.append(char) index += 1 continue start = index @@ -112,6 +144,11 @@ def template_of(name): followed_by_k = index < length and name[index] in ("k", "K") if at_boundary and followed_by_k: parts.append(run) + # The K is upper-cased in the TEMPLATE only, never in the name. + # Patterns are compiled case-insensitively, so 4k and 4K are one + # family; keeping the letter verbatim split it in two. + parts.append("K") + index += 1 continue parts.append(SLOT) numbers.append(run) @@ -132,8 +169,8 @@ def suggested_pattern(template): length = len(template) while index < length: char = template[index] - if char == "\\" and index + 1 < length and template[index + 1] == SLOT: - literal.append(SLOT) + if char == "\\" and index + 1 < length and template[index + 1] in (SLOT, "\\"): + literal.append(template[index + 1]) index += 2 continue if char == SLOT: @@ -152,7 +189,8 @@ def _has_epg_identifier(stream): return bool((stream.get("tvg_id") or "").strip()) -def scan_families(streams, patterns): +def scan_families(streams, patterns, on_yield=None, yield_every=YIELD_EVERY, + max_name_len=None, budget_seconds=None, stats=None): """Group `streams` by template and describe every numbered family found. `patterns` is the list of compiled placeholder patterns already configured. @@ -163,25 +201,71 @@ def scan_families(streams, patterns): Returns a list of dicts, best candidate first, ranked by how many members carry an EPG identifier and then by size. A family is reported whether or not it is covered; the caller decides what to show. + + RUNTIME CONTAINMENT, because this runs inside a uWSGI worker running gevent + where a loop that never yields freezes the whole worker and every request on + it. The pattern safety gate in the plugin deliberately admits patterns that + can backtrack polynomially, on the stated promise that the runtime bounds + the input, and that promise has to be kept here as well as on the regex + pre-processing path: + + `on_yield` is called every `yield_every` names, so the caller can hand + control back to the hub. + + `max_name_len` skips a name longer than the cap rather than applying up to + fifty operator-supplied patterns to it. + + `budget_seconds` stops the walk once that much time has been spent. + + A stopped walk is reported as partial through `stats`, never passed off as a + finished one: a family the walk never reached would otherwise be reported as + uncovered on no evidence. `stats`, when a dict is supplied, receives + `scanned`, `skipped_long` and `budget_tripped`. """ grouped = {} - for stream in streams: - name = stream.get("name") or "" + scanned = 0 + skipped_long = 0 + budget_tripped = False + deadline = (time.monotonic() + budget_seconds) if budget_seconds is not None else None + + for index, stream in enumerate(streams): + if on_yield is not None and index and index % yield_every == 0: + on_yield(index) + if deadline is not None and time.monotonic() >= deadline: + budget_tripped = True + break + # The rows come from whatever the database returned. A row that is not a + # dict must be skipped, not raise: an exception out of this walk reaches + # the operator as a stack trace instead of a readout. + name = stream.get("name") if isinstance(stream, dict) else None + name = name or "" if not name: continue + if max_name_len is not None and len(name) > max_name_len: + skipped_long += 1 + continue + scanned += 1 template, numbers = template_of(name) if not numbers: continue family = grouped.setdefault(template, { "template": template, "count": 0, - "numbers": set(), + "slots": None, "example": name, "with_epg_id": 0, "covered": 0, }) family["count"] += 1 - family["numbers"].add(numbers) + # Distinct numbers are counted WITHIN a slot, not across slots. Counting + # combinations let a two-slot template with two values in each slot show + # four "different numbers" while neither slot held three, which is not + # what the threshold or the readout says. + if family["slots"] is None: + family["slots"] = [set() for _ in numbers] + for position, value in enumerate(numbers): + if position < len(family["slots"]): + family["slots"][position].add(value) if name < family["example"]: family["example"] = name if _has_epg_identifier(stream): @@ -189,9 +273,14 @@ def scan_families(streams, patterns): if patterns and any(p.fullmatch(name) for p in patterns): family["covered"] += 1 + if isinstance(stats, dict): + stats["scanned"] = scanned + stats["skipped_long"] = skipped_long + stats["budget_tripped"] = budget_tripped + families = [] for family in grouped.values(): - distinct = len(family["numbers"]) + distinct = max((len(slot) for slot in family["slots"] or []), default=0) if distinct < MIN_DISTINCT_NUMBERS: continue families.append({ @@ -208,7 +297,8 @@ def scan_families(streams, patterns): return families -def render_report(families, total_streams, pattern_count, feature_enabled): +def render_report(families, total_streams, pattern_count, feature_enabled, + stats=None): """The operator-facing readout, plain ASCII. ASCII on purpose, the same rule as the CSV export preamble: this file is @@ -247,6 +337,24 @@ def render_report(families, total_streams, pattern_count, feature_enabled): "", ] + stats = stats if isinstance(stats, dict) else {} + if stats.get("skipped_long"): + lines += [ + f"NOTE: {stats['skipped_long']} name(s) were too long to scan and were", + "skipped. A very long name is not read, because a pattern can take a long", + "time over one and this runs inside a request.", + "", + ] + if stats.get("budget_tripped"): + lines += [ + "NOTE: the scan did not finish. It stopped at its time limit, so the", + "figures above cover only the names it reached and a family it never", + "reached is missing rather than covered. Run it again when the server is", + "quieter, or simplify the placeholder patterns, which are applied to every", + "name.", + "", + ] + if not feature_enabled: lines += [ "NOTE: EPG-Based Placeholder Matching is OFF, so the pattern list is", @@ -277,7 +385,9 @@ def render_report(families, total_streams, pattern_count, feature_enabled): lines.append(f" different slot numbers : {family['distinct_numbers']}") lines.append(f" carrying an EPG id : {family['with_epg_id']}") lines.append(" example : " + ascii_safe(family["example"])) - lines.append(f" pattern to paste : {family['suggested']}") + suffix = (" (TOO LONG for the setting, it will be skipped)" + if len(family["suggested"]) > SUGGESTION_MAX_LEN else "") + lines.append(f" pattern to paste : {family['suggested']}{suffix}") lines.append("") if len(actionable) > DETAIL_LIMIT: lines.append(f" and {len(actionable) - DETAIL_LIMIT} more, not described here.") @@ -297,6 +407,15 @@ def render_report(families, total_streams, pattern_count, feature_enabled): lines.append(f" and {len(no_epg) - DETAIL_LIMIT * 4} more, not listed here.") lines.append("") + too_long = [f for f in families if len(f["suggested"]) > SUGGESTION_MAX_LEN] + if too_long: + lines.append(f"{len(too_long)} of the suggested patterns are too long for the") + lines.append(f"Placeholder Name Patterns setting, which refuses anything over") + lines.append(f"{SUGGESTION_MAX_LEN} characters. Those are marked below and would") + lines.append("be skipped in silence if pasted. Shorten the name or write a") + lines.append("shorter pattern by hand.") + lines.append("") + if not uncovered: lines.append("No uncovered numbered families were found.") lines.append("") diff --git a/Stream-Mapparr/plugin.py b/Stream-Mapparr/plugin.py index 14b82cb..0140b31 100644 --- a/Stream-Mapparr/plugin.py +++ b/Stream-Mapparr/plugin.py @@ -2639,10 +2639,26 @@ def scan_placeholder_names_action(self, settings, logger, context=None): settings, 'epg_placeholder_matching_enabled', PluginConfig.DEFAULT_EPG_PLACEHOLDER_MATCHING_ENABLED) - families = scan.scan_families(streams, patterns) + # RUNTIME CONTAINMENT, the same three limits the regex pre-processing + # path uses, for the same reason. This action is dispatched + # synchronously, so it runs inside the request in a uWSGI worker + # running gevent, where a loop that never yields freezes the whole + # worker and every other request on it (bug-117). The pattern safety + # gate deliberately admits patterns that can backtrack polynomially on + # the stated promise that the runtime bounds the input, and up to + # REGEX_RULES_MAX of them are applied to each of about 25,000 names. + stats = {} + cfg = PluginConfig + families = scan.scan_families( + streams, patterns, + on_yield=lambda _index: self._cooperative_yield(), + yield_every=cfg.REGEX_YIELD_EVERY, + max_name_len=cfg.REGEX_NAME_MAX_LEN, + budget_seconds=cfg.REGEX_PASS_BUDGET_S, + stats=stats) text = scan.render_report(families, total_streams=len(streams), pattern_count=len(patterns), - feature_enabled=enabled) + feature_enabled=enabled, stats=stats) uncovered = [f for f in families if f["uncovered"] > 0] path = None @@ -2677,12 +2693,25 @@ def scan_placeholder_names_action(self, settings, logger, context=None): ] if actionable: top = actionable[0] - parts.append(f"Best candidate: {top['template']} " + # ascii_safe here too, not only in the file: a provider name can + # carry characters the readout deliberately escapes, and the + # toast should show the same text the file does. + parts.append(f"Best candidate: {scan.ascii_safe(top['template'])} " f"({top['count']} streams, " f"{top['with_epg_id']} with an EPG id).") parts.append(f"Pattern to paste: {top['suggested']}") if not enabled: parts.append("EPG-Based Placeholder Matching is currently off.") + if stats.get("budget_tripped"): + parts.append("The scan stopped at its time limit, so this is partial.") + if path is None: + # FIRST, not appended. The readout exists only in that file, so a run + # that could not write it has produced nothing the operator can read. + # _fit_toast drops whole lines from the END, so appending this put the + # one line that must survive in the position most likely to be cut, + # which a test caught. + parts.insert(0, "The full readout could NOT be written to disk, so only " + "this summary exists.") result = {"status": "success", "message": self._fit_toast(parts)} if path: diff --git a/tests/test_placeholder_scan.py b/tests/test_placeholder_scan.py index d386de7..b8b1d66 100644 --- a/tests/test_placeholder_scan.py +++ b/tests/test_placeholder_scan.py @@ -53,7 +53,9 @@ def test_a_resolution_tag_is_not_a_slot(): names under one template on a real installation.""" assert template_of("US: CINEMANIA 4K") == ("US: CINEMANIA 4K", ()) assert template_of("PPV 12 | 4K") == ("PPV # | 4K", ("12",)) - assert template_of("Movie 8k") == ("Movie 8k", ()) + # The K is upper-cased in the TEMPLATE only, so 4k and 4K are one family. + # The name itself is never altered. + assert template_of("Movie 8k") == ("Movie 8K", ()) def test_a_resolution_tag_needs_a_boundary_before_the_digit(): @@ -367,3 +369,237 @@ def test_the_action_message_fits_a_toast(plugin_module, tmp_path, monkeypatch): plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) result = plugin.scan_placeholder_names_action({}, _Logger()) assert len(result["message"]) <= plugin_module.PluginConfig.TOAST_BUDGET_CHARS + + +# --------------------------------------------------------------------------- # +# Review findings, 2026-09-06. Each was reproduced before it was fixed. +# --------------------------------------------------------------------------- # + +def test_a_character_above_the_basic_plane_is_escaped_in_the_wider_form(): + """The four-digit escape has a MINIMUM width, not a fixed one, so an emoji + produced five hex digits and Python's regular expressions read only four. + The pasted pattern then matched nothing and the family kept being reported + as uncovered, with nothing saying why. Measured before the fix.""" + name = "MAX " + chr(0x1F600) + " 1" + template, _ = template_of(name) + pattern = suggested_pattern(template) + pattern.encode("ascii") + assert re.compile(pattern, re.IGNORECASE).fullmatch(name) + + +def test_a_backslash_before_a_digit_is_not_confused_with_a_literal_hash(): + """A literal hash is stored escaped. A bare backslash followed by a digit + produced those same two characters, so two different names grouped into one + family and the suggested pattern did not match either. Measured before the + fix: A-backslash-1-B and A-hash-B both became the template A-backslash-hash-B.""" + backslash = chr(92) + with_backslash = "A" + backslash + "1B" + with_hash = "A#B" + assert template_of(with_backslash)[0] != template_of(with_hash)[0] + pattern = suggested_pattern(template_of(with_backslash)[0]) + assert re.compile(pattern).fullmatch(with_backslash) + + +def test_distinct_numbers_are_counted_within_one_slot_not_across_slots(): + """Two slots holding two values each produced four combinations and cleared + a threshold of three, while neither slot held three different numbers. The + report says "different numbers appear in its slot", so the count has to mean + that.""" + names = ["S 1-1", "S 1-2", "S 2-1", "S 2-2"] + streams = [{"id": i, "name": n, "tvg_id": ""} for i, n in enumerate(names)] + assert scan_families(streams, []) == [] + names += ["S 3-1"] + streams = [{"id": i, "name": n, "tvg_id": ""} for i, n in enumerate(names)] + families = scan_families(streams, []) + assert [f["distinct_numbers"] for f in families] == [3] + + +def test_the_resolution_tag_groups_regardless_of_the_case_of_the_k(): + """Patterns are compiled case-insensitively, so 4k and 4K are one family. + Keeping the tag verbatim split it in two.""" + names = ["Sky 4k 1", "Sky 4K 2", "Sky 4k 3"] + streams = [{"id": i, "name": n, "tvg_id": ""} for i, n in enumerate(names)] + families = scan_families(streams, []) + assert len(families) == 1 + assert families[0]["count"] == 3 + pattern = re.compile(families[0]["suggested"], re.IGNORECASE) + assert all(pattern.fullmatch(n) for n in names) + + +def test_the_scan_yields_to_the_worker_while_it_walks_the_names(): + """The plugin runs under gevent, where a loop that never yields freezes the + whole worker and every request on it (bug-117). The scan reads 25,000 names + and applies up to 50 operator-supplied patterns to each.""" + calls = [] + streams = [{"id": i, "name": "N %d" % i, "tvg_id": ""} for i in range(1200)] + scan_families(streams, [], on_yield=calls.append, yield_every=500) + assert len(calls) >= 2 + + +def test_an_over_long_name_is_skipped_and_counted_not_matched(): + """The pattern safety gate deliberately admits polynomial patterns on the + promise that the runtime bounds the input. That bound has to exist here too.""" + stats = {} + streams = [{"id": 1, "name": "X" * 600 + " 1", "tvg_id": ""}] + streams += _streams("OK #", [1, 2, 3]) + families = scan_families(streams, [], max_name_len=500, stats=stats) + assert [f["template"] for f in families] == ["OK #"] + assert stats["skipped_long"] == 1 + + +def test_a_tripped_time_budget_stops_the_scan_and_is_reported_as_partial(): + """Reporting a partial scan is honest. Freezing the worker, or silently + reporting a family as uncovered because the run gave up on it, is not.""" + stats = {} + streams = _streams("A #", [1, 2, 3]) + _streams("B #", [4, 5, 6], start_id=90) + scan_families(streams, [], budget_seconds=0.0, stats=stats) + assert stats["budget_tripped"] is True + text = render_report([], total_streams=6, pattern_count=0, feature_enabled=True, + stats=stats) + assert "did not finish" in text + + +def test_the_report_says_how_many_names_it_skipped(): + text = render_report([], total_streams=10, pattern_count=0, feature_enabled=True, + stats={"skipped_long": 4, "budget_tripped": False}) + assert "4" in text and "too long" in text + + +def test_the_detail_limit_is_twenty_five(): + """Pinned as a literal. The cap test above sizes its input from the constant, + so lowering the constant moved the test with it: measured, DETAIL_LIMIT could + be cut from 25 to 5 with the whole suite still green.""" + assert DETAIL_LIMIT == 25 + + +def test_the_header_numbers_are_the_numbers_and_not_labels_alone(): + """This project has already shipped a CSV preamble whose lines were false for + the life of the feature, because only the label was ever checked.""" + streams = _streams("A #", [1, 2, 3], tvg="x") + _streams("B #", [1, 2, 3], + start_id=40) + families = scan_families(streams, _pat(r"^A \d+$")) + text = render_report(families, total_streams=6, pattern_count=1, + feature_enabled=True) + assert "Streams scanned : 6" in text + assert "Numbered families found : 2" in text + assert "Placeholder patterns configured : 1" in text + assert "Families no pattern covers : 1" in text + + +def test_the_second_listing_is_capped_and_says_how_many_it_left_out(): + """On live data this is the 115-family list, the one most likely to run long. + A silent cut is the thing this readout exists to avoid.""" + limit = DETAIL_LIMIT * 4 + labels = [a + b for a in string.ascii_uppercase for b in string.ascii_uppercase] + streams = [] + for i, label in enumerate(labels[:limit + 3]): + streams += _streams("NOEPG %s #" % label, [1, 2, 3], tvg="", start_id=100 * i) + text = render_report(scan_families(streams, []), total_streams=len(streams), + pattern_count=0, feature_enabled=True) + assert text.count("(3 streams)") == limit + assert "3 more" in text + + +def test_a_family_survives_duplicate_rows_only_through_its_other_numbers(): + """Five rows of `HBO 1` plus `HBO 2` and `HBO 3` is a real family of three + numbers carried seven times. The all-duplicates test alone cannot tell the + implemented rule from one keyed on the whole name.""" + streams = _streams("HBO #", [1] * 5) + _streams("HBO #", [2, 3], start_id=60) + families = scan_families(streams, []) + assert families[0]["count"] == 7 + assert families[0]["distinct_numbers"] == 3 + + +def test_regex_metacharacters_in_a_name_round_trip(): + """The escaping is hand-rolled rather than re.escape, because re.escape also + escapes the space and the hash and would make every suggestion unreadable. + Hand-rolled means it needs its own test.""" + for name in ["A (B) [C] {D} 1", "A.B*C+D?E 1", "A|B^C$D 1", "50% off 3"]: + template, _ = template_of(name) + pattern = suggested_pattern(template) + assert re.compile(pattern, re.IGNORECASE).fullmatch(name), name + + +def test_a_suggestion_too_long_for_the_setting_is_flagged(): + """The setting refuses a pattern over 500 characters. A suggestion longer + than that would be skipped in silence if pasted, so the readout says so + rather than offering it as if it worked.""" + long_name = "X" * 520 + " 1" + streams = [{"id": i, "name": "X" * 520 + " %d" % n, "tvg_id": "x"} + for i, n in enumerate([1, 2, 3])] + text = render_report(scan_families(streams, []), total_streams=3, + pattern_count=0, feature_enabled=True) + assert "too long" in text.lower() + assert len(long_name) > 500 + + +def test_a_row_that_is_not_a_dict_is_skipped_rather_than_raising(): + """scan_families is called with whatever the database returned. An exception + out of the walk reaches the operator as a stack trace, not as a readout.""" + streams = [None, "not a dict", 7] + _streams("Z #", [1, 2, 3]) + assert [f["template"] for f in scan_families(streams, [])] == ["Z #"] + + +def test_the_ordering_of_equally_ranked_families_is_stable(): + """131 families on live data. Two runs must list them in the same order or + the readout cannot be compared with the previous one.""" + streams = (_streams("B #", [1, 2, 3], tvg="") + _streams("A #", [1, 2, 3], + tvg="", start_id=40)) + first = [f["template"] for f in scan_families(streams, [])] + second = [f["template"] for f in scan_families(list(reversed(streams)), [])] + assert first == second == ["A #", "B #"] + + +def test_the_suggestion_length_limit_matches_the_setting_that_enforces_it(plugin_module): + """Two constants in two modules, because placeholder_scan.py has no plugin + import. If they drift, the readout promises a pattern the setting rejects.""" + from placeholder_scan import SUGGESTION_MAX_LEN + assert SUGGESTION_MAX_LEN == plugin_module.PluginConfig.REGEX_PATTERN_MAX_LEN + + +def test_the_action_says_when_nothing_numbered_was_found( + plugin_module, tmp_path, monkeypatch): + streams = [{"id": 1, "name": "CNN", "tvg_id": ""}] + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert "No numbered stream-name families" in result["message"] + + +def test_the_action_says_when_every_family_is_covered( + plugin_module, tmp_path, monkeypatch): + streams = _streams("PPV EVENT #", [1, 2, 3], tvg="x") + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + result = plugin.scan_placeholder_names_action( + {"epg_placeholder_matching_enabled": True, + "epg_placeholder_name_patterns": r"^PPV EVENT \d+$"}, _Logger()) + assert "cover them all" in result["message"] + + +def test_the_action_counts_the_families_that_carry_epg_data_not_all_of_them( + plugin_module, tmp_path, monkeypatch): + """MEASURED on this installation: 131 uncovered families and 16 carrying an + EPG identifier. Counting all of them reports a problem eight times larger + than the one worth acting on.""" + with_epg = _streams("PPV #", [1, 2, 3], tvg="ppv.us") + without = _streams("KARAOKE #", [1, 2, 3], tvg="", start_id=50) + plugin = _plugin(plugin_module, tmp_path, monkeypatch, with_epg + without) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert "1 uncovered placeholder family carrying EPG data" in result["message"] + assert "out of 2 uncovered in total" in result["message"] + assert "PPV #" in result["message"] + + +def test_the_action_says_so_when_the_readout_could_not_be_written( + plugin_module, tmp_path, monkeypatch): + """The full readout exists ONLY in that file, so a run that could not write + it produced nothing the operator can read. Reporting a plain success there + is the failure this catch exists to prevent.""" + streams = _streams("MAX #", [1, 2, 3], tvg="x") + plugin = _plugin(plugin_module, tmp_path, monkeypatch, streams) + + def _refuse(*args, **kwargs): + raise OSError("read-only file system") + monkeypatch.setattr(plugin_module.os, "makedirs", _refuse) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert "file" not in result + assert "could NOT be written" in result["message"] diff --git a/tests/test_worker_yield.py b/tests/test_worker_yield.py index b6ccb68..5ea0932 100644 --- a/tests/test_worker_yield.py +++ b/tests/test_worker_yield.py @@ -90,6 +90,40 @@ def test_hot_loop_yields(plugin_ast, action, iterated): f"handing the worker back") +# Not a loop in plugin.py. The placeholder family scan walks the stream names +# inside placeholder_scan.py, which has no Django import and no knowledge of +# gevent, so the action hands it the yield as a callback instead. It is counted +# here because the pinned total below must account for every call site. +CALLBACK_SITES = [ + ("scan_placeholder_names_action", "scan.scan_families"), +] + + +@pytest.mark.parametrize("action,called", CALLBACK_SITES) +def test_the_scan_hands_its_walk_a_yield(plugin_ast, action, called): + """The walk reads about 25,000 names and applies every configured + placeholder pattern to each, inside the request. Without this the worker is + held for the whole run, which is the half of bug-117 that the sync-versus- + background gate does not solve.""" + func = _function(plugin_ast, action) + for node in ast.walk(func): + if isinstance(node, ast.Call) and ast.unparse(node.func) == called: + source = ast.unparse(node) + assert "_cooperative_yield" in source, ( + f"{action} calls {called} without handing it a yield") + # The VALUES, not the argument names. Setting them to None keeps the + # names in the source and removes the bound, and that mutation went + # uncaught until this assertion was tightened. + assert "max_name_len=cfg.REGEX_NAME_MAX_LEN" in source, ( + f"{action} calls {called} without the input cap, so an operator " + f"pattern that backtracks has nothing bounding the name length") + assert "budget_seconds=cfg.REGEX_PASS_BUDGET_S" in source, ( + f"{action} calls {called} without the time budget, so a slow " + f"pattern holds the worker for as long as it takes") + return + raise AssertionError(f"no call to {called} in {action}") + + def test_yield_call_sites_are_pinned(plugin_ast): """Pin the count so a new matching loop is a deliberate decision. @@ -100,5 +134,7 @@ def test_yield_call_sites_are_pinned(plugin_ast): module_calls = [node for node in ast.walk(plugin_ast) if isinstance(node, ast.Call) and ast.unparse(node.func).endswith("_cooperative_yield")] - assert len(module_calls) == len(HOT_LOOPS), ( - f"expected one call per hot loop ({len(HOT_LOOPS)}), found {len(module_calls)}") + expected = len(HOT_LOOPS) + len(CALLBACK_SITES) + assert len(module_calls) == expected, ( + f"expected one call per hot loop ({len(HOT_LOOPS)}) plus one per callback " + f"site ({len(CALLBACK_SITES)}), found {len(module_calls)}") From 89f457d7e9b2bf123220770ad6196aac21df8465 Mon Sep 17 00:00:00 2001 From: PiratesIRC <98669745+PiratesIRC@users.noreply.github.com> Date: Sun, 6 Sep 2026 13:32:29 -0500 Subject: [PATCH 3/3] fix(epg): rank placeholder families by size, not by EPG evidence (#43) The first version ranked uncovered families by how many of their streams carried an EPG identifier, and split the readout in two, suggesting a pattern only for families that carried one. Four independent reviews and a measurement of the live database agree that was wrong in both directions. WHICH DIRECTION THE RISK RUNS, read from the code. In _resolve_current_epg_title_for_stream a stream with no EPG identifier returns immediately, so the matching name is never reassigned and adding a pattern for such a family cannot change matching at all. A stream that does resolve has its matching name replaced by the programme currently airing, and on the channel side the channel name is replaced. So the old ranking promoted the families where a pattern changes behaviour and demoted the families where it cannot. MEASURED against the live installation rather than argued. 3,700 streams carry an EPG identifier. Only 218 of those identifiers match a guide row, and 28 streams could resolve a programme at the moment of measurement. Of the 16 families the old ranking promoted, exactly one held a stream whose identifier matched a guide row, and none could resolve a programme. That one is a numbered channel lineup, which is the case a pattern harms. The largest family on the installation, at 273 streams, was reduced to a single line with no suggested pattern. Separately, the 16 families carrying identifiers and the 9 families whose names contain an event word do not overlap at all. WHAT CHANGED. Families are ranked by size, largest first, which is what the reporter of the issue did by hand. Every uncovered family gets a suggested pattern. The EPG identifier count stays as a note on each family, reported as how many of its streams carry one out of how many. A family where at least half of them do is marked with a caution saying a pattern there replaces a working name with whatever is airing. The notification leads with the largest family and carries the same caution when it applies. On the live data the caution fires on 13 of the 131 uncovered families, every one a recognisable channel lineup such as ITV, BeIN Sports or TNT Sport, and those families moved from ranks 1 to 16 down to ranks 31 to 60. The three largest families now lead the readout, each with a pattern to paste, and all three carry no EPG identifier at all. Verification: 1,510 tests pass. Twenty-nine mutations were applied one at a time and every one was caught by a named test, with a comment-only control leaving the suite green. That includes putting the old ranking back, which now fails a test that says why it was removed. Re-measured over the same 25,068 live stream names: output plain ASCII, all 132 suggestions match their own example name, no name skipped and the time budget not tripped. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_017ZBufwmvsXqzkq2h2VkAwB --- CHANGELOG.md | 32 ++++++++ README.md | 2 +- Stream-Mapparr/placeholder_scan.py | 107 ++++++++++++++---------- Stream-Mapparr/plugin.py | 33 ++++---- tests/test_placeholder_scan.py | 127 ++++++++++++++++++++--------- 5 files changed, 200 insertions(+), 101 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4257375..f410563 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -80,6 +80,38 @@ exists only in that file, so the notification now says the file is missing, and says it first, because the notification drops lines from the end. + **The ranking was reversed after four independent reviews and a measurement + of the live database.** The first version ranked uncovered families by how + many of their streams carried an EPG identifier, and split the readout into + families that carried one and families that did not, suggesting a pattern only + for the first group. Measured against the live installation, that was wrong in + both directions. + + Reading the code settles which direction the risk runs. A stream with no EPG + identifier returns from the resolver immediately, so adding a pattern for such + a family cannot change matching at all. A stream that does resolve has its + matching name replaced by the programme currently airing. So the old ranking + promoted the families where a pattern changes behaviour and demoted the ones + where it cannot. + + Measured on the database rather than argued: 3,700 streams carry an EPG + identifier, only 218 of those identifiers match a guide row, and 28 streams + could resolve a programme at the moment of measurement. Of the 16 families the + old ranking promoted, one contained a stream whose identifier matched a guide + row, and none could resolve a programme. That one was a numbered channel + lineup, which is the case a pattern harms. Meanwhile the largest family on the + installation, at 273 streams, was reduced to a single line with no suggested + pattern. + + Families are now ranked by size, largest first, which is what the reporter of + the issue did by hand. Every uncovered family gets a suggested pattern. The + EPG identifier count stays as a note on each family, saying how many of its + streams carry one out of how many, and a family where at least half of them do + is marked with a caution explaining that a pattern there replaces a working + name with whatever is airing. On the live data that caution fires on 13 + families, every one a recognisable channel lineup such as ITV or BeIN Sports, + and those families moved from the top of the list to ranks 31 to 60. + Also from those reviews: distinct slot numbers are counted within one slot rather than across slots, a resolution tag groups regardless of the case of its K, a suggestion too long for the setting to accept is marked as such diff --git a/README.md b/README.md index 41c485d..36dcd07 100644 --- a/README.md +++ b/README.md @@ -228,7 +228,7 @@ the operation lock prevents concurrent runs and auto-expires after 10 minutes. | **Validate Settings** | Check configuration, profiles, groups and databases | | **Test Regex Rules** | Preview what your regex rules would change, with before and after samples and invisible characters made visible. Writes the full readout to `/config/stream-mapparr/test-regex-rules.txt`, since a notification shows only about 280 characters | | **Check Stream Country Labels** | Compare each stream's group country against its EPG identifier suffix and report where they disagree. Reads two database columns, opens no provider connection and changes nothing. A disagreement is not automatically a fault: a channel carried in one country and made in another is ordinary | -| **Scan for Placeholder Patterns** | Group every stream name into a numbered family, by replacing its numbers with a slot, and report the families that no Placeholder Name Pattern covers, each with a regex to paste. Families whose streams carry EPG data are listed first, because a placeholder can only be resolved when there is guide data to resolve it from. A digit followed by K is left alone, since `4K` is a resolution tag rather than a slot number. Writes the full readout to `/config/stream-mapparr/placeholder-name-scan.txt`. Reads one database column, opens no provider connection and changes nothing | +| **Scan for Placeholder Patterns** | Group every stream name into a numbered family, by replacing its numbers with a slot, and report the families that no Placeholder Name Pattern covers, each with a regex to paste. The largest families come first. Each one says how many of its streams carry an EPG identifier, and a family where most of them do is marked, because that is more likely to be an ordinary numbered channel lineup, where a pattern replaces a name that is already matching. A digit followed by K is left alone, since `4K` is a resolution tag rather than a slot number. Writes the full readout to `/config/stream-mapparr/placeholder-name-scan.txt`. Reads one database column, opens no provider connection and changes nothing | | **Load/Process Channels** | Load channel and stream data from the database | | **Preview Changes** | Dry run with a CSV export | | **Match & Assign Streams** | Fuzzy match and assign streams to channels | diff --git a/Stream-Mapparr/placeholder_scan.py b/Stream-Mapparr/placeholder_scan.py index 8c97ea8..c7e460b 100644 --- a/Stream-Mapparr/placeholder_scan.py +++ b/Stream-Mapparr/placeholder_scan.py @@ -185,6 +185,24 @@ def suggested_pattern(template): return ascii_safe("^" + "".join(out) + "$") +# A family at or above this share of members carrying an EPG identifier is more +# likely to be an ordinary numbered channel lineup than a provider event slot, +# so the readout warns before suggesting a pattern for it. MEASURED: a pattern +# over such a family REPLACES the name that is matching today with whatever +# programme is airing, which changes as programming does, whereas a pattern over +# a family carrying no identifiers cannot change matching at all. +MOSTLY_EPG_SHARE = 0.5 + + +def _mostly_carries_epg(family): + """True when at least MOSTLY_EPG_SHARE of the family's streams carry an + EPG identifier.""" + count = family.get("count") or 0 + if not count: + return False + return (family.get("with_epg_id") or 0) >= MOSTLY_EPG_SHARE * count + + def _has_epg_identifier(stream): return bool((stream.get("tvg_id") or "").strip()) @@ -293,7 +311,14 @@ def scan_families(streams, patterns, on_yield=None, yield_every=YIELD_EVERY, "uncovered": family["count"] - family["covered"], "suggested": suggested_pattern(family["template"]), }) - families.sort(key=lambda f: (-f["with_epg_id"], -f["count"], f["template"])) + # Largest first. The EPG identifier count is NOT a ranking key. It was, and + # MEASURED on 25,068 live stream names that ranked 16 families above the + # other 115, of which exactly one held a stream whose identifier matched a + # guide row and none could resolve a programme at all, while the largest + # family on the installation, at 273 streams, was pushed to a single line. + # The signal is kept as an annotation on each family, where it says what it + # actually is. + families.sort(key=lambda f: (-f["count"], f["template"])) return families @@ -363,67 +388,63 @@ def render_report(families, total_streams, pattern_count, feature_enabled, "", ] - actionable = [f for f in uncovered if f["with_epg_id"] > 0] - no_epg = [f for f in uncovered if f["with_epg_id"] == 0] + too_long = [f for f in families if len(f["suggested"]) > SUGGESTION_MAX_LEN] + if too_long: + lines.append(f"{len(too_long)} of the suggested patterns are too long for the") + lines.append("Placeholder Name Patterns setting, which refuses anything over") + lines.append(f"{SUGGESTION_MAX_LEN} characters. Those are marked below and would") + lines.append("be skipped in silence if pasted. Shorten the name or write a") + lines.append("shorter pattern by hand.") + lines.append("") if uncovered: - word = "family" if len(actionable) == 1 else "families" - lines.append(f"{len(actionable)} likely placeholder {word} not covered by your " - f"current patterns") - lines.append("whose streams carry EPG data, best candidate first. The ranking is") - lines.append("how many streams in the family carry an EPG identifier, because a") - lines.append("placeholder can only ever be resolved when its streams carry EPG") - lines.append("data to resolve it from.") + word = "family" if len(uncovered) == 1 else "families" + lines.append(f"{len(uncovered)} numbered {word} that no pattern of yours covers,") + lines.append("largest first. Size is the ordering because a large family is the") + lines.append("one most likely to be worth your attention, and because it is what") + lines.append("a person auditing the list by hand would sort by.") lines.append("") - if not actionable: - lines.append(" None. Every uncovered family is listed below instead.") - lines.append("") - for family in actionable[:DETAIL_LIMIT]: + lines.append("Each entry says how many of its streams carry an EPG identifier.") + lines.append("Read that as a note, not as a score. A family carrying none cannot") + lines.append("be resolved from guide data today, and adding its pattern changes") + lines.append("nothing until that data arrives, which costs nothing meanwhile. A") + lines.append("family where most streams carry one is more likely to be an ordinary") + lines.append("numbered channel lineup than an event slot, and a pattern there") + lines.append("REPLACES a working name with whatever is airing. Those are marked.") + lines.append("") + for family in uncovered[:DETAIL_LIMIT]: lines.append(" " + ascii_safe(family["template"])) lines.append(f" streams : {family['count']}" f" ({family['uncovered']} not covered)") lines.append(f" different slot numbers : {family['distinct_numbers']}") - lines.append(f" carrying an EPG id : {family['with_epg_id']}") + lines.append(f" carrying an EPG id : {family['with_epg_id']}" + f" of {family['count']}") lines.append(" example : " + ascii_safe(family["example"])) suffix = (" (TOO LONG for the setting, it will be skipped)" if len(family["suggested"]) > SUGGESTION_MAX_LEN else "") lines.append(f" pattern to paste : {family['suggested']}{suffix}") + if _mostly_carries_epg(family): + lines.append(" CAUTION : most of these streams carry an") + lines.append(" EPG identifier, so this looks like") + lines.append(" a numbered channel lineup rather") + lines.append(" than an event slot. A pattern here") + lines.append(" replaces the name that is matching") + lines.append(" today with the programme airing,") + lines.append(" which changes as programming does.") lines.append("") - if len(actionable) > DETAIL_LIMIT: - lines.append(f" and {len(actionable) - DETAIL_LIMIT} more, not described here.") + if len(uncovered) > DETAIL_LIMIT: + lines.append(f" and {len(uncovered) - DETAIL_LIMIT} more, not described here.") lines.append("") - - if no_epg: - lines.append(f"{len(no_epg)} further uncovered families carry no EPG data at all.") - lines.append("A placeholder pattern for one of these could not resolve anything") - lines.append("today, so no pattern is suggested. They are listed because a family") - lines.append("can start carrying EPG data later, and because a numbered channel") - lines.append("family whose names are already informative belongs here rather than") - lines.append("in the list above.") - lines.append("") - for family in no_epg[:DETAIL_LIMIT * 4]: - lines.append(" " + ascii_safe(family["template"]) + f" ({family['count']} streams)") - if len(no_epg) > DETAIL_LIMIT * 4: - lines.append(f" and {len(no_epg) - DETAIL_LIMIT * 4} more, not listed here.") - lines.append("") - - too_long = [f for f in families if len(f["suggested"]) > SUGGESTION_MAX_LEN] - if too_long: - lines.append(f"{len(too_long)} of the suggested patterns are too long for the") - lines.append(f"Placeholder Name Patterns setting, which refuses anything over") - lines.append(f"{SUGGESTION_MAX_LEN} characters. Those are marked below and would") - lines.append("be skipped in silence if pasted. Shorten the name or write a") - lines.append("shorter pattern by hand.") - lines.append("") - - if not uncovered: + else: lines.append("No uncovered numbered families were found.") lines.append("") if covered: lines.append("Families your patterns already cover in full:") - for family in covered: + for family in covered[:DETAIL_LIMIT * 4]: lines.append(" " + ascii_safe(family["template"]) + f" ({family['count']} streams)") + if len(covered) > DETAIL_LIMIT * 4: + lines.append(f" and {len(covered) - DETAIL_LIMIT * 4} more, not listed here.") lines.append("") return "\n".join(lines) diff --git a/Stream-Mapparr/plugin.py b/Stream-Mapparr/plugin.py index 0140b31..2ac4d6e 100644 --- a/Stream-Mapparr/plugin.py +++ b/Stream-Mapparr/plugin.py @@ -2680,26 +2680,23 @@ def scan_placeholder_names_action(self, settings, logger, context=None): parts = [f"{len(families)} numbered families found, and your " f"{len(patterns)} pattern(s) cover them all."] else: - # The headline counts the families whose streams carry EPG data, - # because only those can ever be resolved. MEASURED on this - # installation: 131 families are uncovered and 16 of them hold a - # stream with an EPG identifier, so counting all of them would - # report a problem eight times larger than the one worth acting on. - actionable = [f for f in uncovered if f["with_epg_id"] > 0] - word = "family" if len(actionable) == 1 else "families" + # The headline leads with the LARGEST uncovered family, not with the + # one carrying the most EPG identifiers. Ranking by identifiers was + # tried and reversed: MEASURED on this installation it put 16 + # families ahead of the other 115, of which one held a stream whose + # identifier matched a guide row and none could resolve a programme, + # while the largest family at 273 streams was reduced to one line. + top = uncovered[0] + word = "family" if len(uncovered) == 1 else "families" parts = [ - f"{len(actionable)} uncovered placeholder {word} carrying EPG data, " - f"out of {len(uncovered)} uncovered in total.", + f"{len(uncovered)} numbered {word} that no pattern of yours covers.", + f"Largest: {scan.ascii_safe(top['template'])} ({top['count']} streams, " + f"{top['with_epg_id']} carrying an EPG id).", + f"Pattern to paste: {top['suggested']}", ] - if actionable: - top = actionable[0] - # ascii_safe here too, not only in the file: a provider name can - # carry characters the readout deliberately escapes, and the - # toast should show the same text the file does. - parts.append(f"Best candidate: {scan.ascii_safe(top['template'])} " - f"({top['count']} streams, " - f"{top['with_epg_id']} with an EPG id).") - parts.append(f"Pattern to paste: {top['suggested']}") + if scan._mostly_carries_epg(top): + parts.append("CAUTION: most of its streams carry an EPG id, so a " + "pattern here replaces a working name.") if not enabled: parts.append("EPG-Based Placeholder Matching is currently off.") if stats.get("budget_tripped"): diff --git a/tests/test_placeholder_scan.py b/tests/test_placeholder_scan.py index b8b1d66..679c3c6 100644 --- a/tests/test_placeholder_scan.py +++ b/tests/test_placeholder_scan.py @@ -182,18 +182,22 @@ def test_coverage_uses_fullmatch_like_the_matcher_does(): # Ranking # --------------------------------------------------------------------------- # -def test_families_whose_streams_carry_epg_identifiers_rank_first(): - """`BBC RED BUTTON #` is bigger, but its streams carry no EPG identifier, so - a placeholder pattern for it could never resolve anything.""" - red_button = _streams("UK: BBC RED BUTTON #", list(range(1, 11)), tvg="") - events = _streams("PPV EVENT #", [1, 2, 3], tvg="ppv.uk", start_id=100) - families = scan_families(red_button + events, []) - assert [f["template"] for f in families] == ["PPV EVENT #", "UK: BBC RED BUTTON #"] - assert families[0]["with_epg_id"] == 3 - assert families[1]["with_epg_id"] == 0 - - -def test_equal_epg_evidence_ranks_by_size(): +def test_the_largest_family_ranks_first_whatever_its_epg_data(): + """Ranking by EPG identifiers was tried and reversed. MEASURED on 25,068 live + stream names: it promoted 16 families above the other 115, of which exactly + one held a stream whose identifier matched a guide row and none could resolve + a programme, while the largest family on the installation, at 273 streams, + was reduced to a single line. Size is what a person auditing the list by hand + sorts by, and it is what the reporter of the issue actually did.""" + big_without = _streams("UK: BBC RED BUTTON #", list(range(1, 11)), tvg="") + small_with = _streams("PPV EVENT #", [1, 2, 3], tvg="ppv.uk", start_id=100) + families = scan_families(big_without + small_with, []) + assert [f["template"] for f in families] == ["UK: BBC RED BUTTON #", "PPV EVENT #"] + assert families[0]["with_epg_id"] == 0 + assert families[1]["with_epg_id"] == 3 + + +def test_ranking_is_by_size(): small = _streams("A #", [1, 2, 3], tvg="") big = _streams("B #", [1, 2, 3, 4, 5], tvg="", start_id=50) families = scan_families(small + big, []) @@ -222,26 +226,45 @@ def test_the_report_leads_with_uncovered_families_and_gives_a_pasteable_regex(): families = scan_families(streams, _pat(r"^PPV EVENT \d+$")) text = render_report(families, total_streams=len(streams), pattern_count=1, feature_enabled=True) - assert "1 likely placeholder famil" in text + assert "1 numbered family that no pattern of yours covers" in text assert r"^MAX \d+$" in text assert text.index("MAX #") < text.index("PPV EVENT #") assert "Streams scanned" in text -def test_families_carrying_no_epg_data_are_separated_not_mixed_in(): - """MEASURED on 25,068 live stream names: 131 families are uncovered and - only 16 hold a stream carrying an EPG identifier. A pattern for one of the - other 115 could never resolve anything, so listing all 131 together is the - long, mostly unactionable report this feature exists to avoid.""" +def test_every_uncovered_family_gets_a_pattern_regardless_of_epg_data(): + """The two-section split withheld a suggested pattern from any family whose + streams carried no EPG identifier. On the maintainer's installation that was + 115 of 131 families, including the largest. A pattern over such a family + cannot change matching until guide data arrives, so withholding it protected + against nothing.""" with_epg = _streams("PPV #", [1, 2, 3], tvg="ppv.us") without = _streams("KARAOKE #", list(range(1, 21)), tvg="", start_id=50) text = render_report(scan_families(with_epg + without, []), total_streams=23, pattern_count=0, feature_enabled=True) - assert "pattern to paste" in text - head, tail = text.split("carry no EPG data", 1) - assert "PPV #" in head - assert "KARAOKE #" in tail - assert "pattern to paste" not in tail + assert text.count("pattern to paste") == 2 + assert text.index("KARAOKE #") < text.index("PPV #") + assert "carrying an EPG id : 0 of 20" in text + assert "carrying an EPG id : 3 of 3" in text + + +def test_a_family_whose_streams_mostly_carry_epg_data_is_marked_as_a_risk(): + """A pattern over a family carrying no identifiers cannot change matching. + A pattern over one where the identifiers resolve REPLACES the name that is + matching today with the programme airing. The readout has to say which it is + looking at, because the reader cannot tell from the name.""" + lineup = _streams("UK: ITV # HD", [1, 2, 3, 4], tvg="itv.uk") + text = render_report(scan_families(lineup, []), total_streams=4, + pattern_count=0, feature_enabled=True) + assert "CAUTION" in text + assert "replaces the name" in text + + +def test_a_family_carrying_no_epg_data_is_not_marked_as_a_risk(): + slots = _streams("PPV EVENT #", [1, 2, 3], tvg="") + text = render_report(scan_families(slots, []), total_streams=3, + pattern_count=0, feature_enabled=True) + assert "CAUTION" not in text def test_the_detailed_list_is_capped_and_says_how_many_it_left_out(): @@ -486,16 +509,17 @@ def test_the_header_numbers_are_the_numbers_and_not_labels_alone(): assert "Families no pattern covers : 1" in text -def test_the_second_listing_is_capped_and_says_how_many_it_left_out(): - """On live data this is the 115-family list, the one most likely to run long. - A silent cut is the thing this readout exists to avoid.""" +def test_the_covered_listing_is_capped_and_says_how_many_it_left_out(): + """A silent cut is the thing this readout exists to avoid, and the list of + families the patterns already cover can run long too.""" limit = DETAIL_LIMIT * 4 labels = [a + b for a in string.ascii_uppercase for b in string.ascii_uppercase] streams = [] for i, label in enumerate(labels[:limit + 3]): - streams += _streams("NOEPG %s #" % label, [1, 2, 3], tvg="", start_id=100 * i) - text = render_report(scan_families(streams, []), total_streams=len(streams), - pattern_count=0, feature_enabled=True) + streams += _streams("COV %s #" % label, [1, 2, 3], tvg="", start_id=100 * i) + families = scan_families(streams, _pat(r"^COV [A-Z]+ " + chr(92) + r"d+$")) + text = render_report(families, total_streams=len(streams), + pattern_count=1, feature_enabled=True) assert text.count("(3 streams)") == limit assert "3 more" in text @@ -575,18 +599,31 @@ def test_the_action_says_when_every_family_is_covered( assert "cover them all" in result["message"] -def test_the_action_counts_the_families_that_carry_epg_data_not_all_of_them( +def test_the_action_leads_with_the_largest_family_not_the_one_with_epg_data( plugin_module, tmp_path, monkeypatch): - """MEASURED on this installation: 131 uncovered families and 16 carrying an - EPG identifier. Counting all of them reports a problem eight times larger - than the one worth acting on.""" - with_epg = _streams("PPV #", [1, 2, 3], tvg="ppv.us") - without = _streams("KARAOKE #", [1, 2, 3], tvg="", start_id=50) - plugin = _plugin(plugin_module, tmp_path, monkeypatch, with_epg + without) + """The notification shows about 280 characters, so what it leads with is + most of what the operator reads. It led with the family carrying the most + EPG identifiers, which MEASURED on this installation meant a numbered + channel lineup, while the largest family went unmentioned.""" + small_with_epg = _streams("PPV #", [1, 2, 3], tvg="ppv.us") + large_without = _streams("KARAOKE #", list(range(1, 21)), tvg="", start_id=50) + plugin = _plugin(plugin_module, tmp_path, monkeypatch, + small_with_epg + large_without) result = plugin.scan_placeholder_names_action({}, _Logger()) - assert "1 uncovered placeholder family carrying EPG data" in result["message"] - assert "out of 2 uncovered in total" in result["message"] - assert "PPV #" in result["message"] + assert "2 numbered families that no pattern of yours covers" in result["message"] + assert "Largest: KARAOKE #" in result["message"] + assert "20 streams, 0 carrying an EPG id" in result["message"] + + +def test_the_action_warns_when_its_top_family_is_a_channel_lineup( + plugin_module, tmp_path, monkeypatch): + """A pattern over a family whose streams carry EPG identifiers replaces the + name that is matching today. The warning has to survive into the toast, + because that is where the pattern is offered.""" + lineup = _streams("UK: ITV # HD", [1, 2, 3, 4], tvg="itv.uk") + plugin = _plugin(plugin_module, tmp_path, monkeypatch, lineup) + result = plugin.scan_placeholder_names_action({}, _Logger()) + assert "CAUTION" in result["message"] def test_the_action_says_so_when_the_readout_could_not_be_written( @@ -603,3 +640,15 @@ def _refuse(*args, **kwargs): result = plugin.scan_placeholder_names_action({}, _Logger()) assert "file" not in result assert "could NOT be written" in result["message"] + + +def test_the_caution_predicate_refuses_an_empty_family(): + """Zero streams carrying zero identifiers satisfies "at least half" on the + arithmetic alone, so an empty family would be marked as a channel lineup. + No such family reaches the readout today, which is exactly why the guard + needs its own test rather than relying on a caller.""" + from placeholder_scan import _mostly_carries_epg + assert _mostly_carries_epg({"count": 0, "with_epg_id": 0}) is False + assert _mostly_carries_epg({}) is False + assert _mostly_carries_epg({"count": 4, "with_epg_id": 2}) is True + assert _mostly_carries_epg({"count": 4, "with_epg_id": 1}) is False