Describe your suggested feature
First, thank you for implementing the new TTS cleanup feature on 2.1.3. It is a massive quality-of-life improvement.
The Problem:
Currently, users who alternate between reading visually and listening via TTS must maintain two separate configurations to filter out obfuscated watermarks and anti-scrape spam: a custom JavaScript TreeWalker for the visible text, and the new JSON ruleset for TTS.
Because these site-specific trash strings and unicode obfuscations change frequently, maintaining both scripts simultaneously requires duplicated effort.
Proposed Solution:
Could we unify the cleanup logic so users only have to maintain one ruleset?
Option A: Allow the TTS JSON ruleset to also apply to and clean up the visible text.
Option B: Allow the visual text cleanup JavaScript to be mirrored and applied to the TTS output.
Below are my current parallel configurations required to achieve the same result for both reading modes:
Other details
{
"format": "lnreader-tts-cleanup",
"version": 1,
"settings": {
"enabled": true,
"normalizeUnicode": true,
"rules": [
{
"id": "tts-rule-msdu7gg1-58",
"enabled": true,
"pattern": "u2014",
"isRegex": false,
"flags": "g",
"replacement": " "
},
{
"id": "tts-rule-msdu7gg1-59",
"enabled": true,
"pattern": "𝐍|𝓝|𝔑|п|η",
"isRegex": true,
"flags": "gi",
"replacement": "N"
},
{
"id": "tts-rule-msdu7gg1-60",
"enabled": true,
"pattern": "о|𝑜|𝐨|𝒐|开启|ο|0",
"isRegex": true,
"flags": "gi",
"replacement": "o"
},
{
"id": "tts-rule-msdu7gg1-61",
"enabled": true,
"pattern": "𝓋|𝐯|𝒗",
"isRegex": true,
"flags": "gi",
"replacement": "v"
},
{
"id": "tts-rule-msdu7gg1-62",
"enabled": true,
"pattern": "е|𝑒|𝐞|𝒆|Ε|ε",
"isRegex": true,
"flags": "gi",
"replacement": "e"
},
{
"id": "tts-rule-msdu7gg1-63",
"enabled": true,
"pattern": "і|ι|𝐢|𝑖|i|𝚒|Ⅰ|𝕚|1|!",
"isRegex": true,
"flags": "gi",
"replacement": "i"
},
{
"id": "tts-rule-msdu7gg1-64",
"enabled": true,
"pattern": "г|𝐠|𝒈",
"isRegex": true,
"flags": "gi",
"replacement": "g"
},
{
"id": "tts-rule-msdu7gg1-65",
"enabled": true,
"pattern": "һ|𝐡|𝒉",
"isRegex": true,
"flags": "gi",
"replacement": "h"
},
{
"id": "tts-rule-msdu7gg1-66",
"enabled": true,
"pattern": "т|𝐭|𝒕",
"isRegex": true,
"flags": "gi",
"replacement": "t"
},
{
"id": "tts-rule-msdu7gg1-67",
"enabled": true,
"pattern": "𝐥|ӏ|ℓ",
"isRegex": true,
"flags": "gi",
"replacement": "l"
},
{
"id": "tts-rule-msdu7gg1-68",
"enabled": true,
"pattern": "ꭆ",
"isRegex": true,
"flags": "gi",
"replacement": "r"
},
{
"id": "tts-rule-msdu7gg1-69",
"enabled": true,
"pattern": "а|α|𝐚|𝒶",
"isRegex": true,
"flags": "gi",
"replacement": "a"
},
{
"id": "tts-rule-msdu7gg1-70",
"enabled": true,
"pattern": "b|𝐛|𝑏",
"isRegex": true,
"flags": "gi",
"replacement": "b"
},
{
"id": "tts-rule-msdu7gg1-71",
"enabled": true,
"pattern": "(?:(?:This[ \t]*translation[^\\n]{0,40}?of|Do[ \t]*not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]*N[^\\w\\n]*o[^\\w\\n]*v[^\\w\\n]e[^\\w\\n](?:l|i)[^\\w\\n]*i[^\\w\\n]*g[^\\w\\n]*h[^\\w\\n]t[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-72",
"enabled": true,
"pattern": "(?:(?:This[ \t]*translation[^\\n]{0,40}?of|Do[ \t]*not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]*r[^\\w\\n]*a[^\\w\\n]n[^\\w\\n]o[^\\w\\n]b[^\\w\\n]e[^\\w\\n]s[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-73",
"enabled": true,
"pattern": "(?:(?:This[ \t]translation[^\\n]{0,40}?of|Do[ \t]not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]n[^\\w\\n]o[^\\w\\n]v[^\\w\\n]e[^\\w\\n]l[^\\w\\n]f[^\\w\\n]i[^\\w\\n]r[^\\w\\n]e[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-74",
"enabled": true,
"pattern": "\b(?:novelfire|ranobes)\.(?:com|net|org|info|cc|ru|biz)\b",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-75",
"enabled": true,
"pattern": "(?:This chapter is updated by|Read latest chapters at|The source of this content is|New novel chapters are published on|Find the original at)[ \t](?:novelfire|ranobes)[a-zA-Z0-9.-]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-76",
"enabled": true,
"pattern": "(?:https?:\/\/)?(?:www\.)?(?:discord\.gg|patreon\.com|ko-fi\.com)[^\\s\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-77",
"enabled": true,
"pattern": "Support (?:me|the (?:author|translator)) on (?:Patreon|Ko-fi)|Buy me a coffee[^\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-78",
"enabled": true,
"pattern": "(?:TL|ED|PR|Author['’]?s?)[ \t]Note[ \t]:",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-79",
"enabled": true,
"pattern": "(?:Find the original at|This novel is translated and hosted on)[^\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-80",
"enabled": true,
"pattern": "[\(\[\{【《「](?:This(?: work| translation)? is the )?intellectual property of [^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-81",
"enabled": true,
"pattern": "[\(\[\{【《「]Don['’]?t copy[^\\n\\)\\]\}】》」]read here[^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-82",
"enabled": true,
"pattern": "[\(\[\{【《「](?:Official version|Read the full story|Original source|Continue reading|Read more on our source|Unauthorized use[^\\n\\)\\]\}】》」])[\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-83",
"enabled": true,
"pattern": "[\(\[\{【《「](?:Only on|Exclusive on) [^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
}
],
"phoneticPairs": []
}
}
Text clean up logic
// 1. Core Target Map
const unicodeMap = {
'𝐍':'N','𝓝':'N','𝔑':'N','п':'n','η':'n',
'о':'o','𝑜':'o','𝐨':'o','𝒐':'o','开启':'o','ο':'o','0':'o',
'𝓋':'v','𝐯':'v','𝒗':'v',
'е':'e','𝑒':'e','𝐞':'e','𝒆':'e','Ε':'e','ε':'e',
'і':'i','ι':'i','𝐢':'i','𝑖':'i','i':'i','𝚒':'i','Ⅰ':'i','𝕚':'i','1':'i','!':'i',
'г':'g','𝐠':'g','𝒈':'g',
'һ':'h','𝐡':'h','𝒉':'h',
'т':'t','𝐭':'t','𝒕':'t',
'𝐥':'l','ӏ':'l','ℓ':'l',
'ꭆ':'r','а':'a','α':'a','𝐚':'a','𝒶':'a',
'b':'b','𝐛':'b','𝑏':'b'
};
// 2. Structural patterns
const connector = /(?:[\s_\u200B-\u200F\u202A-\u202E\uFEFF•·°+=:,./-]|\u2014)/.source;
// 3. baseNamePattern
const baseNamePattern = (?:N${connector}o${connector}v${connector}e${connector}[li]${connector}i${connector}g${connector}h${connector}t|r${connector}a${connector}n${connector}o${connector}b${connector}e${connector}s);
// 4. Safe border targeting
const wrap = /(?:[\s\u200B-\u200F\u202A-\u202E\uFEFF/_~✶*❖★☆✧✦✨◆◇■□●○◈▼▲▶◀►◄➤◊❀✿❁✾❃☘♡♥【】《》「」『』〔〕〈〉〘〙«»{}"'“”‘’♪♫✔✖✗✘⊛⊙⊚✪°•.,!?:;|()[]-]|\u2014)/.source;
// 5. Common standalone spam phrases
const spamPhrases = /(?:(Official version)|(Read the full story)|(Original source)|(Don['’]t copy,?\sread here)|(Continue reading)|(Read more on our source)|(\bOnly on\b)|(\bExclusive on\b)|(\bUnauthorized use\b)|(This translation is the intellectual property of)|([([{【《「]\s*[)]}】》」]))/.source;
// 6. Combined Regex
const watermarkRegex = new RegExp(
wrap + (?:(?:This${connector}translation.?of${wrap}|Do${connector}not${connector}rehost.*?${wrap})?${baseNamePattern}(?:${wrap}[\(\[\{【《「].?[\)\]\}】》」])?|${spamPhrases}) + wrap,
'gi'
);
const container = document.querySelector('.chapter-content') || document.querySelector('#chapter-inner') || document.body;
// Use TreeWalker to process raw text nodes directly
const walker = document.createTreeWalker(container, NodeFilter.SHOW_TEXT, null, false);
const textNodes = [];
let node;
while ((node = walker.nextNode())) {
if (node.nodeValue.trim()) textNodes.push(node);
}
textNodes.forEach(textNode => {
let original = textNode.nodeValue;
let text = original.replaceAll('\u2014', ' ');
let sanitized = text;
for (let bad in unicodeMap) {
sanitized = sanitized.replaceAll(bad, unicodeMap[bad].padEnd(bad.length, '\u200B'));
}
sanitized = sanitized.normalize("NFD").replace(/[\u0300-\u036f]/g, "");
let match;
let offset = 0;
let rawText = text;
watermarkRegex.lastIndex = 0;
while ((match = watermarkRegex.exec(sanitized)) !== null) {
if (match[0].length === 0) { watermarkRegex.lastIndex++; continue; }
let start = match.index - offset;
let end = start + match[0].length;
rawText = rawText.substring(0, start) + " " + rawText.substring(end);
offset += (match[0].length - 1);
}
if (rawText !== original) {
textNode.nodeValue = rawText.replace(/ {2,}/g, ' ').trim();
const parent = textNode.parentElement;
if (parent && parent.textContent.trim() === '') parent.remove();
}
});
Acknowledgements
Describe your suggested feature
First, thank you for implementing the new TTS cleanup feature on 2.1.3. It is a massive quality-of-life improvement.
The Problem:
Currently, users who alternate between reading visually and listening via TTS must maintain two separate configurations to filter out obfuscated watermarks and anti-scrape spam: a custom JavaScript TreeWalker for the visible text, and the new JSON ruleset for TTS.
Because these site-specific trash strings and unicode obfuscations change frequently, maintaining both scripts simultaneously requires duplicated effort.
Proposed Solution:
Could we unify the cleanup logic so users only have to maintain one ruleset?
Option A: Allow the TTS JSON ruleset to also apply to and clean up the visible text.
Option B: Allow the visual text cleanup JavaScript to be mirrored and applied to the TTS output.
Below are my current parallel configurations required to achieve the same result for both reading modes:
Other details
{
"format": "lnreader-tts-cleanup",
"version": 1,
"settings": {
"enabled": true,
"normalizeUnicode": true,
"rules": [
{
"id": "tts-rule-msdu7gg1-58",
"enabled": true,
"pattern": "u2014",
"isRegex": false,
"flags": "g",
"replacement": " "
},
{
"id": "tts-rule-msdu7gg1-59",
"enabled": true,
"pattern": "𝐍|𝓝|𝔑|п|η",
"isRegex": true,
"flags": "gi",
"replacement": "N"
},
{
"id": "tts-rule-msdu7gg1-60",
"enabled": true,
"pattern": "о|𝑜|𝐨|𝒐|开启|ο|0",
"isRegex": true,
"flags": "gi",
"replacement": "o"
},
{
"id": "tts-rule-msdu7gg1-61",
"enabled": true,
"pattern": "𝓋|𝐯|𝒗",
"isRegex": true,
"flags": "gi",
"replacement": "v"
},
{
"id": "tts-rule-msdu7gg1-62",
"enabled": true,
"pattern": "е|𝑒|𝐞|𝒆|Ε|ε",
"isRegex": true,
"flags": "gi",
"replacement": "e"
},
{
"id": "tts-rule-msdu7gg1-63",
"enabled": true,
"pattern": "і|ι|𝐢|𝑖|i|𝚒|Ⅰ|𝕚|1|!",
"isRegex": true,
"flags": "gi",
"replacement": "i"
},
{
"id": "tts-rule-msdu7gg1-64",
"enabled": true,
"pattern": "г|𝐠|𝒈",
"isRegex": true,
"flags": "gi",
"replacement": "g"
},
{
"id": "tts-rule-msdu7gg1-65",
"enabled": true,
"pattern": "һ|𝐡|𝒉",
"isRegex": true,
"flags": "gi",
"replacement": "h"
},
{
"id": "tts-rule-msdu7gg1-66",
"enabled": true,
"pattern": "т|𝐭|𝒕",
"isRegex": true,
"flags": "gi",
"replacement": "t"
},
{
"id": "tts-rule-msdu7gg1-67",
"enabled": true,
"pattern": "𝐥|ӏ|ℓ",
"isRegex": true,
"flags": "gi",
"replacement": "l"
},
{
"id": "tts-rule-msdu7gg1-68",
"enabled": true,
"pattern": "ꭆ",
"isRegex": true,
"flags": "gi",
"replacement": "r"
},
{
"id": "tts-rule-msdu7gg1-69",
"enabled": true,
"pattern": "а|α|𝐚|𝒶",
"isRegex": true,
"flags": "gi",
"replacement": "a"
},
{
"id": "tts-rule-msdu7gg1-70",
"enabled": true,
"pattern": "b|𝐛|𝑏",
"isRegex": true,
"flags": "gi",
"replacement": "b"
},
{
"id": "tts-rule-msdu7gg1-71",
"enabled": true,
"pattern": "(?:(?:This[ \t]*translation[^\\n]{0,40}?of|Do[ \t]*not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]*N[^\\w\\n]*o[^\\w\\n]*v[^\\w\\n]e[^\\w\\n](?:l|i)[^\\w\\n]*i[^\\w\\n]*g[^\\w\\n]*h[^\\w\\n]t[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-72",
"enabled": true,
"pattern": "(?:(?:This[ \t]*translation[^\\n]{0,40}?of|Do[ \t]*not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]*r[^\\w\\n]*a[^\\w\\n]n[^\\w\\n]o[^\\w\\n]b[^\\w\\n]e[^\\w\\n]s[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-73",
"enabled": true,
"pattern": "(?:(?:This[ \t]translation[^\\n]{0,40}?of|Do[ \t]not[ \t]rehost[^\\n]{0,40}?)[^\\w\\n])?[^a-zA-Z0-9\\n]n[^\\w\\n]o[^\\w\\n]v[^\\w\\n]e[^\\w\\n]l[^\\w\\n]f[^\\w\\n]i[^\\w\\n]r[^\\w\\n]e[^a-zA-Z0-9\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-74",
"enabled": true,
"pattern": "\b(?:novelfire|ranobes)\.(?:com|net|org|info|cc|ru|biz)\b",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-75",
"enabled": true,
"pattern": "(?:This chapter is updated by|Read latest chapters at|The source of this content is|New novel chapters are published on|Find the original at)[ \t](?:novelfire|ranobes)[a-zA-Z0-9.-]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-76",
"enabled": true,
"pattern": "(?:https?:\/\/)?(?:www\.)?(?:discord\.gg|patreon\.com|ko-fi\.com)[^\\s\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-77",
"enabled": true,
"pattern": "Support (?:me|the (?:author|translator)) on (?:Patreon|Ko-fi)|Buy me a coffee[^\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-78",
"enabled": true,
"pattern": "(?:TL|ED|PR|Author['’]?s?)[ \t]Note[ \t]:",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-79",
"enabled": true,
"pattern": "(?:Find the original at|This novel is translated and hosted on)[^\\n]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg1-80",
"enabled": true,
"pattern": "[\(\[\{【《「](?:This(?: work| translation)? is the )?intellectual property of [^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-81",
"enabled": true,
"pattern": "[\(\[\{【《「]Don['’]?t copy[^\\n\\)\\]\}】》」]read here[^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-82",
"enabled": true,
"pattern": "[\(\[\{【《「](?:Official version|Read the full story|Original source|Continue reading|Read more on our source|Unauthorized use[^\\n\\)\\]\}】》」])[\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
},
{
"id": "tts-rule-msdu7gg2-83",
"enabled": true,
"pattern": "[\(\[\{【《「](?:Only on|Exclusive on) [^\\n\\)\\]\}】》」][\)\]\}】》」]",
"isRegex": true,
"flags": "gi",
"replacement": ""
}
],
"phoneticPairs": []
}
}
Text clean up logic
// 1. Core Target Map
const unicodeMap = {
'𝐍':'N','𝓝':'N','𝔑':'N','п':'n','η':'n',
'о':'o','𝑜':'o','𝐨':'o','𝒐':'o','开启':'o','ο':'o','0':'o',
'𝓋':'v','𝐯':'v','𝒗':'v',
'е':'e','𝑒':'e','𝐞':'e','𝒆':'e','Ε':'e','ε':'e',
'і':'i','ι':'i','𝐢':'i','𝑖':'i','i':'i','𝚒':'i','Ⅰ':'i','𝕚':'i','1':'i','!':'i',
'г':'g','𝐠':'g','𝒈':'g',
'һ':'h','𝐡':'h','𝒉':'h',
'т':'t','𝐭':'t','𝒕':'t',
'𝐥':'l','ӏ':'l','ℓ':'l',
'ꭆ':'r','а':'a','α':'a','𝐚':'a','𝒶':'a',
'b':'b','𝐛':'b','𝑏':'b'
};
// 2. Structural patterns
const connector = /(?:[\s_\u200B-\u200F\u202A-\u202E\uFEFF•·°+=:,./-]|\u2014)/.source;
// 3. baseNamePattern
const baseNamePattern =
(?:N${connector}o${connector}v${connector}e${connector}[li]${connector}i${connector}g${connector}h${connector}t|r${connector}a${connector}n${connector}o${connector}b${connector}e${connector}s);// 4. Safe border targeting
const wrap = /(?:[\s\u200B-\u200F\u202A-\u202E\uFEFF/_~✶*❖★☆✧✦✨◆◇■□●○◈▼▲▶◀►◄➤◊❀✿❁✾❃☘♡♥【】《》「」『』〔〕〈〉〘〙«»{}"'“”‘’♪♫✔✖✗✘⊛⊙⊚✪°•.,!?:;|()[]-]|\u2014)/.source;
// 5. Common standalone spam phrases
const spamPhrases = /(?:(Official version)|(Read the full story)|(Original source)|(Don['’]t copy,?\sread here)|(Continue reading)|(Read more on our source)|(\bOnly on\b)|(\bExclusive on\b)|(\bUnauthorized use\b)|(This translation is the intellectual property of)|([([{【《「]\s*[)]}】》」]))/.source;
// 6. Combined Regex
const watermarkRegex = new RegExp(
wrap +
(?:(?:This${connector}translation.?of${wrap}|Do${connector}not${connector}rehost.*?${wrap})?${baseNamePattern}(?:${wrap}[\(\[\{【《「].?[\)\]\}】》」])?|${spamPhrases})+ wrap,'gi'
);
const container = document.querySelector('.chapter-content') || document.querySelector('#chapter-inner') || document.body;
// Use TreeWalker to process raw text nodes directly
const walker = document.createTreeWalker(container, NodeFilter.SHOW_TEXT, null, false);
const textNodes = [];
let node;
while ((node = walker.nextNode())) {
if (node.nodeValue.trim()) textNodes.push(node);
}
textNodes.forEach(textNode => {
let original = textNode.nodeValue;
let text = original.replaceAll('\u2014', ' ');
let sanitized = text;
for (let bad in unicodeMap) {
sanitized = sanitized.replaceAll(bad, unicodeMap[bad].padEnd(bad.length, '\u200B'));
}
sanitized = sanitized.normalize("NFD").replace(/[\u0300-\u036f]/g, "");
let match;
let offset = 0;
let rawText = text;
watermarkRegex.lastIndex = 0;
while ((match = watermarkRegex.exec(sanitized)) !== null) {
if (match[0].length === 0) { watermarkRegex.lastIndex++; continue; }
let start = match.index - offset;
let end = start + match[0].length;
rawText = rawText.substring(0, start) + " " + rawText.substring(end);
offset += (match[0].length - 1);
}
if (rawText !== original) {
textNode.nodeValue = rawText.replace(/ {2,}/g, ' ').trim();
const parent = textNode.parentElement;
if (parent && parent.textContent.trim() === '') parent.remove();
}
});
Acknowledgements