Skip to content

Commit 4d279f0

Browse files
committed
Redesign automod scoring and warning appeals
1 parent 8ce0581 commit 4d279f0

6 files changed

Lines changed: 562 additions & 132 deletions

File tree

‎README.md‎

Lines changed: 36 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -15,8 +15,8 @@ community operations:
1515

1616
- slash-command moderation for bans, kicks, timeouts, warnings, purge, locks,
1717
slowmode, nicknames, and role tools
18-
- automod protections for spam, duplicate messages, invite links, blocked
19-
words, caps abuse, mention flooding, and raid-mode responses
18+
- context-aware automod protections for spam, duplicate messages, suspicious
19+
links, invite abuse, blocked words, mention flooding, and raid-mode responses
2020
- security guardrails for anti-nuke detection, audit logging, and SQLite
2121
backups
2222
- startup seeding for bundled and curated automod datasets after deploys or
@@ -38,8 +38,8 @@ This repository's source code is open source under the MIT license. See
3838
slowmode, nicknames, and role tools
3939
- SQLite-backed case history, warning points, temp-ban scheduling, and server
4040
config
41-
- automod for spam, duplicate messages, invite links, blocked words, caps, and
42-
mention flooding
41+
- scoring-based automod for spam, duplicate messages, suspicious links, invite
42+
abuse, blocked words, caps context, and mention flooding
4343
- anti-nuke protection for destructive server bursts such as mass bans, kicks,
4444
channel deletes, and role deletes
4545
- richer audit logging for message edits/deletes, role changes, channel
@@ -94,6 +94,38 @@ public Bluesky account into the fixed Discord channel `1490277253949558975`.
9494
The relay uses Bluesky's public AppView HTTP endpoint, so no extra Bluesky
9595
credentials are required for read-only mirroring.
9696

97+
## Moderation Model
98+
99+
Memact AutoMod uses a conservative scoring model instead of warning members for
100+
every single keyword or formatting trigger. The bot weighs message context,
101+
member trust, account age, join age, links, mentions, repeated behavior, and
102+
raid mode before deciding what to do.
103+
104+
In practice:
105+
106+
- all-caps text alone does not create warning points
107+
- GIFs and normal media-only messages are ignored by automod
108+
- words such as casino, sale, promo, or giveaway are not punished by themselves
109+
- invite links and promotional links are usually deleted/logged first, not
110+
instantly converted into warning points
111+
- repeated soft violations can become a warning if the same member keeps doing
112+
it in a short window
113+
- scam-link patterns, message floods, repeated spam, and mass mentions still
114+
trigger stronger actions
115+
116+
Useful staff commands:
117+
118+
- `/automod view`: show current automod mode and thresholds
119+
- `/automod toggle`: enable or disable specific filters
120+
- `/mod warnings`: show a member's active warning points and active warning
121+
cases
122+
- `/mod unwarn_latest`: revoke the latest active warning for a member without
123+
hunting for a case ID
124+
- `/mod unwarn`: revoke a specific warning when staff already know the case ID
125+
- `/mod clearwarns`: clear all active warnings for a member
126+
- `/appeal reason:<text>`: appeal the user's latest active moderation case
127+
- `/appeal reason:<text> case_id:<id>`: appeal a specific case
128+
97129
### Persistence
98130

99131
The catch-up state is stored in the same SQLite database as the rest of the

‎cogs/automod.py‎

Lines changed: 93 additions & 120 deletions
Original file line numberDiff line numberDiff line change
@@ -20,13 +20,10 @@
2020
fetch_dataset_terms_sync,
2121
fetch_lenient_terms_sync,
2222
)
23+
from utils.moderation_engine import AutomodDecision, evaluate_automod
2324
from utils.ui import build_embed, send_interaction
2425

2526

26-
INVITE_RE = re.compile(r"(discord\.gg|discord\.com/invite)/[A-Za-z0-9-]+", re.IGNORECASE)
27-
URL_RE = re.compile(r"https?://\S+", re.IGNORECASE)
28-
SUSPICIOUS_LINK_TOKENS = ("discord-gifts", "nitro-free", "steamcomrnunity", "free-nitro", "claim-prize")
29-
3027
PROMO_KEYWORDS = (
3128
"promo",
3229
"promotion",
@@ -59,19 +56,6 @@
5956
"commission",
6057
)
6158

62-
PROMO_PHRASE_RE = re.compile(
63-
r"\b("
64-
r"join (my|our) (server|discord)"
65-
r"|follow (me|us)"
66-
r"|subscribe to (my|our)"
67-
r"|check out (my|our)"
68-
r"|visit (my|our)"
69-
r"|use code"
70-
r"|limited time"
71-
r"|free giveaway"
72-
r")\b",
73-
re.IGNORECASE,
74-
)
7559
PROMO_DATASET_PATH = Path(__file__).resolve().parent.parent / "data" / "promo_keywords.txt"
7660

7761

@@ -80,6 +64,7 @@ def __init__(self, bot: MemactAutoModBot) -> None:
8064
self.bot = bot
8165
self.spam_history: dict[tuple[int, int], deque[float]] = defaultdict(deque)
8266
self.repeat_history: dict[tuple[int, int], deque[tuple[float, str]]] = defaultdict(deque)
67+
self.soft_violation_history: dict[tuple[int, int], deque[float]] = defaultdict(deque)
8368
self.blocked_word_cache: dict[int, list[tuple[str, re.Pattern[str]]]] = {}
8469
self._last_history_cleanup_ts = 0.0
8570
self._dataset_seed_task: asyncio.Task | None = None
@@ -185,6 +170,7 @@ def _clear_member_history(self, guild_id: int, user_id: int) -> None:
185170
key = (guild_id, user_id)
186171
self.spam_history.pop(key, None)
187172
self.repeat_history.pop(key, None)
173+
self.soft_violation_history.pop(key, None)
188174

189175
async def _assign_join_role(self, member: nextcord.Member, role_id: int) -> bool:
190176
role = member.guild.get_role(role_id)
@@ -268,38 +254,73 @@ async def _acknowledge_intro_message(self, message: nextcord.Message) -> None:
268254
f"in guild {message.guild.id}: {type(error).__name__}: {error}"
269255
)
270256

271-
async def _handle_violation(
257+
def _record_soft_violation(self, guild_id: int, user_id: int, now_ts: float) -> int:
258+
key = (guild_id, user_id)
259+
history = self.soft_violation_history[key]
260+
history.append(now_ts)
261+
while history and now_ts - history[0] > 900:
262+
history.popleft()
263+
if not history:
264+
self.soft_violation_history.pop(key, None)
265+
return len(history)
266+
267+
async def _handle_automod_decision(
272268
self,
273269
message: nextcord.Message,
274270
*,
275-
rule_name: str,
276-
reason: str,
277-
points: int,
271+
decision: AutomodDecision,
278272
) -> None:
279273
if not isinstance(message.author, nextcord.Member):
280274
return
281-
try:
282-
await message.delete()
283-
except (nextcord.Forbidden, nextcord.HTTPException):
284-
pass
285-
moderator = self.bot.user or message.author
286-
await self.bot.apply_warning(
275+
if decision.action == "none":
276+
return
277+
278+
if decision.delete_message:
279+
try:
280+
await message.delete()
281+
except (nextcord.Forbidden, nextcord.HTTPException):
282+
pass
283+
284+
warning_points = decision.warn_points
285+
soft_count = 0
286+
if warning_points <= 0 and decision.soft_strike:
287+
soft_count = self._record_soft_violation(
288+
message.guild.id,
289+
message.author.id,
290+
message.created_at.timestamp(),
291+
)
292+
if soft_count >= 3:
293+
warning_points = 1
294+
295+
if warning_points > 0:
296+
moderator = self.bot.user or message.author
297+
reason = decision.reason
298+
if soft_count >= 3:
299+
reason = f"Repeated soft automod violations ({soft_count}/3). Latest: {decision.reason}"
300+
await self.bot.apply_warning(
301+
message.guild,
302+
message.author,
303+
moderator=moderator,
304+
reason=reason,
305+
points=warning_points,
306+
source="automod",
307+
rule_name="Automod scoring",
308+
)
309+
return
310+
311+
action_label = "Deleted" if decision.delete_message else "Logged"
312+
await self.bot.send_log(
287313
message.guild,
288-
message.author,
289-
moderator=moderator,
290-
reason=reason,
291-
points=points,
292-
source="automod",
293-
rule_name=rule_name,
314+
title=f"Automod {action_label}",
315+
description=decision.reason,
316+
fields=[
317+
("Member", message.author.mention, True),
318+
("Channel", message.channel.mention, True),
319+
("Score", f"{decision.score:.1f}", True),
320+
("Signals", ", ".join(signal.kind for signal in decision.signals), False),
321+
],
294322
)
295323

296-
def _caps_ratio(self, text: str) -> tuple[int, float]:
297-
letters = [char for char in text if char.isalpha()]
298-
if not letters:
299-
return 0, 0.0
300-
uppercase = sum(1 for char in letters if char.isupper())
301-
return len(letters), uppercase / len(letters)
302-
303324
def _check_spam(self, guild_id: int, user_id: int, content: str, *, config: dict, now_ts: float) -> tuple[bool, str | None]:
304325
self._cleanup_history_cache(
305326
now_ts,
@@ -326,21 +347,10 @@ def _check_spam(self, guild_id: int, user_id: int, content: str, *, config: dict
326347
if not repeat_history:
327348
self.repeat_history.pop(key, None)
328349
if normalized and sum(1 for _, value in repeat_history if value == normalized) >= config["repeat_threshold"]:
329-
return True, "No spam"
350+
return True, "Repeat spam"
330351

331352
return False, None
332353

333-
def _check_promotion(self, guild_id: int, content: str) -> bool:
334-
normalized = " ".join(content.lower().split())
335-
promo_keywords = set(PROMO_KEYWORDS)
336-
promo_keywords.update(self.bot.db.list_promo_keywords(guild_id))
337-
multiword_keywords = [term for term in promo_keywords if " " in term]
338-
has_url = URL_RE.search(content) is not None
339-
if has_url:
340-
return any(keyword in normalized for keyword in promo_keywords) or PROMO_PHRASE_RE.search(normalized) is not None
341-
return PROMO_PHRASE_RE.search(normalized) is not None or any(term in normalized for term in multiword_keywords)
342-
343-
344354
@commands.Cog.listener()
345355
async def on_message(self, message: nextcord.Message) -> None:
346356
if message.guild is None or message.author.bot:
@@ -355,64 +365,8 @@ async def on_message(self, message: nextcord.Message) -> None:
355365
should_run_automod = config["automod_enabled"] and not is_moderator_member(message.author, config)
356366

357367
if should_run_automod and content:
358-
normalized_content = content.casefold()
359-
for word, pattern in self._get_blocked_word_patterns(message.guild.id):
360-
if pattern.search(normalized_content):
361-
await self._handle_violation(
362-
message,
363-
rule_name="Be respectful",
364-
reason=f"Blocked word detected: `{word}`.",
365-
points=2,
366-
)
367-
return
368-
369-
if any(token in normalized_content for token in SUSPICIOUS_LINK_TOKENS) and URL_RE.search(content):
370-
await self._handle_violation(
371-
message,
372-
rule_name="No scams or malicious links",
373-
reason="Suspicious link pattern detected.",
374-
points=3,
375-
)
376-
return
377-
378-
if config["invite_filter_enabled"] and INVITE_RE.search(content):
379-
await self._handle_violation(
380-
message,
381-
rule_name="No unsolicited advertising",
382-
reason="Invite link detected without approval.",
383-
points=1,
384-
)
385-
return
386-
387-
if self._check_promotion(message.guild.id, content):
388-
await self._handle_violation(
389-
message,
390-
rule_name="No unsolicited advertising",
391-
reason="Promotional content detected.",
392-
points=1,
393-
)
394-
return
395-
396-
if config["mention_filter_enabled"] and len(message.mentions) >= config["mention_threshold"]:
397-
await self._handle_violation(
398-
message,
399-
rule_name="No spam",
400-
reason=f"Mass mention detected with {len(message.mentions)} mentions.",
401-
points=1,
402-
)
403-
return
404-
405-
if config["caps_filter_enabled"]:
406-
letter_count, caps_ratio = self._caps_ratio(content)
407-
if letter_count >= config["caps_min_length"] and caps_ratio >= config["caps_ratio"]:
408-
await self._handle_violation(
409-
message,
410-
rule_name="No spam",
411-
reason=f"Excessive caps detected at {caps_ratio:.0%}.",
412-
points=1,
413-
)
414-
return
415-
368+
spam_triggered = False
369+
repeat_triggered = False
416370
if config["spam_filter_enabled"] or config["repeat_filter_enabled"]:
417371
triggered, rule_name = self._check_spam(
418372
message.guild.id,
@@ -421,14 +375,30 @@ async def on_message(self, message: nextcord.Message) -> None:
421375
config=config,
422376
now_ts=message.created_at.timestamp(),
423377
)
424-
if triggered:
425-
await self._handle_violation(
426-
message,
427-
rule_name=rule_name or "No spam",
428-
reason="Spam or repeated message threshold reached.",
429-
points=1,
430-
)
431-
return
378+
spam_triggered = bool(triggered and rule_name == "No spam")
379+
repeat_triggered = bool(triggered and rule_name != "No spam")
380+
381+
now = message.created_at
382+
account_age_hours = (now - message.author.created_at).total_seconds() / 3600
383+
joined_at = message.author.joined_at or now
384+
joined_age_hours = (now - joined_at).total_seconds() / 3600
385+
promo_keywords = set(PROMO_KEYWORDS)
386+
promo_keywords.update(self.bot.db.list_promo_keywords(message.guild.id))
387+
decision = evaluate_automod(
388+
content=content,
389+
config=config,
390+
blocked_patterns=self._get_blocked_word_patterns(message.guild.id),
391+
promo_keywords=promo_keywords,
392+
mention_count=len(message.mentions),
393+
account_age_hours=account_age_hours,
394+
joined_age_hours=joined_age_hours,
395+
has_attachments=bool(message.attachments or message.stickers),
396+
spam_triggered=spam_triggered,
397+
repeat_triggered=repeat_triggered,
398+
)
399+
if decision.action != "none":
400+
await self._handle_automod_decision(message, decision=decision)
401+
return
432402

433403

434404
if message.channel.id == INTRO_CHANNEL_ID:
@@ -482,6 +452,8 @@ async def on_guild_remove(self, guild: nextcord.Guild) -> None:
482452
self.spam_history.pop(key, None)
483453
for key in [key for key in self.repeat_history if key[0] == guild.id]:
484454
self.repeat_history.pop(key, None)
455+
for key in [key for key in self.soft_violation_history if key[0] == guild.id]:
456+
self.soft_violation_history.pop(key, None)
485457
self.blocked_word_cache.pop(guild.id, None)
486458

487459
@nextcord.slash_command(
@@ -513,10 +485,11 @@ async def view(self, interaction: nextcord.Interaction) -> None:
513485
("Spam Filter", "On" if config["spam_filter_enabled"] else "Off", True),
514486
("Repeat Filter", "On" if config["repeat_filter_enabled"] else "Off", True),
515487
("Mention Filter", "On" if config["mention_filter_enabled"] else "Off", True),
488+
("Mode", "Scored and conservative", True),
516489
("Blocked Words", f"{len(blocked_words)} configured" if blocked_words else "None", False),
517490
("Lenient Words", f"{len(lenient_words)} allowlisted" if lenient_words else "None", False),
518491
("Promo Keywords", f"{len(promo_keywords)} configured" if promo_keywords else "None", False),
519-
("Caps Threshold", f"{config['caps_ratio']:.0%} with minimum {config['caps_min_length']} letters", False),
492+
("Caps Threshold", f"{config['caps_ratio']:.0%} with minimum {max(config['caps_min_length'], 24)} letters", False),
520493
("Spam Threshold", f"{config['spam_threshold']} messages / {config['spam_window_seconds']}s", False),
521494
("Repeat Threshold", f"{config['repeat_threshold']} duplicates / {config['repeat_window_seconds']}s", False),
522495
("Mention Threshold", str(config["mention_threshold"]), False),
@@ -775,7 +748,7 @@ async def settings(
775748
self,
776749
interaction: nextcord.Interaction,
777750
caps_ratio_percent: int = nextcord.SlashOption(required=False, default=75, min_value=1, max_value=100),
778-
caps_min_length: int = nextcord.SlashOption(required=False, default=12, min_value=1, max_value=200),
751+
caps_min_length: int = nextcord.SlashOption(required=False, default=24, min_value=24, max_value=200),
779752
mention_threshold: int = nextcord.SlashOption(required=False, default=5, min_value=1, max_value=50),
780753
spam_threshold: int = nextcord.SlashOption(required=False, default=6, min_value=2, max_value=50),
781754
spam_window_seconds: int = nextcord.SlashOption(required=False, default=12, min_value=2, max_value=300),

0 commit comments

Comments
 (0)