From 39ef91354378b34d9b1dae3c4dcd8b9dfd82dfe9 Mon Sep 17 00:00:00 2001 From: tdgao Date: Mon, 24 Aug 2026 15:19:09 -0600 Subject: [PATCH 01/40] feat: implement profanity and non-standard text validators --- packages/moderation/package.json | 2 + packages/moderation/src/index.ts | 2 + .../src/validators/non-standard-text/index.ts | 238 ++++++++ .../src/validators/non-standard-text/tests.ts | 118 ++++ .../src/validators/profanity/index.ts | 572 ++++++++++++++++++ .../src/validators/profanity/tests.ts | 123 ++++ 6 files changed, 1055 insertions(+) create mode 100644 packages/moderation/src/validators/non-standard-text/index.ts create mode 100644 packages/moderation/src/validators/non-standard-text/tests.ts create mode 100644 packages/moderation/src/validators/profanity/index.ts create mode 100644 packages/moderation/src/validators/profanity/tests.ts diff --git a/packages/moderation/package.json b/packages/moderation/package.json index 5409c54d5e..a7a05f3023 100644 --- a/packages/moderation/package.json +++ b/packages/moderation/package.json @@ -2,10 +2,12 @@ "name": "@modrinth/moderation", "version": "0.0.0", "private": true, + "type": "module", "main": "./src/index.ts", "scripts": { "lint": "eslint . && prettier --check .", "fix": "eslint . --fix && prettier --write .", + "test": "node --test src/validators/*/tests.ts", "intl:extract": "formatjs extract \"**/*.{vue,ts,tsx,js,jsx,mts,cts,mjs,cjs}\" --ignore \"**/*.d.ts\" --ignore \"node_modules/**/*\" --out-file src/locales/en-US/index.json --preserve-whitespace", "intl:prune-local": "pnpm -w scripts i18n-icu-contract prune-local --scope packages/moderation" }, diff --git a/packages/moderation/src/index.ts b/packages/moderation/src/index.ts index 08861dbf4c..5a9611071e 100644 --- a/packages/moderation/src/index.ts +++ b/packages/moderation/src/index.ts @@ -19,3 +19,5 @@ export * from './types/quick-reply' export * from './types/reports' export * from './types/settings' export * from './utils' +export * from './validators/non-standard-text' +export * from './validators/profanity' diff --git a/packages/moderation/src/validators/non-standard-text/index.ts b/packages/moderation/src/validators/non-standard-text/index.ts new file mode 100644 index 0000000000..72a46c84e6 --- /dev/null +++ b/packages/moderation/src/validators/non-standard-text/index.ts @@ -0,0 +1,238 @@ +// the following are non-stardard text that are detected +export type NonStandardTextIssueKind = + | 'fancy' // styled characters such as `𝐀`, `Ⓐ`, or `A`. + | 'zalgo' // detached or excessive combining marks such as `a̴̵̶`. + | 'invisible' // hidden formatting such as a zero-width space or bidi override. + | 'control' // disallowed control characters such as a null byte. + | 'private-use' // characters from Unicode private-use areas. + | 'unassigned' // code points with no assigned Unicode character. + | 'surrogate' // malformed standalone UTF-16 surrogate code units. + +export interface NonStandardTextIssue { + kind: NonStandardTextIssueKind + character: string + codePoint: string + index: number +} + +export interface NonStandardTextResult { + valid: boolean + issues: NonStandardTextIssue[] + counts: Record +} + +export interface NonStandardTextOptions { + allowNewlines?: boolean + allowTabs?: boolean + maxCombiningMarksPerCharacter?: number +} + +export const DEFAULT_MAX_COMBINING_MARKS_PER_CHARACTER = 2 + +const FANCY_RANGES: ReadonlyArray = [ + [0x02b0, 0x02ff], + [0x1d400, 0x1d7ff], + [0x2460, 0x24ff], + [0x2070, 0x209f], + [0x2100, 0x214f], + [0xfb00, 0xfb06], + [0xff01, 0xff60], + [0x1f100, 0x1f1ad], +] + +const MARK_PATTERN = /\p{M}/u +const CONTROL_PATTERN = /\p{Cc}/u +const FORMAT_PATTERN = /\p{Cf}/u +const PRIVATE_USE_PATTERN = /\p{Co}/u +const UNASSIGNED_PATTERN = /\p{Cn}/u +const LETTER_PATTERN = /\p{L}/u +const EXTENDED_PICTOGRAPHIC_PATTERN = /\p{Extended_Pictographic}/u +const UNIFIED_IDEOGRAPH_PATTERN = /\p{Unified_Ideograph}/u + +function createCounts(): Record { + return { + fancy: 0, + zalgo: 0, + invisible: 0, + control: 0, + 'private-use': 0, + unassigned: 0, + surrogate: 0, + } +} + +function isInRanges(codePoint: number, ranges: ReadonlyArray) { + return ranges.some(([start, end]) => codePoint >= start && codePoint <= end) +} + +function isVariationSelector(codePoint: number) { + return ( + (codePoint >= 0xfe00 && codePoint <= 0xfe0f) || (codePoint >= 0xe0100 && codePoint <= 0xe01ef) + ) +} + +function isEmojiModifier(codePoint: number) { + return codePoint >= 0x1f3fb && codePoint <= 0x1f3ff +} + +function isAscii(character: string) { + return character.codePointAt(0)! <= 0x7f +} + +function isAllowedZeroWidthNonJoiner( + characters: readonly string[], + characterIndex: number, +): boolean { + const previous = characters[characterIndex - 1] + const next = characters[characterIndex + 1] + if (!previous || !next || !LETTER_PATTERN.test(previous) || !LETTER_PATTERN.test(next)) { + return false + } + return !isAscii(previous) || !isAscii(next) +} + +function findAdjacentEmojiCharacter( + characters: readonly string[], + start: number, + direction: -1 | 1, +): string | undefined { + for (let index = start; index >= 0 && index < characters.length; index += direction) { + const character = characters[index] + const codePoint = character.codePointAt(0)! + if (isVariationSelector(codePoint) || isEmojiModifier(codePoint)) continue + return character + } + return undefined +} + +function isAllowedZeroWidthJoiner(characters: readonly string[], characterIndex: number): boolean { + const previous = findAdjacentEmojiCharacter(characters, characterIndex - 1, -1) + const next = findAdjacentEmojiCharacter(characters, characterIndex + 1, 1) + return ( + previous !== undefined && + next !== undefined && + EXTENDED_PICTOGRAPHIC_PATTERN.test(previous) && + EXTENDED_PICTOGRAPHIC_PATTERN.test(next) + ) +} + +function isAllowedVariationSelector( + characters: readonly string[], + characterIndex: number, + codePoint: number, +): boolean { + const previous = characters[characterIndex - 1] + if (!previous) return false + if (codePoint >= 0xfe00 && codePoint <= 0xfe0f) { + return EXTENDED_PICTOGRAPHIC_PATTERN.test(previous) || /^[0-9#*]$/u.test(previous) + } + return UNIFIED_IDEOGRAPH_PATTERN.test(previous) +} + +function codePointLabel(codePoint: number) { + return `U+${codePoint.toString(16).toUpperCase().padStart(4, '0')}` +} + +export function validateNonStandardText( + text: string, + options: NonStandardTextOptions = {}, +): NonStandardTextResult { + const allowNewlines = options.allowNewlines ?? true + const allowTabs = options.allowTabs ?? true + const maxCombiningMarks = + options.maxCombiningMarksPerCharacter ?? DEFAULT_MAX_COMBINING_MARKS_PER_CHARACTER + if (!Number.isInteger(maxCombiningMarks) || maxCombiningMarks < 0) { + throw new Error('Maximum combining marks must be a non-negative integer') + } + + const issues: NonStandardTextIssue[] = [] + const counts = createCounts() + const characters = Array.from(text) + let utf16Index = 0 + let hasBaseCharacter = false + let combiningMarkCount = 0 + + function addIssue( + kind: NonStandardTextIssueKind, + character: string, + codePoint: number, + index: number, + ) { + issues.push({ + kind, + character, + codePoint: codePointLabel(codePoint), + index, + }) + counts[kind]++ + } + + for (let characterIndex = 0; characterIndex < characters.length; characterIndex++) { + const character = characters[characterIndex] + const codePoint = character.codePointAt(0)! + const currentIndex = utf16Index + utf16Index += character.length + + if (codePoint >= 0xd800 && codePoint <= 0xdfff) { + addIssue('surrogate', character, codePoint, currentIndex) + hasBaseCharacter = false + combiningMarkCount = 0 + continue + } + + if (PRIVATE_USE_PATTERN.test(character)) { + addIssue('private-use', character, codePoint, currentIndex) + } else if (UNASSIGNED_PATTERN.test(character)) { + addIssue('unassigned', character, codePoint, currentIndex) + } + + if (CONTROL_PATTERN.test(character)) { + const allowedNewline = allowNewlines && (character === '\n' || character === '\r') + const allowedTab = allowTabs && character === '\t' + if (!allowedNewline && !allowedTab) { + addIssue('control', character, codePoint, currentIndex) + } + hasBaseCharacter = false + combiningMarkCount = 0 + continue + } + + if (FORMAT_PATTERN.test(character)) { + const allowed = + (codePoint === 0x200c && isAllowedZeroWidthNonJoiner(characters, characterIndex)) || + (codePoint === 0x200d && isAllowedZeroWidthJoiner(characters, characterIndex)) + if (!allowed) addIssue('invisible', character, codePoint, currentIndex) + hasBaseCharacter = false + combiningMarkCount = 0 + continue + } + + if (MARK_PATTERN.test(character)) { + if (isVariationSelector(codePoint)) { + if (!isAllowedVariationSelector(characters, characterIndex, codePoint)) { + addIssue('invisible', character, codePoint, currentIndex) + } + continue + } + + combiningMarkCount++ + if (!hasBaseCharacter || combiningMarkCount > maxCombiningMarks) { + addIssue('zalgo', character, codePoint, currentIndex) + } + continue + } + + combiningMarkCount = 0 + hasBaseCharacter = !/^\s$/u.test(character) + + if (isInRanges(codePoint, FANCY_RANGES)) { + addIssue('fancy', character, codePoint, currentIndex) + } + } + + return { + valid: issues.length === 0, + issues, + counts, + } +} diff --git a/packages/moderation/src/validators/non-standard-text/tests.ts b/packages/moderation/src/validators/non-standard-text/tests.ts new file mode 100644 index 0000000000..9352871df1 --- /dev/null +++ b/packages/moderation/src/validators/non-standard-text/tests.ts @@ -0,0 +1,118 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import { validateNonStandardText } from './index.ts' + +test('accepts ordinary multilingual text and punctuation', () => { + const result = validateNonStandardText( + 'Hello, “world”! Français — Русский — العربية — 中文 — 日本語', + ) + + assert.equal(result.valid, true) + assert.deepEqual(result.issues, []) +}) + +test('accepts composed and normally decomposed accents', () => { + assert.equal(validateNonStandardText('café').valid, true) + assert.equal(validateNonStandardText('cafe\u0301').valid, true) + assert.equal(validateNonStandardText('a\u0301\u0327').valid, true) +}) + +test('detects excessive and leading combining marks as zalgo text', () => { + const excessive = validateNonStandardText('a\u0301\u0327\u0308') + const leading = validateNonStandardText('\u0301text') + + assert.equal(excessive.valid, false) + assert.equal(excessive.counts.zalgo, 1) + assert.equal(excessive.issues[0].index, 3) + assert.equal(leading.counts.zalgo, 1) +}) + +test('supports a custom combining-mark threshold', () => { + assert.equal( + validateNonStandardText('a\u0301\u0327', { + maxCombiningMarksPerCharacter: 1, + }).counts.zalgo, + 1, + ) + assert.throws( + () => validateNonStandardText('text', { maxCombiningMarksPerCharacter: -1 }), + /non-negative integer/, + ) +}) + +test('detects common fancy alphabets and presentation forms', () => { + const result = validateNonStandardText('𝐇 Ⓗ ʰ ℌ h ff') + + assert.equal(result.valid, false) + assert.equal(result.counts.fancy, 6) + assert.deepEqual( + result.issues.map(({ codePoint }) => codePoint), + ['U+1D407', 'U+24BD', 'U+02B0', 'U+210C', 'U+FF48', 'U+FB00'], + ) +}) + +test('allows ordinary emoji and valid emoji joiner sequences', () => { + assert.equal(validateNonStandardText('Hello 👋🏽').valid, true) + assert.equal(validateNonStandardText('Family: 👨‍👩‍👧‍👦').valid, true) + assert.equal(validateNonStandardText('Developer: 🧑🏽‍💻').valid, true) + assert.equal(validateNonStandardText('Heart: ❤️').valid, true) +}) + +test('detects suspicious invisible and directional characters', () => { + const result = validateNonStandardText('ab\u200Bcd\u202Eef\u2060gh f\uFE0F') + + assert.equal(result.counts.invisible, 4) + assert.deepEqual( + result.issues.map(({ codePoint }) => codePoint), + ['U+200B', 'U+202E', 'U+2060', 'U+FE0F'], + ) +}) + +test('allows contextual non-joiners but catches ASCII separator evasion', () => { + assert.equal(validateNonStandardText('می‌خواهم').valid, true) + assert.equal(validateNonStandardText('f‌uck').counts.invisible, 1) + assert.equal(validateNonStandardText('a‍b').counts.invisible, 1) +}) + +test('allows newlines and tabs by default and can reject them', () => { + assert.equal(validateNonStandardText('line one\n\tline two').valid, true) + + const result = validateNonStandardText('line one\n\tline two', { + allowNewlines: false, + allowTabs: false, + }) + assert.equal(result.counts.control, 2) +}) + +test('detects other disallowed control characters', () => { + const result = validateNonStandardText(`hello\u0000world`) + + assert.equal(result.counts.control, 1) + assert.equal(result.issues[0].codePoint, 'U+0000') +}) + +test('detects private-use, unassigned, and lone surrogate code points', () => { + const privateUse = validateNonStandardText('\uE000') + const unassigned = validateNonStandardText('\uFDD0') + const surrogate = validateNonStandardText('\uD800') + + assert.equal(privateUse.counts['private-use'], 1) + assert.equal(unassigned.counts.unassigned, 1) + assert.equal(surrogate.counts.surrogate, 1) +}) + +test('reports UTF-16 indexes consistently around astral characters', () => { + const result = validateNonStandardText('🙂\u200Btext') + + assert.equal(result.issues[0].index, 2) +}) + +test('reports multiple issue categories in source order', () => { + const result = validateNonStandardText('𝐀\u200B\u0000') + + assert.deepEqual( + result.issues.map(({ kind }) => kind), + ['fancy', 'invisible', 'control'], + ) +}) diff --git a/packages/moderation/src/validators/profanity/index.ts b/packages/moderation/src/validators/profanity/index.ts new file mode 100644 index 0000000000..8c897dec8c --- /dev/null +++ b/packages/moderation/src/validators/profanity/index.ts @@ -0,0 +1,572 @@ +export type ProfanityKind = 'profanity' | 'slur' + +export interface ProfanityPattern { + kind: ProfanityKind + exceptions: readonly string[] +} + +export interface ProfanityConfig { + patterns: Readonly> +} + +export interface ProfanityMatch { + kind: ProfanityKind + term: string + sanitizedStart: number + sanitizedEnd: number +} + +export interface ProfanityResult { + valid: boolean + profanityCount: number + slurCount: number + firstMatch?: ProfanityMatch + matches: ProfanityMatch[] +} + +export interface ProfanityValidator { + sanitize(text: string): string + findFirst(text: string): ProfanityMatch | undefined + findAll(text: string): ProfanityMatch[] + validate(text: string): ProfanityResult +} + +interface NegativeMatch { + prefix: string + suffix: string +} + +interface TrieNode { + children: Map + negatives: NegativeMatch[] + terminal?: { + kind: ProfanityKind + term: string + } +} + +const CHARACTER_REPLACEMENTS: Readonly> = { + '4': 'a', + '@': 'a', + '3': 'e', + '1': 'i', + '0': 'o', + '5': 's', + '7': 't', + '8': 'b', + '9': 'g', + '+': 't', + $: 's', + '(': 'c', + '{': 'c', + '[': 'c', + '!': 'i', + '|': 'i', + '£': 'e', + '€': 'e', + '¥': 'y', + '¢': 'c', + '<': 'c', +} + +const MULTI_CHARACTER_REPLACEMENTS: Readonly>>> = { + '(': { ')': 'o' }, + '[': { ']': 'o' }, + '{': { '}': 'o' }, + '<': { '>': 'o' }, +} + +const TERM_EXCEPTIONS: Readonly> = { + anuslick: [], + arsehol: [], + arselick: [], + asslick: [], + arsch: [], + asshol: [], + auschwitz: [], + beaner: [], + bestiality: [], + baise: [], + bakachon: [], + bakatyon: [], + bastard: ['bastardized'], + bitch: [], + btch: [], + biatch: [], + bussy: [], + blowjob: [], + blowme: [], + bukakke: [], + buttplug: [], + buttchug: [], + butagorosi: [], + cagada: [], + caralho: [], + cameljockey: [], + castrate: [], + cazzo: [], + ceemen: [], + chankoro: [], + chink: [], + chingchong: [], + choad: [], + chode: [], + chlamydia: [], + clit: ['clitheroe'], + clitoris: [], + cock: [ + 'cockade', + 'cockatiel', + 'cockatiels', + 'cockatoo', + 'cockatoos', + 'cockatrice', + 'cockayne', + 'cockburn', + 'cockcroft', + 'cocked', + 'cocker', + 'cockerel', + 'cockers', + 'cockeyed', + 'cockiness', + 'cocking', + 'cocklebur', + 'cockney', + 'cockpit', + 'cockpits', + 'cockroach', + 'cockroaches', + 'cockscomb', + 'cockspur', + 'cocktail', + 'gamecock', + 'hancock', + 'haycock', + 'hitchcock', + 'leacock', + 'peacock', + 'shuttlecock', + 'stopcock', + 'woodcock', + ], + coon: ['cocoon', 'laocoon', 'raccoon', 'racoon', 'tycoon'], + cocain: [], + coitus: [], + cottonpic: [], + cottonpik: [], + cum: [ + 'acumen', + 'acuminate', + 'altocumulus', + 'cumber', + 'cumbing', + 'cumbria', + 'cumbrian', + 'cumbrous', + 'cummerbund', + 'cumming', + 'cumulat', + 'cumuli', + 'cumulonimbus', + 'cumulus', + 'encumber', + 'encumbrance', + 'scumbag', + 'locum', + 'modicum', + 'magnacumlaude', + 'macumba', + 'practicum', + 'recumbent', + 'slocum', + 'stratocumulus', + 'succumb', + 'talcum', + 'taraxacum', + 'tecumseh', + 'tucuman', + 'capsicum', + 'cecum', + 'circum', + 'colchicum', + 'document', + 'ecumeni', + 'illyricum', + 'incumben', + ], + cunt: ['scunthorpe'], + cvnt: [], + cunny: [], + cunnie: [], + csam: [], + cyka: [], + darkie: [], + dick: [ + 'chappaquiddick', + 'dickens', + 'dickensian', + 'dickerson', + 'dickey', + 'dickies', + 'dickinson', + 'dickson', + 'dickvandyke', + 'dicky', + ], + dildo: [], + douchebag: [], + dyke: ['vandyke'], + downie: [], + dumbass: [], + ejaculate: [], + fag: ['antofagasta', 'serfage', 'wharfage', 'fagin', 'leafage'], + feck: [], + fellate: [], + fellatio: [], + felch: [], + fuck: [], + fvck: [], + fxck: [], + fack: [], + fzck: [], + fck: [], + fudgepacker: [], + flange: ['flanged', 'flanges'], + gestapo: [], + gook: [], + horny: ['thorny'], + hooker: [], + hitler: [], + incest: [], + jap: ['japan'], + jizz: [], + jigabo: [], + junglebunny: [], + kkk: [], + kike: [], + klux: [], + kluklux: [], + klukluxklan: [], + koon: [], + lickmy: [], + masturbat: [], + molest: [], + muff: [ + 'muffed', + 'muffin', + 'muffins', + 'muffle', + 'muffled', + 'muffler', + 'mufflers', + 'muffles', + 'muffling', + 'muffs', + 'ragamuffin', + 'earmuff', + 'earmuffs', + ], + nazi: ['ashkenazi', 'ashkenazic', 'ashkenazim', 'monazite'], + nigg: [], + niqa: [], + nigga: [], + niqqa: [], + niggu: [], + niqqu: [], + niggr: [], + nigger: [], + niglet: [], + nignog: [], + paki: ['pakistan'], + penis: ['openis', 'penistone'], + porn: [], + prostitut: [], + pube: [], + pussie: [], + pussy: ['pussycat', 'pussyfoot'], + raghead: [], + rape: [ + 'grape', + 'forape', + 'trapeze', + 'trapezium', + 'trapezius', + 'trapezoid', + 'therapeutic', + 'drape', + 'parapet', + 'rapeseed', + 'scrape', + 'serape', + ], + rapist: ['therapist'], + retard: ['retardant', 'retarder', 'retarding'], + rimjob: [], + shit: [ + 'cushitic', + 'shitake', + 'pushit', + 'peshitta', + 'libshitz', + 'shitzu', + 'wishit', + 'yamashita', + 'finishit', + 'shitbox', + 'shitmg', + 'publishit', + 'englishit', + ], + slut: [], + spunk: ['spunky'], + suckmy: [], + sodom: [], + semen: ['sement'], + teensex: [], + tittie: [], + titty: [], + trannie: [], + tranny: [], + vagina: [], + wank: ['swank', 'wankel'], + wetback: [], + whore: ['whores', 'whorev', 'whoreturned'], + whitepower: [], + fondle: [], + minestorm: [], + kissmy: [], + blowmy: [], + jelqing: [], + dafuq: [], +} + +const SLUR_TERMS = new Set([ + 'beaner', + 'cameljockey', + 'chankoro', + 'chink', + 'chingchong', + 'coon', + 'cottonpic', + 'cottonpik', + 'darkie', + 'downie', + 'dyke', + 'fag', + 'gook', + 'jap', + 'jigabo', + 'junglebunny', + 'kike', + 'koon', + 'nigg', + 'niqa', + 'nigga', + 'niqqa', + 'niggu', + 'niqqu', + 'niggr', + 'nigger', + 'niglet', + 'nignog', + 'paki', + 'raghead', + 'retard', + 'trannie', + 'tranny', + 'wetback', +]) + +export const DEFAULT_PROFANITY_PATTERNS: Readonly> = + Object.fromEntries( + Object.entries(TERM_EXCEPTIONS).map(([term, exceptions]) => { + const kind: ProfanityKind = SLUR_TERMS.has(term) ? 'slur' : 'profanity' + return [term, { kind, exceptions }] as const + }), + ) + +export const DEFAULT_PROFANITY_CONFIG: ProfanityConfig = { + patterns: DEFAULT_PROFANITY_PATTERNS, +} + +function newTrieNode(): TrieNode { + return { + children: new Map(), + negatives: [], + } +} + +export function sanitizeProfanityText(text: string): string { + const transformed = text + .normalize('NFD') + .replaceAll(/\p{Mn}/gu, '') + .normalize('NFC') + const characters = Array.from(transformed) + let sanitized = '' + + for (let index = 0; index < characters.length; index++) { + const character = characters[index] + const multiReplacement = MULTI_CHARACTER_REPLACEMENTS[character] + const nextCharacter = characters[index + 1] + + if (multiReplacement && nextCharacter && multiReplacement[nextCharacter]) { + sanitized += multiReplacement[nextCharacter] + index++ + continue + } + + const replacement = CHARACTER_REPLACEMENTS[character] + if (replacement) { + sanitized += replacement + } else if (character >= 'A' && character <= 'Z') { + sanitized += character.toLowerCase() + } else if ((character >= 'a' && character <= 'z') || (character >= '0' && character <= '9')) { + sanitized += character + } + } + + return sanitized +} + +function createTrie(patterns: Readonly>): TrieNode { + const root = newTrieNode() + + for (const [rawTerm, pattern] of Object.entries(patterns)) { + const term = rawTerm.toLowerCase() + if (!term || sanitizeProfanityText(term) !== term) { + throw new Error(`Profanity term must already be sanitized: ${rawTerm}`) + } + + let current = root + for (const character of term) { + let child = current.children.get(character) + if (!child) { + child = newTrieNode() + current.children.set(character, child) + } + current = child + } + + if (current.terminal) { + throw new Error(`Duplicate sanitized profanity term: ${term}`) + } + current.terminal = { kind: pattern.kind, term } + + for (const rawException of pattern.exceptions) { + const exception = rawException.toLowerCase() + if (sanitizeProfanityText(exception) !== exception) { + throw new Error(`Profanity exception must already be sanitized: ${rawException}`) + } + + const termIndex = exception.indexOf(term) + if (termIndex < 0) { + throw new Error(`Profanity exception must contain ${term}: ${rawException}`) + } + + current.negatives.push({ + prefix: exception.slice(0, termIndex), + suffix: exception.slice(termIndex + term.length), + }) + } + } + + return root +} + +function negativeMatches(negative: NegativeMatch, text: string, start: number, end: number) { + const prefixIndex = start - negative.prefix.length + const suffixIndex = end + negative.suffix.length + + if (prefixIndex < 0 || suffixIndex > text.length) return false + return ( + text.slice(prefixIndex, start) === negative.prefix && + text.slice(end, suffixIndex) === negative.suffix + ) +} + +function findAt(root: TrieNode, text: string, start: number): ProfanityMatch | undefined { + let current = root + + for (let index = start; index < text.length; index++) { + const child = current.children.get(text[index]) + if (!child) return undefined + current = child + + if (!current.terminal) continue + + const end = index + 1 + const matchesNegative = current.negatives.some((negative) => + negativeMatches(negative, text, start, end), + ) + if (matchesNegative && current.children.size === 0) return undefined + + return { + kind: current.terminal.kind, + term: text.slice(start, end), + sanitizedStart: start, + sanitizedEnd: end, + } + } + + return undefined +} + +export function createProfanityValidator( + config: ProfanityConfig = DEFAULT_PROFANITY_CONFIG, +): ProfanityValidator { + const root = createTrie(config.patterns) + + function findFirst(text: string): ProfanityMatch | undefined { + const sanitized = sanitizeProfanityText(text) + for (let index = 0; index < sanitized.length; index++) { + const match = findAt(root, sanitized, index) + if (match) return match + } + return undefined + } + + function findAll(text: string): ProfanityMatch[] { + const sanitized = sanitizeProfanityText(text) + const matches: ProfanityMatch[] = [] + + for (let index = 0; index < sanitized.length; ) { + const match = findAt(root, sanitized, index) + if (match) { + matches.push(match) + index = match.sanitizedEnd + } else { + index++ + } + } + + return matches + } + + function validate(text: string): ProfanityResult { + const matches = findAll(text) + const profanityCount = matches.filter((match) => match.kind === 'profanity').length + const slurCount = matches.length - profanityCount + + return { + valid: matches.length === 0, + profanityCount, + slurCount, + firstMatch: matches[0], + matches, + } + } + + return { + sanitize: sanitizeProfanityText, + findFirst, + findAll, + validate, + } +} + +export const profanityValidator = createProfanityValidator() + +export function validateProfanity(text: string): ProfanityResult { + return profanityValidator.validate(text) +} diff --git a/packages/moderation/src/validators/profanity/tests.ts b/packages/moderation/src/validators/profanity/tests.ts new file mode 100644 index 0000000000..fdfe6101e6 --- /dev/null +++ b/packages/moderation/src/validators/profanity/tests.ts @@ -0,0 +1,123 @@ +import assert from 'node:assert/strict' +import test from 'node:test' + +import { createProfanityValidator, sanitizeProfanityText, validateProfanity } from './index.ts' + +test('sanitizes text with the configured single-character replacements', () => { + assert.equal(sanitizeProfanityText('4@3105789+$([{!|£€¥¢<'), 'aaeiostbgtsccciieeycc') +}) + +test('sanitizes paired characters before their single-character replacements', () => { + assert.equal(sanitizeProfanityText('()[]{}<>'), 'oooo') +}) + +test('strips accents, separators, emoji, and ASCII casing', () => { + assert.equal(sanitizeProfanityText('F Ü.C—K🙂'), 'fuck') +}) + +test('matches profanity across separators and common substitutions', () => { + assert.equal(validateProfanity('f.u c-k').firstMatch?.term, 'fuck') + assert.equal(validateProfanity('$h!t').firstMatch?.term, 'shit') + assert.equal(validateProfanity('p()rn').firstMatch?.term, 'porn') +}) + +test('honors exact negative prefix and suffix matches', () => { + const validator = createProfanityValidator({ + patterns: { + bad: { kind: 'profanity', exceptions: ['notbadword'] }, + }, + }) + + assert.equal(validator.findFirst('not bad word'), undefined) + assert.equal(validator.findFirst('very bad word')?.term, 'bad') + assert.equal(validator.findFirst('not bad phrase')?.term, 'bad') +}) + +test('matches the first profanity while ignoring a later negative match', () => { + const validator = createProfanityValidator({ + patterns: { + shit: { kind: 'profanity', exceptions: ['horseshit', 'bullshit'] }, + fuck: { kind: 'profanity', exceptions: [] }, + }, + }) + + assert.equal(validator.findFirst('this horseshit'), undefined) + assert.equal(validator.findFirst('fuck this bullshit')?.term, 'fuck') +}) + +test('uses the first terminal when one configured term prefixes another', () => { + const validator = createProfanityValidator({ + patterns: { + bad: { kind: 'profanity', exceptions: [] }, + badword: { kind: 'profanity', exceptions: [] }, + }, + }) + + assert.equal(validator.findFirst('badword')?.term, 'bad') +}) + +test('does not match configured false-positive substrings', () => { + assert.equal(validateProfanity('Scunthorpe and peacock').valid, true) + assert.equal(validateProfanity('cock and cunt').profanityCount, 2) +}) + +test('rejects any uncensored configured profanity', () => { + assert.equal(validateProfanity('A clean project').valid, true) + assert.equal(validateProfanity('This is shit').valid, false) +}) + +test('allows redacted profanity when the removed letters cannot reconstruct a term', () => { + assert.equal(validateProfanity('f**k').valid, true) + assert.equal(validateProfanity('f**k works in titles, summaries, and descriptions').valid, true) + assert.equal(validateProfanity('f.u.c.k').valid, false) +}) + +test('rejects slurs', () => { + const validator = createProfanityValidator({ + patterns: { + forbidden: { kind: 'slur', exceptions: [] }, + }, + }) + + assert.equal(validator.validate('forbidden').valid, false) + assert.equal(validator.validate('forbidden').slurCount, 1) +}) + +test('counts matches from left to right without overlaps', () => { + const validator = createProfanityValidator({ + patterns: { + bad: { kind: 'profanity', exceptions: [] }, + }, + }) + + assert.deepEqual( + validator + .findAll('bad-bad') + .map(({ sanitizedStart, sanitizedEnd }) => [sanitizedStart, sanitizedEnd]), + [ + [0, 3], + [3, 6], + ], + ) +}) + +test('rejects invalid configuration', () => { + assert.throws( + () => + createProfanityValidator({ + patterns: { + 'not sanitized': { kind: 'profanity', exceptions: [] }, + }, + }), + /term must already be sanitized/, + ) + assert.throws( + () => + createProfanityValidator({ + patterns: { + bad: { kind: 'profanity', exceptions: ['innocent'] }, + }, + }), + /exception must contain bad/, + ) +}) From bc37c6dc9ac9e18ff66d3e08487e18f30b9b6890 Mon Sep 17 00:00:00 2001 From: chyz <32403637+chyzman@users.noreply.github.com> Date: Mon, 24 Aug 2026 17:23:12 -0400 Subject: [PATCH 02/40] Better External Link Validation (#7005) * complain about links we know are incorrect * dont complain about the specific part of the url being incorrect if it wouldn't be correct regardless * UnsavedChangesPopup had a place where it was supposed to disable the save button but the actual impl didn't implement that property so i fixed that and also more work on link checks (including start of license link checks too) * github sponsor note * show the warning directly cuz it's not particularly clear there's a tooltip migrate links page to actually good unsaved changes thing that every other page uses also like self-hosted git support also add 9minecraft to blacklist * this dont look correct warning + oops duplicate id * better/more blacklist * clean up after merge * imports --------- Co-authored-by: tdgao --- .../src/components/LinkCheckMessage.vue | 29 + .../[type]/[project]/settings/license.vue | 18 +- .../pages/[type]/[project]/settings/links.vue | 512 ++++++------ packages/moderation/src/data/nags/index.ts | 1 + .../moderation/src/data/nags/link-checks.ts | 749 ++++++++++++++++++ 5 files changed, 1037 insertions(+), 272 deletions(-) create mode 100644 apps/frontend/src/components/LinkCheckMessage.vue create mode 100644 packages/moderation/src/data/nags/link-checks.ts diff --git a/apps/frontend/src/components/LinkCheckMessage.vue b/apps/frontend/src/components/LinkCheckMessage.vue new file mode 100644 index 0000000000..d273bef595 --- /dev/null +++ b/apps/frontend/src/components/LinkCheckMessage.vue @@ -0,0 +1,29 @@ + + + diff --git a/apps/frontend/src/pages/[type]/[project]/settings/license.vue b/apps/frontend/src/pages/[type]/[project]/settings/license.vue index fea81bc4ff..9d6079b9b2 100644 --- a/apps/frontend/src/pages/[type]/[project]/settings/license.vue +++ b/apps/frontend/src/pages/[type]/[project]/settings/license.vue @@ -72,7 +72,7 @@ -
+
+
@@ -145,7 +146,8 @@ !( current.license.friendly === 'Custom' && (current.license.short === '' || current.licenseUrl === '') - ) + ) && + effectiveLicenseCheck?.severity !== 'error' " @reset="reset" @save="save" @@ -154,6 +156,7 @@ diff --git a/apps/frontend/src/components/ui/create/ProjectCreateModal.vue b/apps/frontend/src/components/ui/create/ProjectCreateModal.vue index 492245bbf6..fc8c70d6aa 100644 --- a/apps/frontend/src/components/ui/create/ProjectCreateModal.vue +++ b/apps/frontend/src/components/ui/create/ProjectCreateModal.vue @@ -43,10 +43,14 @@ /> -