From f2bcdd21de6c61495644bec8466401f294405f66 Mon Sep 17 00:00:00 2001 From: Carlton McFarlane Date: Mon, 26 Oct 2020 12:14:04 +0100 Subject: [PATCH] =?UTF-8?q?fix(banned=20words):=20fix=20partial=20matching?= =?UTF-8?q?=20of=20words=20containing=20diacritic=E2=80=A6=20(#12444)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix(banned words): fix partial matching of words containing diacritics against banned words list (#12309) * lint: remove whitespace to fix error * test: add test to prevent partial matching of words containing diacritics against banned words list (#12309) * doc: add link to Unicode table of diacritical marks (#12309) --- test/api/unit/libs/stringUtils.test.js | 5 +++++ website/server/libs/stringUtils.js | 11 ++++++++++- 2 files changed, 15 insertions(+), 1 deletion(-) diff --git a/test/api/unit/libs/stringUtils.test.js b/test/api/unit/libs/stringUtils.test.js index 0596cd16b9..947274df53 100644 --- a/test/api/unit/libs/stringUtils.test.js +++ b/test/api/unit/libs/stringUtils.test.js @@ -8,5 +8,10 @@ describe('stringUtils', () => { const matches = getMatchesByWordArray(message, bannedWords); expect(matches.length).to.equal(bannedWords.length); }); + it('doesn\'t flag names with accented characters', () => { + const name = 'TESTPLACEHOLDERSWEARWORDHEREé'; + const matches = getMatchesByWordArray(name, bannedWords); + expect(matches.length).to.equal(0); + }); }); }); diff --git a/website/server/libs/stringUtils.js b/website/server/libs/stringUtils.js index a7e4b08aa8..39f75356c5 100644 --- a/website/server/libs/stringUtils.js +++ b/website/server/libs/stringUtils.js @@ -4,12 +4,21 @@ export function removePunctuationFromString (str) { // NOTE: the wordsToMatch aren't escaped in order to support regular expressions, // so this method should not be used if wordsToMatch contains unsanitized user input + export function getMatchesByWordArray (str, wordsToMatch) { + // remove accented characters from the string, which would trip up the regEx + // later on, by using the built-in Unicode normalisation methods + // https://developer.mozilla.org/en-US/docs/Web/JavaScript/Reference/Global_Objects/String/normalize + // https://www.unicode.org/reports/tr15/#Canon_Compat_Equivalence + // https://unicode-table.com/en/#combining-diacritical-marks + + const normalizedStr = str.normalize('NFD').replace(/[\u0300-\u036f]/g, ''); + const matchedWords = []; const wordRegexs = wordsToMatch.map(word => new RegExp(`\\b([^a-z]+)?${word}([^a-z]+)?\\b`, 'i')); for (let i = 0; i < wordRegexs.length; i += 1) { const regEx = wordRegexs[i]; - const match = str.match(regEx); + const match = normalizedStr.match(regEx); if (match !== null && match[0] !== null) { const trimmedMatch = removePunctuationFromString(match[0]).trim(); matchedWords.push(trimmedMatch);