diff --git a/init.js b/init.js index 768d02999..ec675d5a4 100644 --- a/init.js +++ b/init.js @@ -1,15 +1,7 @@ const Setting = require('./models/setting'); -const wordlist = require('./services/wordlist'); module.exports = () => Promise.all([ // Upsert the settings object. - Setting - .init({id: '1', moderation: 'pre'}) - .then(() => { - - // Load in the wordlist now that settings have been init'd. - return wordlist.init(); - }) - + Setting.init({id: '1', moderation: 'pre'}) ]); diff --git a/services/wordlist.js b/services/wordlist.js index 2103c7267..49b646bcb 100644 --- a/services/wordlist.js +++ b/services/wordlist.js @@ -8,184 +8,194 @@ const Setting = require('../models/setting'); * The root wordlist object. * @type {Object} */ -const wordlist = { - lists: { - banned: [], - suspect: [] - }, - enabled: false -}; +class Wordlist { -/** - * Loads wordlists in from the naughty-words package based on languages - * selected. - * @param {Array} languages language codes to add to the wordlist - */ -wordlist.init = () => { - return Setting - .retrieve() - .then((settings) => { + constructor() { + this.lists = { + banned: [], + suspect: [] + }; + } - // Insert the settings wordlist. - wordlist.insert(settings.wordlist); + /** + * Loads wordlists in from the database + */ + load() { + return Setting + .retrieve() + .then((settings) => { + + // Insert the settings wordlist. + this.upsert(settings.wordlist); + }); + } + + /** + * Inserts the wordlist data + * @param {Array} list list of words to be set to the wordlist + */ + upsert(lists) { + + // Add the words to this array, but also lowercase the words so that an + // easy comparison can take place. + ['banned', 'suspect'].forEach((k) => { + if (!(k in lists)) { + return; + } + + this.lists[k] = Wordlist.parseList(lists[k]); + + debug(`Added ${lists[k].length} words to the ${k} wordlist.`); }); -}; -/** - * Inserts the wordlist data and enables the wordlist. - * @param {Array} list list of words to be added to the wordlist - */ -wordlist.insert = (lists) => { + return Promise.resolve(this); + } - // Add the words to this array, but also lowercase the words so that an - // easy comparison can take place. - ['banned', 'suspect'].forEach((k) => { - if (!(k in lists)) { - return; - } + /** + * Parses the list content. + * @param {Array} list array of words to parse for a list. + * @return {Array} the parsed list + */ + static parseList(list) { + return _.uniq(list.map((word) => tokenizer.tokenize(word.toLowerCase()))); + } - wordlist.lists[k] = wordlist.parseList(lists[k]); + /** + * Tests the phrase to see if it contains any of the defined blockwords. + * @param {String} phrase value to check for blockwords. + * @return {Boolean} true if a blockword is found, false otherwise. + */ + match(list, phrase) { - debug(`Added ${lists[k].length} words to the ${k} wordlist.`); - }); + // Lowercase the word to ensure that we don't miss a match due to + // capitalization. + let lowerPhraseWords = tokenizer.tokenize(phrase.toLowerCase()); - // Enable the wordlist. - wordlist.enabled = true; + // This will return true in the event that at least one blockword is found + // in the phrase. + return list.some((blockphrase) => { - return Promise.resolve(wordlist); -}; + // First, let's see if we can find the first word in the blockphrase in the + // source phrase. + let idx = lowerPhraseWords.indexOf(blockphrase[0]); -/** - * Parses the list content. - * @param {Array} list array of words to parse for a list. - * @return {Array} the parsed list - */ -wordlist.parseList = (list) => _.uniq(list.map((word) => { - return tokenizer.tokenize(word.toLowerCase()); -})); + if (idx === -1) { -/** - * Tests the phrase to see if it contains any of the defined blockwords. - * @param {String} phrase value to check for blockwords. - * @return {Boolean} true if a blockword is found, false otherwise. - */ -wordlist.match = (list, phrase) => { - - // Lowercase the word to ensure that we don't miss a match due to - // capitalization. - let lowerPhraseWords = tokenizer.tokenize(phrase.toLowerCase()); - - // This will return true in the event that at least one blockword is found - // in the phrase. - return list.some((blockphrase) => { - - // First, let's see if we can find the first word in the blockphrase in the - // source phrase. - let idx = lowerPhraseWords.indexOf(blockphrase[0]); - - if (idx === -1) { - - // The first blockword in the blockphrase did not match the source phrase - // anywhere. - return false; - } - - // Here we'll quick respond with true in the event that the blockphrase was - // just a single word. - if (blockphrase.length === 1) { - return true; - } - - // We found the first word in the source phrase! Lets ensure it matches the - // rest of the blockphrase... - - // Check to see if it even has the length to support this word! - if (lowerPhraseWords.length < idx + blockphrase.length - 1) { - - // We couldn't possibly have the entire phrase here because we don't have - // enough entries! - return false; - } - - for (let i = 1; i < blockphrase.length; i++) { - - // Check to see if the next word also matches! - if (lowerPhraseWords[idx + i] !== blockphrase[i]) { + // The first blockword in the blockphrase did not match the source phrase + // anywhere. return false; } - } - // We've walked over all the words of the blockphrase, and haven't had a - // mismatch... It does contain the whole word! - return true; - }); -}; + // Here we'll quick respond with true in the event that the blockphrase was + // just a single word. + if (blockphrase.length === 1) { + return true; + } + + // We found the first word in the source phrase! Lets ensure it matches the + // rest of the blockphrase... + + // Check to see if it even has the length to support this word! + if (lowerPhraseWords.length < idx + blockphrase.length - 1) { + + // We couldn't possibly have the entire phrase here because we don't have + // enough entries! + return false; + } + + for (let i = 1; i < blockphrase.length; i++) { + + // Check to see if the next word also matches! + if (lowerPhraseWords[idx + i] !== blockphrase[i]) { + return false; + } + } + + // We've walked over all the words of the blockphrase, and haven't had a + // mismatch... It does contain the whole word! + return true; + }); + } + + filter(...fields) { + return (req, res, next) => { + + // Start with the sensible default that the content does not contain + // profanity. + req.wordlist = { + matched: false + }; + + // Loop over all the fields from the body that we want to check. + for (let i = 0; i < fields.length; i++) { + let field = fields[i]; + + let phrase = _.get(req.body, field, false); + + // If the field doesn't exist in the body, then it can't be profane! + if (!phrase) { + + // Return that there wasn't a profane word here. + continue; + } + + // Check if the field contains a banned word. + if (this.match(this.lists.banned, phrase)) { + debug(`the field "${field}" contained a phrase "${phrase}" which contained a banned word/phrase`); + + req.wordlist.banned = ErrContainsProfanity; + + // Stop looping through the fields now, we discovered the worst possible + // situation (a banned word). + break; + } + + // Check if the field contains a banned word. + if (this.match(this.lists.suspect, phrase)) { + debug(`the field "${field}" contained a phrase "${phrase}" which contained a suspected word/phrase`); + + req.wordlist.suspect = ErrContainsProfanity; + + // Continue looping through the fields now, we discovered a possible bad + // word (suspect). + continue; + } + } + + next(); + }; + } + + /** + * Connect middleware for scanning request bodies for wordlisted words and + * attaching a ErrContainsProfanity to the req.wordlisted parameter, otherwise + * it will just set that parameter to false. + * @param {Array} fields selectors for the body to extract the fields to be + * tested + * @return {Function} the Connect middleware + */ + static filter(...fields) { + return (req, res, next) => { + + // Create a new instance of the Wordlist. + const wl = new Wordlist(); + + wl + .load() + .then(() => { + + // Perform a filtering operation using the new instance of the + // Wordlist. + wl.filter(...fields)(req, res, next); + }); + }; + } +} // ErrContainsProfanity is returned in the event that the middleware detects // profanity/wordlisted words in the payload. const ErrContainsProfanity = new Error('contains profanity'); ErrContainsProfanity.status = 400; -/** - * Connect middleware for scanning request bodies for wordlisted words and - * attaching a ErrContainsProfanity to the req.wordlisted parameter, otherwise - * it will just set that parameter to false. - * @param {Array} fields selectors for the body to extract the fields to be - * tested - * @return {Function} the Connect middleware - */ -wordlist.filter = (...fields) => (req, res, next) => { - - // Start with the sensible default that the content does not contain - // profanity. - req.wordlist = { - matched: false - }; - - // If the wordlist isn't enabled, then don't actually perform checking and - // forward the request! - if (!wordlist.enabled) { - return next(); - } - - // Loop over all the fields from the body that we want to check. - for (let i = 0; i < fields.length; i++) { - let field = fields[i]; - - let phrase = _.get(req.body, field, false); - - // If the field doesn't exist in the body, then it can't be profane! - if (!phrase) { - - // Return that there wasn't a profane word here. - continue; - } - - // Check if the field contains a banned word. - if (wordlist.match(wordlist.lists.banned, phrase)) { - debug(`the field "${field}" contained a phrase "${phrase}" which contained a banned word/phrase`); - - req.wordlist.banned = ErrContainsProfanity; - - // Stop looping through the fields now, we discovered the worst possible - // situation (a banned word). - break; - } - - // Check if the field contains a banned word. - if (wordlist.match(wordlist.lists.suspect, phrase)) { - debug(`the field "${field}" contained a phrase "${phrase}" which contained a suspected word/phrase`); - - req.wordlist.suspect = ErrContainsProfanity; - - // Continue looping through the fields now, we discovered a possible bad - // word (suspect). - continue; - } - } - - next(); -}; - -module.exports = wordlist; +module.exports = Wordlist; module.exports.ErrContainsProfanity = ErrContainsProfanity; diff --git a/tests/routes/api/comments/index.js b/tests/routes/api/comments/index.js index ea439299e..b3c3a44fb 100644 --- a/tests/routes/api/comments/index.js +++ b/tests/routes/api/comments/index.js @@ -8,22 +8,18 @@ const expect = chai.expect; chai.should(); chai.use(require('chai-http')); -const wordlist = require('../../../../services/wordlist'); const Comment = require('../../../../models/comment'); const Asset = require('../../../../models/asset'); const Action = require('../../../../models/action'); const User = require('../../../../models/user'); const Setting = require('../../../../models/setting'); -const settings = {id: '1', moderation: 'pre'}; +const settings = {id: '1', moderation: 'pre', wordlist: {banned: ['bad words'], suspect: ['suspect words']}}; describe('/api/v1/comments', () => { // Ensure that the settings are always available. - beforeEach(() => Promise.all([ - wordlist.insert({banned: ['bad words'], suspect: ['suspect words']}), - Setting.init(settings) - ])); + beforeEach(() => Setting.init(settings)); describe('#get', () => { const comments = [{ diff --git a/tests/services/wordlist.js b/tests/services/wordlist.js index c65be3b11..dd3e5257e 100644 --- a/tests/services/wordlist.js +++ b/tests/services/wordlist.js @@ -1,6 +1,6 @@ const expect = require('chai').expect; -const wordlist = require('../../services/wordlist'); +const Wordlist = require('../../services/wordlist'); describe('wordlist: services', () => { @@ -16,21 +16,22 @@ describe('wordlist: services', () => { ] }; + let wordlist = new Wordlist(); + describe('#init', () => { - before(() => wordlist.insert(wordlists)); + before(() => wordlist.upsert(wordlists)); it('has entries', () => { expect(wordlist.lists.banned).to.not.be.empty; expect(wordlist.lists.suspect).to.not.be.empty; - expect(wordlist.enabled).to.be.true; }); }); describe('#match', () => { - const bannedList = wordlist.parseList(wordlists.banned); + const bannedList = Wordlist.parseList(wordlists.banned); it('does match on a bad word', () => { [ @@ -61,7 +62,7 @@ describe('wordlist: services', () => { describe('#filter', () => { - before(() => wordlist.insert(wordlists)); + before(() => wordlist.upsert(wordlists)); it('matches on bodies containing bad words', (done) => { @@ -75,7 +76,7 @@ describe('wordlist: services', () => { expect(err).to.be.undefined; expect(req).to.have.property('wordlist'); expect(req.wordlist).to.have.property('matched'); - expect(req.wordlist.banned).to.be.equal(wordlist.ErrContainsProfanity); + expect(req.wordlist.banned).to.be.equal(Wordlist.ErrContainsProfanity); done(); });