JSFiddle - React, Tailwind, and code Playground
by raisch
HTML
<div id="result">[result]</div>
JavaScript
// tokenize(str)
// extracts semantically useful tokens from a string containing English-language sentences
// @param {String} the string to tokenize
// @returns {Array} containing extracted tokens
function tokenize(str) {
var punct = '\\[' + '\\!' + '\\"' + '\\#' + '\\$' + // since javascript does not
'\\%' + '\\&' + '\\\'' + '\\(' + '\\)' + // support POSIX character
'\\*' + '\\+' + '\\,' + '\\\\' + '\\-' + // classes, we'll need our
'\\.' + '\\/' + '\\:' + '\\;' + '\\<' + // own version of [:punct:]
'\\=' + '\\>' + '\\?' + '\\@' + '\\[' +
'\\]' + '\\^' + '\\_' + '\\`' + '\\{' +
'\\|' + '\\}' + '\\~' + '\\]',
re = new RegExp( // tokenizer
'\\s*' + // discard possible leading whitespace
'(' + // start capture group #1
'\\.{3}' + // ellipsis (must appear before punct)
'|' + // alternator
'\\w+\\-\\w+' + // hyphenated words (must appear before punct)
'|' + // alternator
'\\w+\'(?:\\w+)?' + // compound words (must appear before punct)
'|' + // alternator
'\\w+' + // other words
'|' + // alternator
'[' + punct + ']' + // punct
')' // end capture group #1
),
tokens=str.split(re), // split string using tokenizing regex
result=[];
// add non-empty tokens to result
for(var i=0,len=tokens.length;i++<len;) {
if(tokens[i]) {
result.push(tokens[i]);
}
}
return result;
} // end tokenize()
// example string to tokenize
var str = 'Here\'s a (good, bad, indifferent, ...) ' +
'example sentence to be used in this test ' +
'of English language "token-extraction".',
...