JSFiddle - React, Tailwind, and code Playground

by raisch

HTML

<div id="result">[result]</div>

JavaScript

// tokenize(str)
// extracts semantically useful tokens from a string containing English-language sentences
// @param {String}    the string to tokenize
// @returns {Array}   containing extracted tokens

function tokenize(str) {

    var punct = '\\[' + '\\!' + '\\"' + '\\#' + '\\$'  + // since javascript does not
                '\\%' + '\\&' + '\\\'' + '\\(' + '\\)' + // support POSIX character
                '\\*' + '\\+' + '\\,' + '\\\\' + '\\-' + // classes, we'll need our
                '\\.' + '\\/' + '\\:' + '\\;' + '\\<'  + // own version of [:punct:]
                '\\=' + '\\>' + '\\?' + '\\@' + '\\['  + 
                '\\]' + '\\^' + '\\_' + '\\`' + '\\{'  + 
                '\\|' + '\\}' + '\\~' + '\\]',
        
        re = new RegExp( // tokenizer
            '\\s*' +           // discard possible leading whitespace
            '(' +              // start capture group #1
                '\\.{3}' +          // ellipsis (must appear before punct)
            '|' +              // alternator
                '\\w+\\-\\w+' +     // hyphenated words (must appear before punct)
            '|' +              // alternator
                '\\w+\'(?:\\w+)?' + // compound words (must appear before punct)
            '|' +              // alternator
                '\\w+' +            // other words
            '|' +              // alternator
                '[' + punct + ']' +  // punct
            ')' // end capture group #1
          ),
        
        tokens=str.split(re),  // split string using tokenizing regex
        result=[];
    
    // add non-empty tokens to result
    for(var i=0,len=tokens.length;i++<len;) {
        if(tokens[i]) {
            result.push(tokens[i]);
        }
    }
    
    return result;

} // end tokenize()

// example string to tokenize
var str = 'Here\'s a (good, bad, indifferent, ...) ' + 
          'example sentence to be used in this test ' +  
          'of English language "token-extraction".',
    
   ...