| [81bc7da] | 1 | 'use strict';
|
|---|
| 2 |
|
|---|
| 3 | /**
|
|---|
| 4 | * Converts tokens for a single address into an address object
|
|---|
| 5 | *
|
|---|
| 6 | * @param {Array} tokens Tokens object
|
|---|
| 7 | * @return {Object} Address object
|
|---|
| 8 | */
|
|---|
| 9 | function _handleAddress(tokens) {
|
|---|
| 10 | let isGroup = false;
|
|---|
| 11 | let state = 'text';
|
|---|
| 12 | let address;
|
|---|
| 13 | let addresses = [];
|
|---|
| 14 | let data = {
|
|---|
| 15 | address: [],
|
|---|
| 16 | comment: [],
|
|---|
| 17 | group: [],
|
|---|
| 18 | text: []
|
|---|
| 19 | };
|
|---|
| 20 | let i;
|
|---|
| 21 | let len;
|
|---|
| 22 |
|
|---|
| 23 | // Filter out <addresses>, (comments) and regular text
|
|---|
| 24 | for (i = 0, len = tokens.length; i < len; i++) {
|
|---|
| 25 | let token = tokens[i];
|
|---|
| 26 | let prevToken = i ? tokens[i - 1] : null;
|
|---|
| 27 | if (token.type === 'operator') {
|
|---|
| 28 | switch (token.value) {
|
|---|
| 29 | case '<':
|
|---|
| 30 | state = 'address';
|
|---|
| 31 | break;
|
|---|
| 32 | case '(':
|
|---|
| 33 | state = 'comment';
|
|---|
| 34 | break;
|
|---|
| 35 | case ':':
|
|---|
| 36 | state = 'group';
|
|---|
| 37 | isGroup = true;
|
|---|
| 38 | break;
|
|---|
| 39 | default:
|
|---|
| 40 | state = 'text';
|
|---|
| 41 | break;
|
|---|
| 42 | }
|
|---|
| 43 | } else if (token.value) {
|
|---|
| 44 | if (state === 'address') {
|
|---|
| 45 | // handle use case where unquoted name includes a "<"
|
|---|
| 46 | // Apple Mail truncates everything between an unexpected < and an address
|
|---|
| 47 | // and so will we
|
|---|
| 48 | token.value = token.value.replace(/^[^<]*<\s*/, '');
|
|---|
| 49 | }
|
|---|
| 50 |
|
|---|
| 51 | if (prevToken && prevToken.noBreak && data[state].length) {
|
|---|
| 52 | // join values
|
|---|
| 53 | data[state][data[state].length - 1] += token.value;
|
|---|
| 54 | } else {
|
|---|
| 55 | data[state].push(token.value);
|
|---|
| 56 | }
|
|---|
| 57 | }
|
|---|
| 58 | }
|
|---|
| 59 |
|
|---|
| 60 | // If there is no text but a comment, replace the two
|
|---|
| 61 | if (!data.text.length && data.comment.length) {
|
|---|
| 62 | data.text = data.comment;
|
|---|
| 63 | data.comment = [];
|
|---|
| 64 | }
|
|---|
| 65 |
|
|---|
| 66 | if (isGroup) {
|
|---|
| 67 | // http://tools.ietf.org/html/rfc2822#appendix-A.1.3
|
|---|
| 68 | data.text = data.text.join(' ');
|
|---|
| 69 | addresses.push({
|
|---|
| 70 | name: data.text || (address && address.name),
|
|---|
| 71 | group: data.group.length ? addressparser(data.group.join(',')) : []
|
|---|
| 72 | });
|
|---|
| 73 | } else {
|
|---|
| 74 | // If no address was found, try to detect one from regular text
|
|---|
| 75 | if (!data.address.length && data.text.length) {
|
|---|
| 76 | for (i = data.text.length - 1; i >= 0; i--) {
|
|---|
| 77 | if (data.text[i].match(/^[^@\s]+@[^@\s]+$/)) {
|
|---|
| 78 | data.address = data.text.splice(i, 1);
|
|---|
| 79 | break;
|
|---|
| 80 | }
|
|---|
| 81 | }
|
|---|
| 82 |
|
|---|
| 83 | let _regexHandler = function (address) {
|
|---|
| 84 | if (!data.address.length) {
|
|---|
| 85 | data.address = [address.trim()];
|
|---|
| 86 | return ' ';
|
|---|
| 87 | } else {
|
|---|
| 88 | return address;
|
|---|
| 89 | }
|
|---|
| 90 | };
|
|---|
| 91 |
|
|---|
| 92 | // still no address
|
|---|
| 93 | if (!data.address.length) {
|
|---|
| 94 | for (i = data.text.length - 1; i >= 0; i--) {
|
|---|
| 95 | // fixed the regex to parse email address correctly when email address has more than one @
|
|---|
| 96 | data.text[i] = data.text[i].replace(/\s*\b[^@\s]+@[^\s]+\b\s*/, _regexHandler).trim();
|
|---|
| 97 | if (data.address.length) {
|
|---|
| 98 | break;
|
|---|
| 99 | }
|
|---|
| 100 | }
|
|---|
| 101 | }
|
|---|
| 102 | }
|
|---|
| 103 |
|
|---|
| 104 | // If there's still is no text but a comment exixts, replace the two
|
|---|
| 105 | if (!data.text.length && data.comment.length) {
|
|---|
| 106 | data.text = data.comment;
|
|---|
| 107 | data.comment = [];
|
|---|
| 108 | }
|
|---|
| 109 |
|
|---|
| 110 | // Keep only the first address occurence, push others to regular text
|
|---|
| 111 | if (data.address.length > 1) {
|
|---|
| 112 | data.text = data.text.concat(data.address.splice(1));
|
|---|
| 113 | }
|
|---|
| 114 |
|
|---|
| 115 | // Join values with spaces
|
|---|
| 116 | data.text = data.text.join(' ');
|
|---|
| 117 | data.address = data.address.join(' ');
|
|---|
| 118 |
|
|---|
| 119 | if (!data.address && isGroup) {
|
|---|
| 120 | return [];
|
|---|
| 121 | } else {
|
|---|
| 122 | address = {
|
|---|
| 123 | address: data.address || data.text || '',
|
|---|
| 124 | name: data.text || data.address || ''
|
|---|
| 125 | };
|
|---|
| 126 |
|
|---|
| 127 | if (address.address === address.name) {
|
|---|
| 128 | if ((address.address || '').match(/@/)) {
|
|---|
| 129 | address.name = '';
|
|---|
| 130 | } else {
|
|---|
| 131 | address.address = '';
|
|---|
| 132 | }
|
|---|
| 133 | }
|
|---|
| 134 |
|
|---|
| 135 | addresses.push(address);
|
|---|
| 136 | }
|
|---|
| 137 | }
|
|---|
| 138 |
|
|---|
| 139 | return addresses;
|
|---|
| 140 | }
|
|---|
| 141 |
|
|---|
| 142 | /**
|
|---|
| 143 | * Creates a Tokenizer object for tokenizing address field strings
|
|---|
| 144 | *
|
|---|
| 145 | * @constructor
|
|---|
| 146 | * @param {String} str Address field string
|
|---|
| 147 | */
|
|---|
| 148 | class Tokenizer {
|
|---|
| 149 | constructor(str) {
|
|---|
| 150 | this.str = (str || '').toString();
|
|---|
| 151 | this.operatorCurrent = '';
|
|---|
| 152 | this.operatorExpecting = '';
|
|---|
| 153 | this.node = null;
|
|---|
| 154 | this.escaped = false;
|
|---|
| 155 |
|
|---|
| 156 | this.list = [];
|
|---|
| 157 | /**
|
|---|
| 158 | * Operator tokens and which tokens are expected to end the sequence
|
|---|
| 159 | */
|
|---|
| 160 | this.operators = {
|
|---|
| 161 | '"': '"',
|
|---|
| 162 | '(': ')',
|
|---|
| 163 | '<': '>',
|
|---|
| 164 | ',': '',
|
|---|
| 165 | ':': ';',
|
|---|
| 166 | // Semicolons are not a legal delimiter per the RFC2822 grammar other
|
|---|
| 167 | // than for terminating a group, but they are also not valid for any
|
|---|
| 168 | // other use in this context. Given that some mail clients have
|
|---|
| 169 | // historically allowed the semicolon as a delimiter equivalent to the
|
|---|
| 170 | // comma in their UI, it makes sense to treat them the same as a comma
|
|---|
| 171 | // when used outside of a group.
|
|---|
| 172 | ';': ''
|
|---|
| 173 | };
|
|---|
| 174 | }
|
|---|
| 175 |
|
|---|
| 176 | /**
|
|---|
| 177 | * Tokenizes the original input string
|
|---|
| 178 | *
|
|---|
| 179 | * @return {Array} An array of operator|text tokens
|
|---|
| 180 | */
|
|---|
| 181 | tokenize() {
|
|---|
| 182 | let list = [];
|
|---|
| 183 |
|
|---|
| 184 | for (let i = 0, len = this.str.length; i < len; i++) {
|
|---|
| 185 | let chr = this.str.charAt(i);
|
|---|
| 186 | let nextChr = i < len - 1 ? this.str.charAt(i + 1) : null;
|
|---|
| 187 | this.checkChar(chr, nextChr);
|
|---|
| 188 | }
|
|---|
| 189 |
|
|---|
| 190 | this.list.forEach(node => {
|
|---|
| 191 | node.value = (node.value || '').toString().trim();
|
|---|
| 192 | if (node.value) {
|
|---|
| 193 | list.push(node);
|
|---|
| 194 | }
|
|---|
| 195 | });
|
|---|
| 196 |
|
|---|
| 197 | return list;
|
|---|
| 198 | }
|
|---|
| 199 |
|
|---|
| 200 | /**
|
|---|
| 201 | * Checks if a character is an operator or text and acts accordingly
|
|---|
| 202 | *
|
|---|
| 203 | * @param {String} chr Character from the address field
|
|---|
| 204 | */
|
|---|
| 205 | checkChar(chr, nextChr) {
|
|---|
| 206 | if (this.escaped) {
|
|---|
| 207 | // ignore next condition blocks
|
|---|
| 208 | } else if (chr === this.operatorExpecting) {
|
|---|
| 209 | this.node = {
|
|---|
| 210 | type: 'operator',
|
|---|
| 211 | value: chr
|
|---|
| 212 | };
|
|---|
| 213 |
|
|---|
| 214 | if (nextChr && ![' ', '\t', '\r', '\n', ',', ';'].includes(nextChr)) {
|
|---|
| 215 | this.node.noBreak = true;
|
|---|
| 216 | }
|
|---|
| 217 |
|
|---|
| 218 | this.list.push(this.node);
|
|---|
| 219 | this.node = null;
|
|---|
| 220 | this.operatorExpecting = '';
|
|---|
| 221 | this.escaped = false;
|
|---|
| 222 |
|
|---|
| 223 | return;
|
|---|
| 224 | } else if (!this.operatorExpecting && chr in this.operators) {
|
|---|
| 225 | this.node = {
|
|---|
| 226 | type: 'operator',
|
|---|
| 227 | value: chr
|
|---|
| 228 | };
|
|---|
| 229 | this.list.push(this.node);
|
|---|
| 230 | this.node = null;
|
|---|
| 231 | this.operatorExpecting = this.operators[chr];
|
|---|
| 232 | this.escaped = false;
|
|---|
| 233 | return;
|
|---|
| 234 | } else if (['"', "'"].includes(this.operatorExpecting) && chr === '\\') {
|
|---|
| 235 | this.escaped = true;
|
|---|
| 236 | return;
|
|---|
| 237 | }
|
|---|
| 238 |
|
|---|
| 239 | if (!this.node) {
|
|---|
| 240 | this.node = {
|
|---|
| 241 | type: 'text',
|
|---|
| 242 | value: ''
|
|---|
| 243 | };
|
|---|
| 244 | this.list.push(this.node);
|
|---|
| 245 | }
|
|---|
| 246 |
|
|---|
| 247 | if (chr === '\n') {
|
|---|
| 248 | // Convert newlines to spaces. Carriage return is ignored as \r and \n usually
|
|---|
| 249 | // go together anyway and there already is a WS for \n. Lone \r means something is fishy.
|
|---|
| 250 | chr = ' ';
|
|---|
| 251 | }
|
|---|
| 252 |
|
|---|
| 253 | if (chr.charCodeAt(0) >= 0x21 || [' ', '\t'].includes(chr)) {
|
|---|
| 254 | // skip command bytes
|
|---|
| 255 | this.node.value += chr;
|
|---|
| 256 | }
|
|---|
| 257 |
|
|---|
| 258 | this.escaped = false;
|
|---|
| 259 | }
|
|---|
| 260 | }
|
|---|
| 261 |
|
|---|
| 262 | /**
|
|---|
| 263 | * Parses structured e-mail addresses from an address field
|
|---|
| 264 | *
|
|---|
| 265 | * Example:
|
|---|
| 266 | *
|
|---|
| 267 | * 'Name <address@domain>'
|
|---|
| 268 | *
|
|---|
| 269 | * will be converted to
|
|---|
| 270 | *
|
|---|
| 271 | * [{name: 'Name', address: 'address@domain'}]
|
|---|
| 272 | *
|
|---|
| 273 | * @param {String} str Address field
|
|---|
| 274 | * @return {Array} An array of address objects
|
|---|
| 275 | */
|
|---|
| 276 | function addressparser(str, options) {
|
|---|
| 277 | options = options || {};
|
|---|
| 278 |
|
|---|
| 279 | let tokenizer = new Tokenizer(str);
|
|---|
| 280 | let tokens = tokenizer.tokenize();
|
|---|
| 281 |
|
|---|
| 282 | let addresses = [];
|
|---|
| 283 | let address = [];
|
|---|
| 284 | let parsedAddresses = [];
|
|---|
| 285 |
|
|---|
| 286 | tokens.forEach(token => {
|
|---|
| 287 | if (token.type === 'operator' && (token.value === ',' || token.value === ';')) {
|
|---|
| 288 | if (address.length) {
|
|---|
| 289 | addresses.push(address);
|
|---|
| 290 | }
|
|---|
| 291 | address = [];
|
|---|
| 292 | } else {
|
|---|
| 293 | address.push(token);
|
|---|
| 294 | }
|
|---|
| 295 | });
|
|---|
| 296 |
|
|---|
| 297 | if (address.length) {
|
|---|
| 298 | addresses.push(address);
|
|---|
| 299 | }
|
|---|
| 300 |
|
|---|
| 301 | addresses.forEach(address => {
|
|---|
| 302 | address = _handleAddress(address);
|
|---|
| 303 | if (address.length) {
|
|---|
| 304 | parsedAddresses = parsedAddresses.concat(address);
|
|---|
| 305 | }
|
|---|
| 306 | });
|
|---|
| 307 |
|
|---|
| 308 | if (options.flatten) {
|
|---|
| 309 | let addresses = [];
|
|---|
| 310 | let walkAddressList = list => {
|
|---|
| 311 | list.forEach(address => {
|
|---|
| 312 | if (address.group) {
|
|---|
| 313 | return walkAddressList(address.group);
|
|---|
| 314 | } else {
|
|---|
| 315 | addresses.push(address);
|
|---|
| 316 | }
|
|---|
| 317 | });
|
|---|
| 318 | };
|
|---|
| 319 | walkAddressList(parsedAddresses);
|
|---|
| 320 | return addresses;
|
|---|
| 321 | }
|
|---|
| 322 |
|
|---|
| 323 | return parsedAddresses;
|
|---|
| 324 | }
|
|---|
| 325 |
|
|---|
| 326 | // expose to the world
|
|---|
| 327 | module.exports = addressparser;
|
|---|