summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorAlon Levy <alon@pobox.com>2015-02-24 14:24:49 +0200
committerAlon Levy <alon@pobox.com>2015-02-24 14:24:49 +0200
commit6a9ed9a51476907d4510f06b66e4f7d8afd9c47c (patch)
treeeb4f0fc7a6227da16be4bd5ae960b23849a0ff9a /src
parentf4245f118d0c4c4d4f46fbbdc55177f56d9e1046 (diff)
client/textanalysis: add new node_token module global variable and use in tokenize
Diffstat (limited to 'src')
-rw-r--r--src/client/textanalysis.js15
1 files changed, 11 insertions, 4 deletions
diff --git a/src/client/textanalysis.js b/src/client/textanalysis.js
index 87395b8f..dd093cf3 100644
--- a/src/client/textanalysis.js
+++ b/src/client/textanalysis.js
@@ -2,6 +2,9 @@ define(['rz_core', 'model/core', 'model/util', 'model/diff', 'consts', 'util'],
function(rz_core, model_core, model_util, model_diff, consts, util) {
"use strict";
+// Constants
+var node_edge_separator = false;
+var separator_symbol = '#'; //' ';
var typeindex = 0,
nodetypes = consts.nodetypes,
@@ -44,6 +47,8 @@ function cleanup(text)
/**
* Tokenizer for input.
*
+ * Assumes the whole input string is available, used for lookahead via slice.
+ *
* node_token is it's own token, represented by itself.
*
* accepts a quotation char which allows whitespace in between.
@@ -76,6 +81,7 @@ function tokenize(text, node_token, quote)
prev = null,
prev_whitespace = true,
start = 0,
+ is_node_token,
next = function() {
if (token.length > 0) {
tokens.push({start: start, end: i, token: token.join('')});
@@ -85,7 +91,8 @@ function tokenize(text, node_token, quote)
};
for (i = 0 ; i < text.length; ++i) {
c = text[i];
- if (prev == '\\') {
+ is_node_token = text.slice(i, i + node_token.length) === node_token;
+ if (prev === '\\') {
token.push(c);
prev = null;
continue;
@@ -103,9 +110,9 @@ function tokenize(text, node_token, quote)
inquote = !inquote;
break;
default:
- if (c == node_token && prev_whitespace) {
- tokens.push({start: i, end: i + 1, token: node_token});
- start = i + 1;
+ if (is_node_token && (node_token.length > 1 || prev_whitespace)) {
+ tokens.push({start: i, end: i + node_token.length, token: node_token});
+ start = i + node_token.length;
} else {
token.push(c);
}