summaryrefslogtreecommitdiff
path: root/utils
diff options
context:
space:
mode:
authorGeo Halkiadakis <gchalkiadakis@sklavenitis.co.gr>2024-04-12 18:31:00 +0300
committerGeo Halkiadakis <gchalkiadakis@sklavenitis.co.gr>2024-04-12 18:31:00 +0300
commitb122279ece06a9381e805c6087f130b41616fe3d (patch)
tree2ec0a8fc3c570417d972b18a7a7b878758ce3a6f /utils
parent762d85c95690f510bec4dd8c37c05454ae1be4fc (diff)
downloadoseine-b122279ece06a9381e805c6087f130b41616fe3d.tar.gz
oseine-b122279ece06a9381e805c6087f130b41616fe3d.tar.bz2
oseine-b122279ece06a9381e805c6087f130b41616fe3d.zip
added utils folder; added several suplamentary modules
Diffstat (limited to 'utils')
-rw-r--r--utils/kb-util.js88
-rw-r--r--utils/match-util.js92
-rw-r--r--utils/mem-usage.js20
-rw-r--r--utils/url-util.js23
4 files changed, 223 insertions, 0 deletions
diff --git a/utils/kb-util.js b/utils/kb-util.js
new file mode 100644
index 0000000..aac88f9
--- /dev/null
+++ b/utils/kb-util.js
@@ -0,0 +1,88 @@
+/**
+ * fast string manipulation utilities
+ * for bi-lingual (EL/EN) words/phrases
+ * based on the keyboard layout
+ */
+
+// suplamentary arrays (mostly for cache)
+// --- -- -- - - -
+
+var ORiGiNal = 'ςερτυθιοπασδφγηξκλζχψωβνμΕΡΤΥΘΙΟΠΑΣΔΦΓΗΞΚΛΖΧΨΩΒΝΜάέήίόύώϊΐϋΆΈΉΊΌΎΏΪΫQWERTYUIOPASDFGHJKLZXCVBNMqwertyuiopasdfghjklzxcvbnm0123456789- '.split('');
+
+var kbKeyZed = 'sertyuiopasdfghjklzxcvbnmertyuiopasdfghjklzxcvbnmaehioyviiyaehioyviyqwertyuiopasdfghjklzxcvbnmqwertyuiopasdfghjklzxcvbnm0123456789- '.split('');
+
+var map = new Map();
+for (var i=0; i<ORiGiNal.length; i++) map.set(ORiGiNal[i], kbKeyZed[i]);
+
+
+// cache (=create a global array)
+// of accended to non-accended vowels mapping
+// --- -- -- - - -
+accented_vowels = [];
+[
+ 'ά α', 'έ ε', 'ή η', 'ί ι', 'ϊ ι', 'ΐ ι', 'ό ο', 'ύ υ', 'ϋ υ', 'ώ ω',
+ 'Ά Α', 'Έ Ε', 'Ή Η', 'Ί Ι', 'Ϊ Ι', 'Ό Ο', 'Ύ Υ', 'Ϋ Υ', 'Ώ Ω'
+].forEach( pair => {
+ ap = pair.split(' ');
+ accented_vowels.push({
+ a: ap[0], // accented
+ p: ap[1] // pure = non accended
+ });
+});
+
+
+// translates string to keyboard-latin keys
+// (the ones that used whan typing each letter of the word)
+const keyboardize = (str) => {
+ str = str.replace('\'','');
+ var out = '';
+ // [map]'s implementation is 40x faster than [for]'s
+ for (var i=0 ; i< str.length; i++) out += map.get(str[i]);
+ return out;
+}
+
+// keyboardize an array of strings
+const keyb_array = (arr) => {
+ kb_arr = [];
+ arr.forEach( w => {
+ kb_arr.push(keyboardize(w));
+ });
+ return kb_arr;
+}
+
+
+// transforms to lowercase; handles sigma-teliko
+const sanitizeGR = (str) => {
+ str = str.toLowerCase();
+
+ // replace accended vowels with pure ones
+ accented_vowels.forEach( v => {
+ str = str.replaceAll(v.a, v.p);
+ });
+
+ // replace sigma on the end of words
+ str = str + ' ';
+ str = str.replaceAll('σ-', 'ς-');
+ str = str.replaceAll('σ ', 'ς ');
+
+ return str;
+}
+
+// removes non keyword characters [+ . , !] and internal multiple-spaces
+// @param txt (string): product description
+const clean = (txt) => {
+ return txt.replace('+',' ').replace('.',' ').replace(',',' ') // change to space
+ .replace('!','').replace('\"', '') // remove character
+ .replace(' ',' ').replace(' ',' '); // remove multiple spaces
+}
+
+
+// exports
+// --- -- -- - - -
+
+module.exports = {
+ keyboardize,
+ keyb_array,
+ sanitizeGR,
+ clean
+}; \ No newline at end of file
diff --git a/utils/match-util.js b/utils/match-util.js
new file mode 100644
index 0000000..a13fddb
--- /dev/null
+++ b/utils/match-util.js
@@ -0,0 +1,92 @@
+/** Ngram fuzzy match algorithm
+ * (simple and fast)
+ */
+const createNgram = (word, n) => { // Ngram creation
+ if (word.length <3) return word;
+ const vector = [];
+ for (let i = 0; i < word.length-n+1; ++i) {
+ vector.push(word.slice(i, i + n));
+ }
+ return vector;
+};
+
+/** similarity
+ * rates similarity between 2 words
+ * based on Ngram matches of N = n letters;
+ * implements a 2-dim check (all a-Ngrams vs all all b-Ngrams)
+ *
+ * @param {string} a : first word
+ * @param {string} b : second word
+ * @param {int} n : Ngram base
+ * @returns {float} : match percentage as a float in [0, 1]
+ */
+const similarity = (a, b, n) => { // Ngram match score
+ if (a.length > 0 && b.length > 0) {
+ const aNgram = createNgram(a, n);
+ const bNgram = createNgram(b, n);
+ let hits = 0;
+ for (let x = 0; x < aNgram.length; ++x) {
+ for (let y = 0; y < bNgram.length; ++y) {
+ if (aNgram[x] === bNgram[y]) {
+ hits += 1;
+ }
+ }
+ }
+ if (hits > 0) {
+ const union = aNgram.length + bNgram.length;
+ return (2.0 * hits) / union;
+ }
+ }
+ return 0;
+};
+
+/** resemblance
+ * is an alternative similarity rating;
+ * implements an 1-dim Ngram similarity check
+ * and it's much faster than similarity()
+ */
+const resemblance = (a, b, n) => {
+ if (a.length > n && b.length >= a.length) {
+ const aNgram = createNgram(a, n);
+ let hits = 0;
+ for (let i = 0; i < aNgram.length; ++i) {
+ if (b.includes(aNgram[i])) {
+ hits++;
+ }
+ }
+ if (hits > 0) {
+ // rate resemblance based on hits and length-similarity
+ return (hits / aNgram.length) * (a.length / b.length);
+ }
+ }
+ return 0;
+}
+
+
+/** is_exact_match
+ *
+ * check if a searching string -> query (string/latin in kb-format)
+ * matches exactly an item of the array of synonyms -> chkArr (array of utf-8/strings)
+ *
+ * @param query (string): searching string; string/latin in kb-format
+ * @param chkArr (array): array of synonyms; (array of utf-8/strings)
+ * @return (boolean): true|false
+ */
+function exact( query, chkArr ) {
+ found = false;
+ chkArr.forEach( w => { if (w == query) found = true });
+ return found;
+}
+
+function partial( query, chkArr ) {
+ found = false;
+ chkArr.forEach( w => { if (w.includes(query)) found = true });
+ return found;
+}
+
+module.exports = {
+ exact,
+ partial,
+ similarity,
+ resemblance
+}
diff --git a/utils/mem-usage.js b/utils/mem-usage.js
new file mode 100644
index 0000000..766a93e
--- /dev/null
+++ b/utils/mem-usage.js
@@ -0,0 +1,20 @@
+/**
+ * memory usage report utility
+ */
+
+const formatMemoryUsage = (data) => `${Math.round(data / 1024 / 1024 * 100) / 100} MB`;
+
+function report() {
+ let memoryData = process.memoryUsage();
+
+ let memoryUsage = {
+ rss: `${formatMemoryUsage(memoryData.rss)} -> Resident Set Size - total memory allocated for the process execution`,
+ heapTotal: `${formatMemoryUsage(memoryData.heapTotal)} -> total size of the allocated heap`,
+ heapUsed: `${formatMemoryUsage(memoryData.heapUsed)} -> actual memory used during the execution`,
+ external: `${formatMemoryUsage(memoryData.external)} -> V8 external memory`,
+ };
+
+ console.log(memoryUsage);
+}
+
+module.exports = { report }
diff --git a/utils/url-util.js b/utils/url-util.js
new file mode 100644
index 0000000..e86ad9b
--- /dev/null
+++ b/utils/url-util.js
@@ -0,0 +1,23 @@
+/**
+ * url utility
+ */
+
+const querystring = require('querystring');
+
+function struct(req, url) {
+ let url_parts = url.split('?');
+ let query = (url_parts.length > 1)
+ ? querystring.decode(url_parts[1])
+ : {};
+ return {
+ method: req.method,
+ host: req.host,
+ path: url_parts[0],
+ query: query
+ }
+}
+
+/**
+ * exports
+ */
+module.exports = { struct } \ No newline at end of file