From b122279ece06a9381e805c6087f130b41616fe3d Mon Sep 17 00:00:00 2001 From: Geo Halkiadakis Date: Fri, 12 Apr 2024 18:31:00 +0300 Subject: added utils folder; added several suplamentary modules --- pieces/kb-util.js | 84 ------------- pieces/match-util.js | 92 -------------- pieces/prepare-streams.js | 5 +- pieces/retro-search.js | 304 ++++++++++++++++++++++++++++++++++++++++++++++ pieces/search.js | 303 --------------------------------------------- pieces/suggest.js | 4 +- 6 files changed, 309 insertions(+), 483 deletions(-) delete mode 100644 pieces/kb-util.js delete mode 100644 pieces/match-util.js create mode 100644 pieces/retro-search.js delete mode 100644 pieces/search.js (limited to 'pieces') diff --git a/pieces/kb-util.js b/pieces/kb-util.js deleted file mode 100644 index 5441861..0000000 --- a/pieces/kb-util.js +++ /dev/null @@ -1,84 +0,0 @@ -// suplamentary arrays (mostly for cache) -// --- -- -- - - - - -var ORiGiNal = 'ςερτυθιοπασδφγηξκλζχψωβνμΕΡΤΥΘΙΟΠΑΣΔΦΓΗΞΚΛΖΧΨΩΒΝΜάέήίόύώϊΐϋΆΈΉΊΌΎΏΪΫQWERTYUIOPASDFGHJKLZXCVBNMqwertyuiopasdfghjklzxcvbnm0123456789- '.split(''); - -var kbKeyZed = 'sertyuiopasdfghjklzxcvbnmertyuiopasdfghjklzxcvbnmaehioyviiyaehioyviyqwertyuiopasdfghjklzxcvbnmqwertyuiopasdfghjklzxcvbnm0123456789- '.split(''); - -var map = new Map(); -for (var i=0; i { - ap = pair.split(' '); - accented_vowels.push({ - a: ap[0], // accented - p: ap[1] // pure = non accended - }); -}); - - -// translates string to keyboard-latin keys -// (the ones that used whan typing each letter of the word) -const keyboardize = (str) => { - str = str.replace('\'',''); - var out = ''; - // [map]'s implementation is 40x faster than [for]'s - for (var i=0 ; i< str.length; i++) out += map.get(str[i]); - return out; -} - -// keyboardize an array of strings -const keyb_array = (arr) => { - kb_arr = []; - arr.forEach( w => { - kb_arr.push(keyboardize(w)); - }); - return kb_arr; -} - - -// transforms to lowercase; handles sigma-teliko -const sanitizeGR = (str) => { - str = str.toLowerCase(); - - // replace accended vowels with pure ones - accented_vowels.forEach( v => { - str = str.replaceAll(v.a, v.p); - }); - - // replace sigma on the end of words - str = str + ' '; - str = str.replaceAll('σ-', 'ς-'); - str = str.replaceAll('σ ', 'ς '); - - return str; -} - -// removes non keyword characters [+ . , !] and internal multiple-spaces -// @param txt (string): product description -const clean = (txt) => { - return txt.replace('+',' ').replace('.',' ').replace(',',' ') // change to space - .replace('!','').replace('\"', '') // remove character - .replace(' ',' ').replace(' ',' '); // remove multiple spaces -} - - -// exports -// --- -- -- - - - - -module.exports = { - map, - accented_vowels, - keyboardize, - keyb_array, - sanitizeGR, - clean -}; \ No newline at end of file diff --git a/pieces/match-util.js b/pieces/match-util.js deleted file mode 100644 index a13fddb..0000000 --- a/pieces/match-util.js +++ /dev/null @@ -1,92 +0,0 @@ -/** Ngram fuzzy match algorithm - * (simple and fast) - */ -const createNgram = (word, n) => { // Ngram creation - if (word.length <3) return word; - const vector = []; - for (let i = 0; i < word.length-n+1; ++i) { - vector.push(word.slice(i, i + n)); - } - return vector; -}; - -/** similarity - * rates similarity between 2 words - * based on Ngram matches of N = n letters; - * implements a 2-dim check (all a-Ngrams vs all all b-Ngrams) - * - * @param {string} a : first word - * @param {string} b : second word - * @param {int} n : Ngram base - * @returns {float} : match percentage as a float in [0, 1] - */ -const similarity = (a, b, n) => { // Ngram match score - if (a.length > 0 && b.length > 0) { - const aNgram = createNgram(a, n); - const bNgram = createNgram(b, n); - let hits = 0; - for (let x = 0; x < aNgram.length; ++x) { - for (let y = 0; y < bNgram.length; ++y) { - if (aNgram[x] === bNgram[y]) { - hits += 1; - } - } - } - if (hits > 0) { - const union = aNgram.length + bNgram.length; - return (2.0 * hits) / union; - } - } - return 0; -}; - -/** resemblance - * is an alternative similarity rating; - * implements an 1-dim Ngram similarity check - * and it's much faster than similarity() - */ -const resemblance = (a, b, n) => { - if (a.length > n && b.length >= a.length) { - const aNgram = createNgram(a, n); - let hits = 0; - for (let i = 0; i < aNgram.length; ++i) { - if (b.includes(aNgram[i])) { - hits++; - } - } - if (hits > 0) { - // rate resemblance based on hits and length-similarity - return (hits / aNgram.length) * (a.length / b.length); - } - } - return 0; -} - - -/** is_exact_match - * - * check if a searching string -> query (string/latin in kb-format) - * matches exactly an item of the array of synonyms -> chkArr (array of utf-8/strings) - * - * @param query (string): searching string; string/latin in kb-format - * @param chkArr (array): array of synonyms; (array of utf-8/strings) - * @return (boolean): true|false - */ -function exact( query, chkArr ) { - found = false; - chkArr.forEach( w => { if (w == query) found = true }); - return found; -} - -function partial( query, chkArr ) { - found = false; - chkArr.forEach( w => { if (w.includes(query)) found = true }); - return found; -} - -module.exports = { - exact, - partial, - similarity, - resemblance -} diff --git a/pieces/prepare-streams.js b/pieces/prepare-streams.js index 4fa0a7e..ee7ba8e 100644 --- a/pieces/prepare-streams.js +++ b/pieces/prepare-streams.js @@ -1,8 +1,9 @@ const fs = require('fs'); -const kb = require('./kb-util.js'); - var request = require('request'); +const kb = require('../utils/kb-util.js'); + + // const https = require("https"); diff --git a/pieces/retro-search.js b/pieces/retro-search.js new file mode 100644 index 0000000..ba4d937 --- /dev/null +++ b/pieces/retro-search.js @@ -0,0 +1,304 @@ +const kb = require('../utils/kb-util.js'); +const match = require('../utils/match-util.js'); + +const products = require('../data/products.json'); + +/** VARIABLES + * may passed as module arguments + * ----------------------------------------------------------------------------- + *////////////////////////////////////////////////////////////////////////////// + + +var _allowFuzzy = true; // enable|disable fuzzy search +var n = 2; // Ngram base +var _fuzzyLimit = .5; // minimum bigram score for being considered a match +var max_list = 24; +var tolerance = 42; +var STORE = { id: 904 }; + +var _kwlinks; // keyword links (word-connections; imported via ajax-get) +var _products = []; // all products (imported via ajax-get) + +// setup options +var _maxResults = options.max_list; // limit suggestions +var _blendProds = 4; // minimum final-produncts to blend with next-word suggestions +var _Ngram_base = 2; // number of N in Ngram spliting algorithm +var _isReady = false; // whether the searchbox is ready to be used + +// product keywords +var keywordsURL = options.keywords_json; + +var cursor_on = { none: true }; // what product is highlighted; if not on product then { none: true } + + + + + + +/** SUPPLEMENTARY FUNCTIONS + * ----------------------------------------------------------------------------- + *////////////////////////////////////////////////////////////////////////////// + + + +// callback function for sorting resulrs per r (=rating) property +function compare_rate(a,b) { + return (a.r < b.r); +} + + + +/** SEARCH ENGINE + * ------------------------------------------------------------------------- + *////////////////////////////////////////////////////////////////////////// + + + + +function matchWordInList(q, list = false) { + + if (list !== false && list.length == 0) return []; // no results + + var result = []; + var firstPass = false; + + if (list === false) { + firstPass = true; // on first pass + list = products; // list is all products + } + + for(let i = 0 ; i < list.length ; i++) { + + // check exact + rate + + // else check partial + rate + + // else check similarity + rate + + let found = false; + let similarity = 0; + products[i].kb.split(' ').forEach( w => { + let sim = match.resemblance(q, w, 2); + if (sim > 0.5) { + found = true; + similarity = (sim > similarity) ? sim : similarity; + } + }); + if (found) { + products[i].similarity = similarity; + products[i].rating = (firstPass) + ? similarity * 2.0 + : products[i].rating + similarity * 2.0 + result.push(products[i]); + } + } + + return result +} + + + +// suggestions engine ////////////////////////////////////////////////// +// --- +function suggestions_engine(query) { + var results = []; + var pot = []; pot.length = 0; + + // clean and sanitize and mark links onto q(uery) string + var q = kb.keyboardize( kb.sanitizeGR( kb.clean(query.trim()) ) ).trim(); + + // TODO: + // construct direct-linked words + // = do unequivocally replaces + // steps: + // 1. replace accented vowels with non accented ones + // 2. replace `/some pattern/gi , 'SOME-REPLACE-PATTERN'` + + var qAr = q.split(' '); // split to words + /// if (space_ended) qAr.push(' '); // if space-end existed, push a space to query array + /// + /// if (qAr.slice(-1) == "") { + /// qAr.pop(); + /// } + + + pot = _products; // potential results // NOTE: CRITICAL: BY REFERENCE + + var wi = 0; // word index (from list) + var wc = qAr.length; + + qAr.forEach( w => { + + let sf = []; // (matches) so far + let mi; // position of match + wi++; + + pot.forEach( it => { + let matched = false; + let tester = ' '+ it.kb + ' '; + + // reset previous history and ratings + if (wi == 1) { + it.r = 0; + it.history = []; + } + + + // rate word-match > start-match > simple-match + // ... up to 8 points + + if (tester.indexOf(' '+ w +' ') != -1) { + it.r += 9; + it.history.push({ w: w, rate: 9 }); + matched = true; + } + else if (tester.indexOf(' '+ w) != -1) { + it.r += 5; + it.history.push({ w: w, rate: 5 }); + matched = true; + } + else if (tester.indexOf(w) != -1) { + it.r += 2; + it.history.push({ w: w, rate: 2 }); + matched = true; + } + + // rate `near-to-start` matching .. up to 7p + // rate `earlyness` of word in query .. up to 7p + + if ((mi = tester.indexOf(' '+w)) != -1) { + let fc1 = 100 - ((mi < 99) ? mi : 99); // near-to-start factor + let fc2 = wc - wi + 1; // query earlyness factor + let r1 = Math.floor(7*fc1/100); + let r2 = Math.floor(7*fc2/wc); + + it.r += (r1 + r2); + it.history.push({ w: w, left: [fc1, r1], early: [fc2, r2] }); + } + if (matched) sf.push(it); + }); + + if ((sf.length > (_maxResults + Math.floor(_maxResults/2))) + || (wi == 1) ) { + // ..if pot has a fair amount (= max + 50%) of results + // ..or these are results of '1st-query-word' + // set sf as new source + pot.lenght = 0; pot = []; + pot = JSON.parse(JSON.stringify(sf)); // copy by value + + } else { + // else.. keep the source list and increase of 'so-far rating' + // console.log('found small list', sf, pot) + pot.forEach( it => { + sf.forEach( si => { + if (it.id == si.id) { + it.r += 10; + it.history.push({ w: w, plus: '+10'}); + } + }); + }); + } + }); + + // sort results, get max-list of best rated + results = (pot.length > _maxResults) + ? pot.sort(compare_rate).slice(0, _maxResults) + : pot.sort(compare_rate) + + if (options.debug) console.log(results); + + return results; +} + + +// sub-module (start) +//////////////////////////////////////////////////////////////////////////// + +function update_common_search_results(q, results) { + let queries = getSessionObj('sr'); + let newSRlist = []; + let isnewQ = true; + if (queries === null) { + setSessionObj('sr', [{ + q: q, + result: result, + t: + new Date() + }]); + return true; + + } else { + + queries.forEach(it => { + if (it.q == q) { + newSRlist.push({ + q:q, + result: result, + t: + new Date() + }); + isnewQ = false; + } else { newSRlist.push(it); } + }); + + if (isnewQ) { + newSRlist.push({ + q:q, + result: result, + t: + new Date() + }); + } + + return true; + } +} + + + +//////////////////////////////////////////////////////////////////////////// +// sub-module (end) + + +function common_search(query) { + // clear ; sanitize ; split + var qAr = keyboardize( sanitize_GR( clean_text(query) ) ).toLowerCase().split(' '); + + // if last item is empty, remove it + if ((qAr.slice(-1) == ' ') || (qAr.slice(-1) == '')) qAr.pop() + + var results = _products; + + // for each key fitler results + qAr.forEach( key => { + results = key_sublist(key, results) + }); + + // echo products (and prepare list to POST) + var list_ = []; + results.forEach( item => { + if (options.debug) console.log(item.id, ':', item.w); + list_.push(item.id) + }) + + // *** TODO: keep results in local storage (or on session storage) + + // update_common_search_results(query, list_); + + let l = list_.join(','); + var url = encodeURI(`${options.visualize_search_results_url}?search=${query}&eys_code=${l}`); + + console.log('common search: search query > location = search') + window.location.href = encodeURI(`${options.visualize_search_results_url}?search=${query}`); + +} + +/** return from list only items that include 'key' + */ +function key_sublist(key, list) { + var result = []; + list.forEach( item => { + if (item.kb.includes(key)) { + result.push(item); + } + }); + + return result; +} diff --git a/pieces/search.js b/pieces/search.js deleted file mode 100644 index 0c05708..0000000 --- a/pieces/search.js +++ /dev/null @@ -1,303 +0,0 @@ -const kb = require('./kb-util.js'); -const match = require('./match-util.js'); -const products = require('../data/products.json'); - -/** VARIABLES - * may passed as module arguments - * ----------------------------------------------------------------------------- - *////////////////////////////////////////////////////////////////////////////// - - -var _allowFuzzy = true; // enable|disable fuzzy search -var n = 2; // Ngram base -var _fuzzyLimit = .5; // minimum bigram score for being considered a match -var max_list = 24; -var tolerance = 42; -var STORE = { id: 904 }; - -var _kwlinks; // keyword links (word-connections; imported via ajax-get) -var _products = []; // all products (imported via ajax-get) - -// setup options -var _maxResults = options.max_list; // limit suggestions -var _blendProds = 4; // minimum final-produncts to blend with next-word suggestions -var _Ngram_base = 2; // number of N in Ngram spliting algorithm -var _isReady = false; // whether the searchbox is ready to be used - -// product keywords -var keywordsURL = options.keywords_json; - -var cursor_on = { none: true }; // what product is highlighted; if not on product then { none: true } - - - - - - -/** SUPPLEMENTARY FUNCTIONS - * ----------------------------------------------------------------------------- - *////////////////////////////////////////////////////////////////////////////// - - - -// callback function for sorting resulrs per r (=rating) property -function compare_rate(a,b) { - return (a.r < b.r); -} - - - -/** SEARCH ENGINE - * ------------------------------------------------------------------------- - *////////////////////////////////////////////////////////////////////////// - - - - -function matchWordInList(q, list = false) { - - if (list !== false && list.length == 0) return []; // no results - - var result = []; - var firstPass = false; - - if (list === false) { - firstPass = true; // on first pass - list = products; // list is all products - } - - for(let i = 0 ; i < list.length ; i++) { - - // check exact + rate - - // else check partial + rate - - // else check similarity + rate - - let found = false; - let similarity = 0; - products[i].kb.split(' ').forEach( w => { - let sim = match.resemblance(q, w, 2); - if (sim > 0.5) { - found = true; - similarity = (sim > similarity) ? sim : similarity; - } - }); - if (found) { - products[i].similarity = similarity; - products[i].rating = (firstPass) - ? similarity * 2.0 - : products[i].rating + similarity * 2.0 - result.push(products[i]); - } - } - - return result -} - - - -// suggestions engine ////////////////////////////////////////////////// -// --- -function suggestions_engine(query) { - var results = []; - var pot = []; pot.length = 0; - - // clean and sanitize and mark links onto q(uery) string - var q = kb.keyboardize( kb.sanitizeGR( kb.clean(query.trim()) ) ).trim(); - - // TODO: - // construct direct-linked words - // = do unequivocally replaces - // steps: - // 1. replace accented vowels with non accented ones - // 2. replace `/some pattern/gi , 'SOME-REPLACE-PATTERN'` - - var qAr = q.split(' '); // split to words - /// if (space_ended) qAr.push(' '); // if space-end existed, push a space to query array - /// - /// if (qAr.slice(-1) == "") { - /// qAr.pop(); - /// } - - - pot = _products; // potential results // NOTE: CRITICAL: BY REFERENCE - - var wi = 0; // word index (from list) - var wc = qAr.length; - - qAr.forEach( w => { - - let sf = []; // (matches) so far - let mi; // position of match - wi++; - - pot.forEach( it => { - let matched = false; - let tester = ' '+ it.kb + ' '; - - // reset previous history and ratings - if (wi == 1) { - it.r = 0; - it.history = []; - } - - - // rate word-match > start-match > simple-match - // ... up to 8 points - - if (tester.indexOf(' '+ w +' ') != -1) { - it.r += 9; - it.history.push({ w: w, rate: 9 }); - matched = true; - } - else if (tester.indexOf(' '+ w) != -1) { - it.r += 5; - it.history.push({ w: w, rate: 5 }); - matched = true; - } - else if (tester.indexOf(w) != -1) { - it.r += 2; - it.history.push({ w: w, rate: 2 }); - matched = true; - } - - // rate `near-to-start` matching .. up to 7p - // rate `earlyness` of word in query .. up to 7p - - if ((mi = tester.indexOf(' '+w)) != -1) { - let fc1 = 100 - ((mi < 99) ? mi : 99); // near-to-start factor - let fc2 = wc - wi + 1; // query earlyness factor - let r1 = Math.floor(7*fc1/100); - let r2 = Math.floor(7*fc2/wc); - - it.r += (r1 + r2); - it.history.push({ w: w, left: [fc1, r1], early: [fc2, r2] }); - } - if (matched) sf.push(it); - }); - - if ((sf.length > (_maxResults + Math.floor(_maxResults/2))) - || (wi == 1) ) { - // ..if pot has a fair amount (= max + 50%) of results - // ..or these are results of '1st-query-word' - // set sf as new source - pot.lenght = 0; pot = []; - pot = JSON.parse(JSON.stringify(sf)); // copy by value - - } else { - // else.. keep the source list and increase of 'so-far rating' - // console.log('found small list', sf, pot) - pot.forEach( it => { - sf.forEach( si => { - if (it.id == si.id) { - it.r += 10; - it.history.push({ w: w, plus: '+10'}); - } - }); - }); - } - }); - - // sort results, get max-list of best rated - results = (pot.length > _maxResults) - ? pot.sort(compare_rate).slice(0, _maxResults) - : pot.sort(compare_rate) - - if (options.debug) console.log(results); - - return results; -} - - -// sub-module (start) -//////////////////////////////////////////////////////////////////////////// - -function update_common_search_results(q, results) { - let queries = getSessionObj('sr'); - let newSRlist = []; - let isnewQ = true; - if (queries === null) { - setSessionObj('sr', [{ - q: q, - result: result, - t: + new Date() - }]); - return true; - - } else { - - queries.forEach(it => { - if (it.q == q) { - newSRlist.push({ - q:q, - result: result, - t: + new Date() - }); - isnewQ = false; - } else { newSRlist.push(it); } - }); - - if (isnewQ) { - newSRlist.push({ - q:q, - result: result, - t: + new Date() - }); - } - - return true; - } -} - - - -//////////////////////////////////////////////////////////////////////////// -// sub-module (end) - - -function common_search(query) { - // clear ; sanitize ; split - var qAr = keyboardize( sanitize_GR( clean_text(query) ) ).toLowerCase().split(' '); - - // if last item is empty, remove it - if ((qAr.slice(-1) == ' ') || (qAr.slice(-1) == '')) qAr.pop() - - var results = _products; - - // for each key fitler results - qAr.forEach( key => { - results = key_sublist(key, results) - }); - - // echo products (and prepare list to POST) - var list_ = []; - results.forEach( item => { - if (options.debug) console.log(item.id, ':', item.w); - list_.push(item.id) - }) - - // *** TODO: keep results in local storage (or on session storage) - - // update_common_search_results(query, list_); - - let l = list_.join(','); - var url = encodeURI(`${options.visualize_search_results_url}?search=${query}&eys_code=${l}`); - - console.log('common search: search query > location = search') - window.location.href = encodeURI(`${options.visualize_search_results_url}?search=${query}`); - -} - -/** return from list only items that include 'key' - */ -function key_sublist(key, list) { - var result = []; - list.forEach( item => { - if (item.kb.includes(key)) { - result.push(item); - } - }); - - return result; -} diff --git a/pieces/suggest.js b/pieces/suggest.js index a420edd..c0a6e04 100644 --- a/pieces/suggest.js +++ b/pieces/suggest.js @@ -1,5 +1,5 @@ -const kb = require('./kb-util.js'); -const match = require('./match-util.js'); +const kb = require('../utils/kb-util.js'); +const match = require('../utils/match-util.js'); var _allowFuzzy = true; // enable|disable fuzzy search var n = 2; // Ngram base -- cgit v1.2.3