diff options
| -rw-r--r-- | javascript/keygen.js | 584 | ||||
| -rw-r--r-- | python/products-dict-v5.py | 14 |
2 files changed, 591 insertions, 7 deletions
diff --git a/javascript/keygen.js b/javascript/keygen.js new file mode 100644 index 0000000..cacd317 --- /dev/null +++ b/javascript/keygen.js @@ -0,0 +1,584 @@ +// requirements +//////////////////////////////////////////////////////////////////////////////// + +const fs = require('fs'); + +const os = require('os'); + +const fetch = require('node-fetch'); + + +// preloaded data +//////////////////////////////////////////////////////////////////////////////// + +// any-character to keyboard-latin mapping +// ----------------------------------------------------------------------------- +var ORiGiNal = 'ςερτυθιοπασδφγηξκλζχψωβνμΕΡΤΥΘΙΟΠΑΣΔΦΓΗΞΚΛΖΧΨΩΒΝΜάέήίόύώϊΐϋΆΈΉΊΌΎΏΪΫQWERTYUIOPASDFGHJKLZXCVBNMqwertyuiopasdfghjklzxcvbnm0123456789- '.split(''); +var kbKeyZed = 'sertyuiopasdfghjklzxcvbnmertyuiopasdfghjklzxcvbnmaehioyviiyaehioyviyqwertyuiopasdfghjklzxcvbnmqwertyuiopasdfghjklzxcvbnm0123456789- '.split(''); +const map = new Map(); +for (var i=0; i<ORiGiNal.length; i++) map.set(ORiGiNal[i], kbKeyZed[i]); + + +// synonyms +// ----------------------------------------------------------------------------- +var synonyms = []; // groups of synonyms +var synonym_kbs = []; // cache kb-formats for performance +var synonyms_Originals = [ + 'μπίρα μπύρα μπίρες μπύρες', + 'αυγά αβγά αυγό', + 'σίκαλης σικάλεως', + 'ξηρά ξερά', + 'ρολό ρολλό', + 'coca-cola cocacola coke', + 'χαρτί-υγείας ρολό-υγείας χαρτί-τουαλέτας', + 'χαρτί-κουζίνας ρολό-κουζίνας', + 'οινος κρασι', + 'ΚΑΤΣΕΛΗΣ ΚΑΤΣΕΛΗ', + 'DR-OETKER OETKER', + 'DR.BECKMANN BECKMANN', + 'NES-CAFE NESCAFE', + 'Ολικής-Άλεσης Ολικής-Aλέσεως Ολικής', + 'τσίπουρο ρακή', + 'Βρώμη Βρώμης', + 'Φράουλα Φράουλες Φράουλας', + 'Μαλλιά Μαλλιών', + 'Κέικ Cake', + 'CRETA-FARMS CRETA-FARM', + 'MARSEILLAIS LE-PETIT-MARSEILLAIS PETIT-MARSEILLAIS', + 'Γαϊδούρας Γαϊδάρου', + 'ΚΑΛΟΓΕΡΑΚΗΣ ΚΑΛΟΓΕΡΑΚΗ', + 'ΚΑΪΔΑΝΤΖΗΣ ΚΑΪΔΑΝΤΖΗ', + 'ΥΦΑΝΤΗΣ ΥΦΑΝΤΗ', + 'ΣΥΝΑΓΡΙΔΑ ΣΥΝΑΓΡΙΔΕΣ', + 'Ντομάτα Ντομάτας', + 'Ελαφρύ Ελαφρά Light', + 'Εγχώρια Εγχώριες Ελληνικό Ελληνική Ελληνικά', + 'τριμμένη τριμμένο', + 'Τόνος Τόνου', + 'Κριθαρένια κρίθινα', + 'Χωρίς-Kαφεϊνη Decaffeine', + 'Το-Μάννα Μάννα', + 'Κράνμπερι Κράνμπερις', + 'Κρήτης Κρητικό', + 'Πέννες Πένες', + 'Μακαρόνια Σπαγγέτι Σπαγγετίνι Σπαγγετόνι', + 'Καρτέλλα Καρτέλα Καρτέλλες' +] +synonyms_Originals.forEach( grp => { + synonyms.push( grp.split(' ') ); + synonym_kbs.push( kb_trans(grp).split(' ') ); +}) + +// significant terms +// ----------------------------------------------------------------------------- +significantExceptios = '7UP 3ΑΛΦΑ 17 3Π 7DAYS K2R'.split(' ') + + +// replaces (correcting descriptions) +// ----------------------------------------------------------------------------- +replaces = []; +replaceSource = [ + '3 ΑΛΦΑ ;3ΑΛΦΑ ', + 'HEAD & SHOULDERS ;HEAD&SHOULDERS ', + 'W.K Kellogg ; ', + 'ΦΙΛΕΤ ;Φιλέτο ', + 'ΕΝΕΛΛΑΔ ;Εν-Ελλάδι ', + 'ΓΑΛΟΠΟΥΛ ;Γαλοπούλα ', + '7 DAYS ;7DAYS ', + 'ΜΠΑΡΜΠΑ ΣΤΑΘΗ ;ΜΠΑΡΜΠΑ-ΣΤΑΘΗΣ ' +] +replaceSource.forEach( it => { + st = it.split(';'); + replaces.push({ src: st[0], trg: st[1] }); +}); + + +// words that shall not be searched first +// ----------------------------------------------------------------------------- +var noRootKeywords = []; +noRoot = [ + 'χωρίς', + 'εισαγωγής', + 'δώρο', + 'γεύση', + 'γεύσεις', + 'φέτες', + 'Χωρίς-Γλουτένη', + 'Γλουτένη', + 'Λακτόζη', + 'Παστερίωσης', + 'Ανθρακικό', + 'Συντηρητικά', + 'Χωρίς-Ζάχαρη', + 'Άλεσης', + 'Χωρίς-Αλάτι', + 'Χωρίς-Λακτόζη', + 'Χωρίς-Συντηρητικά', + 'Χωρίς-Αλκοόλ', + 'Χωρίς-Kαφεϊνη', + 'Χωρίς-Γλυκάνισο', + 'Χωρίς-Ανθρακικό', + 'Υψηλής-Παστερίωσης', + 'Ολες-τις-Χρήσεις', + 'Ολικής-Άλεσης', + 'Ολικής-Aλέσεως', + 'Ολικής', + 'Γαϊδούρας', + 'Γαϊδάρου', + 'Ρούχων', + 'Πιάτων', + 'πλύσεις', + 'Πλυντηρίου', + 'Φύλλων', + 'Γάλακτος', + 'Χρήσης', + 'Τύπου', + 'Ολλανδίας', + 'Απορριμμάτων', + 'Medium', + 'Μαλλιά', + 'Μαλλιών', + 'Γενικής', + 'Plus', + 'Classic', + 'Έκπληξη', + 'Μάνης', + 'Ελάτου', + 'Άγριων', + 'Βοτάνων', + 'Λακωνίας', + 'ΠΑΡΑΓΓΕΛΙΩΝ' +] +noRoot.forEach( w => { noRootKeywords.push(kb_trans(w)); }); + + +// list of linked-words +// ----------------------------------------------------------------------------- +linkedWords = [ + 'Χωρίς-Γλουτένη', + 'Χωρίς-Ζάχαρη', + 'Χωρίς-Αλάτι', + 'Χωρίς-Λακτόζη', + 'Χωρίς-Συντηρητικά', + 'Χωρίς-Αλκοόλ', + 'Χωρίς-Kαφεϊνη', + 'Χωρίς-Γλυκάνισο', + 'Χωρίς-Ανθρακικό', + 'Υψηλής-Παστερίωσης', + 'Ολικής-Άλεσης', + 'Ολικής-Aλέσεως', + 'Χαρτί-Υγείας', + 'ρολό-υγείας', + 'χαρτί-τουαλέτας', + 'Χαρτί-Κουζίνας', + 'Μπάρες-Δημητριακών', + 'Μπαρμπα-Στάθης', + 'COCA-COLA', + 'Aς-Μαγειρέψουμε', + 'ΚΡΙΣ-ΚΡΙΣ', + 'ΚΡΙ-ΚΡΙ', + 'ΕΛ-ΓΚΡΕΚΟ', + 'FREE-STEP', + 'EL-SABOR', + 'LE-PETIT-MARSEILLAIS', + 'DOUWE-EGBERTS', + 'ΕΝ-ΕΛΛΑΔΙ', + 'SPIN-SPAN', + 'CRETA-FARMS', + 'CRETA-FARM', + 'NES-CAFE', + 'Ολες-τις-Χρήσεις', + 'Το-Μάννα', + 'Χωρίς-προσθήκη-ζάχαρης' +] + + +// list of words to exclude from keywords +// ----------------------------------------------------------------------------- +// NOTE: APPLIED in PER-WORD base -> after spliting description to words +removeList = [] +removeOriginals = 'μας με σε για του της των από στο στον & r s ft l τ e g h k m n o p s x'.split(' ') +removeOriginals.forEach( w => { removeList.push( kb_trans(w)); }); + + + + + +/* +// ready to get main data to preccess +//////////////////////////////////////////////////////////////////////////////// + +// db connection parametres +// ----------------------------------------------------------------------------- +var con = mysql.createConnection({ + // host: "/cloudsql/pythia-251711:europe-west4:pythia-db-eu", + socketPath: "/cloudsql/pythia-251711:europe-west4:pythia-db-eu", + user: "pythia_services", + password: process.env.DB_PASSWORD, + database: "pythia_db" +}); + +*/ + + +let url = "https://emarket-laravel-dlqjpfxz5q-oa.a.run.app/api/v1/productsSearch"; + +let settings = { method: "Get" }; + +fetch(url, settings) + .then(res => res.json()) + .then((json) => { + + console.log('data loaded'); + + // do something with JSON + var linkeys = do_proccess(json.data); + + save_local(kwlinks_); +}); + + + +// keyword links (word-links dictionary; array of objects) +// ----------------------------------------------------------------------------- +var kwlinks_ = []; ////////////////////////////// MAIN OUTPUT OF THE SCRIPT + + +// connect; +// get records to proccess; +// call main proccess function; +// save dictionary; +// end script; +// ----------------------------------------------------------------------------- + +exports.main = () => { + + con.connect(function(err) { + // connect; + if (err) throw err; + console.log("Connected!"); + + var sql = "SELECT count(pl.eys_code) as FREQuency,\ + pl.product_id as product_id,\ + pb.brand_name,\ + pl.barcode, pl.skl_code, pl.eys_code,\ + IF( pd.description IS NOT NULL , pd.description , pl.product_description ) AS product_description,\ + pl.bpcs_code,\ + pd.image_path\ + FROM product_list as pl\ + LEFT JOIN delivery_orders_products AS dop ON dop.product_id = pl.eys_code\ + LEFT JOIN product_details AS pd ON pl.eys_code = pd.eys_code\ + LEFT JOIN product_brands pb ON pl.brand_id = pb.id\ + WHERE pl.active = 1 AND pl.sap_code IS NOT NULL AND pl.product_category_sap_4 NOT LIKE '72%'\ + GROUP BY pl.product_id\ + ORDER BY FREQuency DESC"; + + // query sql + con.query(sql, function (err, result) { + if (err) throw err; + console.log('Records from database received!') + + do_proccess(result); // proccess + console.log('Keywords proccesed!') + + save_keywords(); // save + upload_file('pythia-files', os.tmpdir()+'/keywords.json', 'uploads/orders/keywords.json'); + console.log('Results saved! Exiting.') + + // echo memory stats + const used = process.memoryUsage(); + for (let key in used) { + console.log(`${key} ${Math.round(used[key] / 1024 / 1024 * 100) / 100} MB`); + } + }); + }); +} + + + +// Google Cloud Functions +//////////////////////////////////////////////////////////////////////////////// +function upload_file( bucketName, filePath, destFileName ) { + // [START storage_upload_file] + + // Sample code + // const bucketName = 'your-unique-bucket-name'; // The ID of your GCS bucket + // const filePath = 'path/to/your/file'; // The path to your file to upload + // const destFileName = 'your-new-file-name'; // The new ID for your GCS file + + // Imports the Google Cloud client library + const {Storage} = require('@google-cloud/storage'); + + // Creates a client + const storage = new Storage(); + + async function uploadFile() { + await storage.bucket(bucketName).upload(filePath, { + destination: destFileName, + gzip: true, // serve compressed + metadata: { // cache for 8 hours + cacheControl: 'public, max-age=28800', + } + }); + + console.log(`${filePath} uploaded to ${bucketName}`); + } + + uploadFile().catch(console.error); + // [END storage_upload_file] +} + + + + + +// main proccess +// ----------------------------------------------------------------------------- +function do_proccess(obj) { + // console.log(JSON.stringify(obj, null, 2)); + + obj.forEach( rec => { + + if (rec.img != 0) { + + var description = preproccess_text(rec.txt); + var fq = 1; + var pid = rec.id; + + var keys = []; + var words = description.split(' '); + + // filter words; keep only significant + words.forEach( w => { + if (removeList.indexOf(kb_trans(w)) == -1) // if not excluded + if (is_significant(w)) // and significant + keys.push(w); // add it to keys + }); + // console.log(pid, description, keys); + + keys.forEach( w => { + var wl = synonym_keys(w); + + // if key CAN be a root word (not a no-Root-keyword) update root-node + if (noRootKeywords.indexOf(kb_trans(w)) == -1) { + + root_key(wl, fq); // update root keyword stats + + // if key is the only in the list of product's keywords + // connect it with a dummy key (to preserve the reference to the product) + if (keys.length == 1) connect_keys(wl, ['*'], pid, fq); + + // connect w with all the other product's keywords + keys.forEach( w2 => { + if (w2 != w) { + var w2syns = synonym_keys(w2); + connect_keys( wl, w2syns, pid, fq); + } + }); + } + + }); + } + }); // main proccessing finished; + + // post proccess + // ------------------------------------------------------------------------- + + // remove cached keys from final array + kwlinks_.forEach( ro => { + delete ro.kb; + ro.c.forEach( ch => { delete ch.kb; }); + }); + + // sort root and child nodes by frequency descanding + kwlinks_.forEach( it => { + it.c = it.c.sort((a, b) => b.f - a.f ); + }); + kwlinks_ = kwlinks_.sort((a, b) => b.f - a.f ); + +} + + +// save proccess +// ----------------------------------------------------------------------------- + +function save_local(json) { + // stringify JSON Object + var jsonContent = JSON.stringify(json); + + fs.writeFile("../results/keywords-node.json", jsonContent, 'utf8', function (err) { + if (err) { + console.log("An error occured while writing JSON Object to File."); + return console.log(err); + } + + console.log("JSON file has been saved."); + }); +} + + + +function save_keywords() { + let jsonStr = JSON.stringify(kwlinks_); + // console.log(jsonStr); + + fs.writeFileSync(os.tmpdir() + "/keywords.json", jsonStr, 'utf8', (err) => { + if (err) { + console.log("An error occured while writing keywords.json"); + return console.log(err); + } + console.log("linkeys file saved."); + }); +} + + +// functions for linking words in keywords dictionary +//////////////////////////////////////////////////////////////////////////////// + +// set root-keyword: wl (if not exist) +// update frequency: f +// * wl is a list of synonym-words +// ** comparison is based on the *keyboard* format +// --- +function root_key ( wl, f ) { + var wkb = kb_trans(wl[0]) // cache kb format + + // check if exists in root keys already + // NOTE: you only need to check the 1st word of synonyms-list + for (i=0; i< kwlinks_.length ; i++) { + if (kwlinks_[i].kb == wkb) { + kwlinks_[i].f += f; + return true; + } + } + // if not exists, append keyword + kwlinks_.push({ + w : wl, + kb : wkb, + f : f, + c : [] + }); + return true; +} + + +// connect keys: a , b (each one is a list of synonmyms) +// of product with id: i +// with frequency: f +// --- +function connect_keys( a, b, id, f ) { + var kbA = kb_trans(a[0]); + var kbB = kb_trans(b[0]); + var bExists = false; + + if (kbA == kbB) return false; // exclude just-in-case + + for (i=0; i< kwlinks_.length ; i++) { + if (kwlinks_[i].kb == kbA) { // found: a; + // update connection to: b + bExists = false; + for (j=0 ; j < kwlinks_[i].c.length ; j++) { + if (kwlinks_[i].c[j].kb == kbB) { + bExists = true; + // update the connection's data + kwlinks_[i].c[j].f += f; + kwlinks_[i].c[j].p.push(id) + break; + } + } + // if connection not exist, init a new one + if (bExists == false) { + // create connection with: b + kwlinks_[i].c.push({ + w : b, + kb : kbB, + f : f, + p : [ i ] + }); + } + return true; + } + } +} + + +// other supplementary functions +//////////////////////////////////////////////////////////////////////////////// + + +// kb_trans translates string to keyboard-latin keys; +// --- +function kb_trans(str) { + str = str.replace('\'',''); + var out = ''; + for (var i=0 ; i< str.length; i++) out += map.get(str[i]); + return out; +} + +// clean text trims some characters (+.') and internal multiple-spaces +// --- +function clean_text(txt) { + return txt.replace('+',' ').replace('.',' ') + .replace(' ',' ') + .replace(' ',' '); +} + +// check if term is significant +// (if not, the term will be excluded from keywords dicionary) +// --- +function is_significant(str) { + if (str == '') return false; + if (significantExceptios.indexOf(str) !== -1) return true; + return !(/\d/.test(str)); +} + +// check if word: w +// ...has synonyms; return list of synonyms +// --- +function synonym_keys(w) { + w_kb = kb_trans(w); + for (i=0 ; i < synonym_kbs.length ; i++) { + if (synonym_kbs.indexOf(w_kb) !== -1) + return synonyms[i]; + } + return [ w ]; +} + +// edit common mistakes +// with suggested replaces +function do_replaces(str) { + replaces.forEach( it => { str = str.replace(it.src, it.trg); }); + return str; +} + +// preproccess description +// --- +function preproccess_text(str) { + str = do_replaces(str); + str = clean_text(str); + str = mark_linked_words(str); + return str; +} + + +// mark linked words (connect them with a dash) +// return new text after "all-links" are marked +// --- +function mark_linked_words(txt) { + linkedWords.forEach( lw => { txt = mark_link( lw, txt ); }); + return txt +} + +// mark a link (lws) to a text (source) +// conecting them with a dash/minus character +// --- +function mark_link(lws, source) { + var src_kb = kb_trans(source.replace(' ', '-')); + var lws_kb = kb_trans(lws.replace(' ', '-')); + var _left = src_kb.toLowerCase().indexOf(lws_kb.toLowerCase()) + if (_left !== -1 ) { + return source.slice(0, _left) + lws + source.slice(_left + lws.length); + } + else return source; +} + diff --git a/python/products-dict-v5.py b/python/products-dict-v5.py index 8d32586..a0a8965 100644 --- a/python/products-dict-v5.py +++ b/python/products-dict-v5.py @@ -9,6 +9,7 @@ import mysql.connector as mysql # mysql connector import re # regex import json # json +import ijson # json import os.path # ... import sys from urllib.request import urlopen @@ -333,7 +334,7 @@ for it in synonymOriginals : ## get data from endpoint (url) -url = "http://localhost/api/v1/productsSearch" +url = "https://emarket-laravel-dlqjpfxz5q-oa.a.run.app/api/v1/productsSearch" json_url = urlopen(url) received = json.loads(json_url.read()) @@ -341,7 +342,7 @@ received = json.loads(json_url.read()) # print(data) # sys.exit() - +print ('products loaded') # --- Lists to fill keywords_ = [] # all data @@ -389,8 +390,7 @@ for rec in received['data'] : description = rec['txt'] pid = ['id'] - fq = 1 - + fq = 1 # setup product # --- @@ -448,11 +448,11 @@ for it in keywords_ : ## OUTPUT final data to a json-format file # ////////////////////////////////////////////////////////////////////////////// -with open("./keywords-v5.json", "w", encoding="utf-8") as outfile : +with open("./results/keywords-v5.json", "w", encoding="utf-8") as outfile : data = json.dump(keywords_, outfile, sort_keys=False, indent=3, ensure_ascii=False) -with open("./minilist-v5.json", "w", encoding="utf-8") as outfile : +with open("./results/minilist-v5.json", "w", encoding="utf-8") as outfile : data = json.dump(minilist_, outfile, sort_keys=False, indent=3, ensure_ascii=False) -with open("./products-v5.json", "w", encoding="utf-8") as outfile : +with open("./results/products-v5.json", "w", encoding="utf-8") as outfile : data = json.dump(products_, outfile, sort_keys=False, indent=3, ensure_ascii=False)
\ No newline at end of file |
