diff options
| author | Geo Halkiadakis <gchalkiadakis@sklavenitis.co.gr> | 2023-03-17 13:30:46 +0200 |
|---|---|---|
| committer | Geo Halkiadakis <gchalkiadakis@sklavenitis.co.gr> | 2023-03-17 13:30:46 +0200 |
| commit | c50d4c645cd3c04204106c4f9f026e5910afa3d5 (patch) | |
| tree | 6cc6bdb9cb159a96b8335419691a08a941ea0165 /products-dictionary.py | |
| parent | 504732c3d35d003fd5067240b98bc35f03c8cad9 (diff) | |
| download | linkeysearch-c50d4c645cd3c04204106c4f9f026e5910afa3d5.tar.gz linkeysearch-c50d4c645cd3c04204106c4f9f026e5910afa3d5.tar.bz2 linkeysearch-c50d4c645cd3c04204106c4f9f026e5910afa3d5.zip | |
Code tree reorganized; older implemenatations act as a start point
Diffstat (limited to 'products-dictionary.py')
| -rw-r--r-- | products-dictionary.py | 228 |
1 files changed, 0 insertions, 228 deletions
diff --git a/products-dictionary.py b/products-dictionary.py deleted file mode 100644 index 1f35a67..0000000 --- a/products-dictionary.py +++ /dev/null @@ -1,228 +0,0 @@ -## LIBRARIES -# ////////////////////////////////////////////////////////////////////////////// - -import pandas as pd # pandas for excel reading -import re # regex -import json # json -import os.path # ... - - -## LOCAL FUNCTIONS -# ////////////////////////////////////////////////////////////////////////////// - -# do-me-INTeger -# --- -def domeInt(x) : - if isinstance(x, str) : # if string - return int(x.strip()) - if isinstance(x, float) : # if float - return round(x) - return x # otherwise is int already - -# do-me-Float -# --- -def domeFloat(x) : - if isinstance(x, str) : - return float(x.strip()) - else : - return x + 0.00 # make sure that result is float - - -def cleanText(x) : - removeList = [ ' με ', ' σε ', ' για ', ' του ', ' της ', ' των ', ' από ', '&', '.', ',', '!', '(', ')', '[', ']', '\'', '\"' ] - - for r in removeList : - x = x.replace(r, ' ') - - x.replace(' ', ' ') # remove spare spaces - x.replace(' ', ' ') - x.replace(' ', ' ') - - return x - - -def isSignificant(x) : - # is significant if words has no digit-characters - return not bool(re.match("\S*\d+\S*", x)) - - - -# set root-keyQ: w (if not exist) -# update frequency: f -# into list: l -## --- -def rootKey ( w, f, l ) : - keyExists = False - for it in l : - if it['w'] == w : - keyExists = True - it['f'] += f - - if keyExists == False : - l.append({ - 'w' : w, - 'f' : f, - 'c' : [] - }) - - - -# connect keys: a , b -# of product with id: i -# with frequency: f -# into list: l -## --- -def connectKeys( a, b, i, f, l ) : - if a == b : - return False ## exclude just-in-case - - for it in l : - if it['w'] == a : - - # word a found; - # lets update the connection to: b - bExists = False - - for jt in it['c'] : - if jt['w'] == b : - bExists = True - # update the connection's data - jt['f'] += f - jt['p'].append(i) - - if bExists == False : - # create connection with word: b - it['c'].append({ - 'w': b, - 'f': f, - 'p': [ i ] - }) - - - - - -## LOCAL CONSTANTS -# ////////////////////////////////////////////////////////////////////////////// - -_COL = { - # -- main info - 'freq' : 0, # frequency (based on recent orders) - 'pid' : 1, # product id - 'brand' : 2, # brand - 'barcd' : 3, # barcode - 'sklcd' : 4, - 'eyscd' : 5, - 'descr' : 6, # product description - 'sap2' : 7 # SAP category level-2 id -} - - - - -## SET SOURCE and EXPORT FileNames -# ------------------------------------------------------------------------------ -# location of excel file -loc = "./data/PRODucts2search-wBrands.xlsx" - -print("default filename:", loc) -newXLfile = input("input other Excel filename [enter to keep default]: ") - -if newXLfile != "" and os.path.exists(newXLfile): - loc = newXLfile -else : - print(newXLfile, "is not a file; default is kept;") - -## baseEXPORTname = input("Base export name: ") - - - -## Read data -# ////////////////////////////////////////////////////////////////////////////// - -df = pd.read_excel(loc) # read data from excel file - -rows = df.iterrows() # set rows list - - -# --- Lists to fill -keywords_ = [] # all data -minilist_ = [] - -## keywords format: -## [ -## { -## w : 'fresh', -## f : 150, -## c : [ -## { w : 'milk', f : 150 , p : [122, 254, 907] }, -## { w : 'juice', f : 50 , p : [254, 351] } -## ] -## }, -## {...}, -## ... -## ] -## --- index: -## w : word -## f : frequency -## c : combos / connections -## p : list of product-ids with this combo - -# --- temporary variables (initialize) - - -## LOOP through the rows to pre-proccess all products -## --- -for idx, row in rows : - - description = row[_COL['descr']].strip() # product description - pid = domeInt( row[_COL['pid']] ) # product-id - fq = domeInt( row[_COL['freq']] ) # frequency - - # TODO: - # identify brands - # then ... - - description = cleanText(description) # clean sescription string - words = description.strip().upper().split() # split to words - ## words = [w.strip('.,!;()[]') for w in words] # clean strings - - # identify significant words - keys = [] - for w in words : - if isSignificant(w) : - keys.append(w) - - print(pid, keys) - # append words (and their combos) to the list - for w in keys : - rootKey( w, fq, keywords_ ) - for w2 in keys : - if w2 != w : - connectKeys( w, w2, pid, fq, keywords_ ) - - -## SORT keywords -# ////////////////////////////////////////////////////////////////////////////// - -# --- sort childs of each key (per frequency, desc) -for it in keywords_ : - it['c'].sort(key=lambda x: x['f'], reverse=True) - - -# --- sort root keys -keywords_.sort(key=lambda x: x['f'], reverse=True) - -# --- create mini list based on the sorted keywords_ -for it in keywords_ : - minilist_.append( it['w'] ) - - -## OUTPUT final data to a json-format file -# ////////////////////////////////////////////////////////////////////////////// - -with open("results/keywords-v2.json", "w", encoding="utf-8") as outfile : - data = json.dump(keywords_, outfile, sort_keys=False, indent=3, ensure_ascii=False) - -with open("results/minilist.json", "w", encoding="utf-8") as outfile : - data = json.dump(minilist_, outfile, sort_keys=False, indent=3, ensure_ascii=False) |
